From ef365e89a90497250874036e6e7a72dfa994191c Mon Sep 17 00:00:00 2001 From: Nic Cope Date: Wed, 30 Sep 2026 21:44:55 -0700 Subject: [PATCH 1/7] Build the host Python package set in its own file The checks build their virtualenvs from a uv2nix package set that checks.nix defines for itself. The apps that run the tests need virtualenvs from the same set, so this commit moves it to nix/python.nix, and flake.nix passes it to the checks. Signed-off-by: Nic Cope --- flake.nix | 18 ++++++++++-------- nix/checks.nix | 14 +------------- nix/python.nix | 19 +++++++++++++++++++ 3 files changed, 30 insertions(+), 21 deletions(-) create mode 100644 nix/python.nix diff --git a/flake.nix b/flake.nix index 02704e802..637947c89 100644 --- a/flake.nix +++ b/flake.nix @@ -115,14 +115,16 @@ checks = forAllSystems ( { pkgs, ... }: import ./nix/checks.nix { - inherit - pkgs - self - functionNames - pyproject-nix - uv2nix - pyproject-build-systems - ; + inherit pkgs self functionNames; + pythonSet = import ./nix/python.nix { + inherit + pkgs + self + pyproject-nix + uv2nix + pyproject-build-systems + ; + }; } ); diff --git a/nix/checks.nix b/nix/checks.nix index 29993c8af..04955f1b5 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -7,23 +7,11 @@ pkgs, self, functionNames, - pyproject-nix, - uv2nix, - pyproject-build-systems, + pythonSet, }: let docs = import ./docs.nix { inherit pkgs self; }; - workspace = uv2nix.lib.workspace.loadWorkspace { workspaceRoot = self; }; - pythonSet = - (pkgs.callPackage pyproject-nix.build.packages { python = pkgs.python312; }).overrideScope - ( - pkgs.lib.composeManyExtensions [ - pyproject-build-systems.overlays.wheel - (workspace.mkPyprojectOverlay { sourcePreference = "wheel"; }) - ] - ); - # Each function exports a 'function' Python module, so tests must run from # a directory where that module is importable via the venv. We copy tests/ # from the source tree and run unittest against the venv's Python. diff --git a/nix/python.nix b/nix/python.nix new file mode 100644 index 000000000..740b9e7d3 --- /dev/null +++ b/nix/python.nix @@ -0,0 +1,19 @@ +# The uv workspace's Python packages, built by uv2nix from uv.lock, for +# virtualenvs that run on the build host. The function images build their own, +# per target architecture (see functions.nix). +{ + pkgs, + self, + pyproject-nix, + uv2nix, + pyproject-build-systems, +}: +let + workspace = uv2nix.lib.workspace.loadWorkspace { workspaceRoot = self; }; +in +(pkgs.callPackage pyproject-nix.build.packages { python = pkgs.python312; }).overrideScope ( + pkgs.lib.composeManyExtensions [ + pyproject-build-systems.overlays.wheel + (workspace.mkPyprojectOverlay { sourcePreference = "wheel"; }) + ] +) From 5617126449cf3b4b744a544448d1120a723489c0 Mon Sep 17 00:00:00 2001 From: Nic Cope Date: Wed, 30 Sep 2026 16:55:54 -0700 Subject: [PATCH 2/7] Run the function unit tests with pytest The e2e tests are moving to pytest, and one test runner for the whole repo is simpler than two. pytest 9 runs the existing unittest suites unchanged and reports each subTest on its own, so this commit switches the runner without touching a test. It configures pytest in strict mode, to list each test, and to print whole assertion diffs, since the tests compare whole responses. nix run .#test runs every function's unit tests outside the sandbox, against the virtualenvs the checks use, or one function's with pytest arguments after its name. Towards #473. Signed-off-by: Nic Cope --- CONTRIBUTING.md | 6 +++-- flake.nix | 10 ++++++++ nix/apps.nix | 62 +++++++++++++++++++++++++++++++++++++++++++++++++ nix/checks.nix | 10 +++++--- pyproject.toml | 12 ++++++++++ uv.lock | 54 ++++++++++++++++++++++++++++++++++++++++++ 6 files changed, 149 insertions(+), 5 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 0c27ee61e..87e4989da 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -331,8 +331,10 @@ the function emits. Some existing tests (`compose-serving-stack`, the second method in `compose-eks-cluster`) predate this form and assert on individual fields. Don't -model new tests on them. Add new cases to the function's `test_fn.py` and run -`nix flake check` to verify they pass. +model new tests on them. Add new cases to the function's `test_fn.py`, and run +them with `nix run .#test`. Name a function to run only its tests, and pass +pytest arguments after it, as in `nix run .#test -- compose-usages -k +namespace`. `nix flake check` runs them too. ### Running locally diff --git a/flake.nix b/flake.nix index 637947c89..5af039a62 100644 --- a/flake.nix +++ b/flake.nix @@ -160,6 +160,15 @@ apps = import ./nix/apps.nix { inherit pkgs; }; crossplane = deps.crossplane { inherit system; }; functionsPkg = self.packages.${system}.functions or null; + pythonSet = import ./nix/python.nix { + inherit + pkgs + self + pyproject-nix + uv2nix + pyproject-build-systems + ; + }; in { fix = apps.fix { }; @@ -177,6 +186,7 @@ }; stop = apps.stop { inherit crossplane; }; e2e = apps.e2e { inherit crossplane functionsPkg; }; + test = apps.test { inherit pythonSet functionNames; }; stacks = apps.stacks { inherit (pkgs) aicr; }; } ); diff --git a/nix/apps.nix b/nix/apps.nix index 48c62ff1f..8dc475f4c 100644 --- a/nix/apps.nix +++ b/nix/apps.nix @@ -296,6 +296,68 @@ ); }; + # Run the composition functions' unit tests outside the sandbox, against the + # same virtualenvs nix flake check uses. With no function named it runs every + # function's tests, each in a pytest session of its own because every + # function's package is named `function`. Arguments after the function name + # go to pytest, e.g. nix run .#test -- compose-usages -k namespace. + test = + { + pythonSet, + functionNames, + }: + let + venvs = map (name: { + inherit name; + venv = pythonSet.mkVirtualEnv "${name}-test-env" { + ${name} = [ ]; + pytest = [ ]; + }; + }) functionNames; + cases = pkgs.lib.concatMapStrings (v: '' + ${v.name}) python=${v.venv}/bin/python ;; + '') venvs; + in + { + type = "app"; + meta.description = "Run the composition functions' unit tests"; + program = pkgs.lib.getExe ( + pkgs.writeShellApplication { + name = "modelplane-test"; + runtimeInputs = [ pkgs.coreutils ]; + inheritPath = false; + text = '' + run() { + local fn="$1" python + shift + case "$fn" in + ${cases} + *) + echo "no such function: $fn" >&2 + return 2 + ;; + esac + "$python" -m pytest "functions/$fn/tests" "$@" + } + + if [ $# -gt 0 ] && [[ "$1" != -* ]]; then + run "$@" + exit + fi + + failed=() + for fn in ${pkgs.lib.concatStringsSep " " functionNames}; do + run "$fn" "$@" || failed+=("$fn") + done + if [ ''${#failed[@]} -gt 0 ]; then + echo "failed: ''${failed[*]}" >&2 + exit 1 + fi + ''; + } + ); + }; + # Run the two-cluster local end-to-end test: a workload # kind cluster registered via source: Existing (serving stack + model) and a # control-plane cluster (crossplane + the InferenceGateway). Two clusters diff --git a/nix/checks.nix b/nix/checks.nix index 04955f1b5..8aa54827a 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -13,18 +13,22 @@ let docs = import ./docs.nix { inherit pkgs self; }; # Each function exports a 'function' Python module, so tests must run from - # a directory where that module is importable via the venv. We copy tests/ - # from the source tree and run unittest against the venv's Python. + # a directory where that module is importable via the venv, and one pytest + # session can't hold two functions' tests. We copy tests/ from the source + # tree and run pytest against the venv's Python. We also copy pyproject.toml + # for its [tool.pytest] config, which pytest finds in its rootdir. mkFunctionTest = name: let venv = pythonSet.mkVirtualEnv "${name}-test-env" { ${name} = [ ]; + pytest = [ ]; }; in pkgs.runCommand "modelplane-test-${name}" { } '' cp -r ${self}/functions/${name}/tests tests - ${venv}/bin/python -m unittest discover -s tests -v + cp ${self}/pyproject.toml pyproject.toml + ${venv}/bin/python -m pytest tests mkdir -p $out touch $out/.tests-passed ''; diff --git a/pyproject.toml b/pyproject.toml index d808d3ae1..7e1e0eeba 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -26,6 +26,8 @@ dev = [ # schemas/ tree, so an edit there survives until the next build. "pyyaml>=6.0", "pydantic>=2.0", + # 9.0 reports unittest subtests natively. + "pytest>=9.0", ] [tool.ruff] @@ -94,3 +96,13 @@ python-version = "3.12" [tool.ty.src] # The generated Pydantic models aren't under our control and aren't checked. exclude = ["schemas/python"] + +[tool.pytest] +# Fail on unknown config, unregistered markers, duplicate parametrize IDs and +# passing xfails. +strict = true +# Tests compare whole responses, and by default pytest cuts the diff between two +# large values down to a summary line. +verbosity_assertions = "2" +# List each test by name as it runs. +addopts = ["-v"] diff --git a/uv.lock b/uv.lock index 9fcb8dba1..15f980c67 100644 --- a/uv.lock +++ b/uv.lock @@ -558,6 +558,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/82/54/acc6a6e684827b0f6bb4e2c27f3d7e25b71322c4078ef5b455c07c43260e/grpcio_reflection-1.62.3-py3-none-any.whl", hash = "sha256:a48ef37df81a3bada78261fc92ef382f061112f989d1312398b945cc69838b9c", size = 22232, upload-time = "2024-08-06T00:30:13.131Z" }, ] +[[package]] +name = "iniconfig" +version = "2.3.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/72/34/14ca021ce8e5dfedc35312d08ba8bf51fdd999c576889fc2c24cb97f4f10/iniconfig-2.3.0.tar.gz", hash = "sha256:c76315c77db068650d49c5b56314774a7804df16fee4402c1f19d6d15d8c4730", size = 20503, upload-time = "2025-10-18T21:55:43.219Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/cb/b1/3846dd7f199d53cb17f49cba7e651e9ce294d8497c8c150530ed11865bb8/iniconfig-2.3.0-py3-none-any.whl", hash = "sha256:f631c04d2c48c52b84d0d0549c99ff3859c98df65b3101406327ecc7d53fbf12", size = 7484, upload-time = "2025-10-18T21:55:41.639Z" }, +] + [[package]] name = "jmespath" version = "1.1.0" @@ -585,6 +594,7 @@ source = { virtual = "." } dev = [ { name = "crossplane-function-sdk-python" }, { name = "pydantic" }, + { name = "pytest" }, { name = "pyyaml" }, { name = "types-protobuf" }, ] @@ -595,10 +605,20 @@ dev = [ dev = [ { name = "crossplane-function-sdk-python", specifier = ">=0.14.0" }, { name = "pydantic", specifier = ">=2.0" }, + { name = "pytest", specifier = ">=9.0" }, { name = "pyyaml", specifier = ">=6.0" }, { name = "types-protobuf", specifier = ">=4.24" }, ] +[[package]] +name = "packaging" +version = "26.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/7d/fa/3944b40b07da9ce895c0e6303a5ab7d53da063554f534556b134a54d6093/packaging-26.3.tar.gz", hash = "sha256:94edc256424af38762eb31306eed28beb9f0efc50a8837492c9d6fd6004aed79", size = 313412, upload-time = "2026-08-04T18:15:28.737Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/63/34/ba1c580383c9eada3711951fef0795c80b829a078d72188184bcab9dd527/packaging-26.3-py3-none-any.whl", hash = "sha256:d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c", size = 129956, upload-time = "2026-08-04T18:15:27.159Z" }, +] + [[package]] name = "pendulum" version = "3.2.0" @@ -649,6 +669,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/02/fb/d65db067a67df7252f18b0cb7420dda84078b9e8bfb375215469c14a50be/pendulum-3.2.0-py3-none-any.whl", hash = "sha256:f3a9c18a89b4d9ef39c5fa6a78722aaff8d5be2597c129a3b16b9f40a561acf3", size = 114111, upload-time = "2026-01-30T11:22:22.361Z" }, ] +[[package]] +name = "pluggy" +version = "1.6.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/f9/e2/3e91f31a7d2b083fe6ef3fa267035b518369d9511ffab804f839851d2779/pluggy-1.6.0.tar.gz", hash = "sha256:7dcc130b76258d33b90f61b658791dede3486c3e6bfb003ee5c9bfb396dd22f3", size = 69412, upload-time = "2025-05-15T12:30:07.975Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/54/20/4d324d65cc6d9205fabedc306948156824eb9f0ee1633355a8f7ec5c66bf/pluggy-1.6.0-py3-none-any.whl", hash = "sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746", size = 20538, upload-time = "2025-05-15T12:30:06.134Z" }, +] + [[package]] name = "protobuf" version = "7.35.0" @@ -751,6 +780,31 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/4b/2d/69abac8f838090bbecd5df894befb2c2619e7996a98ddb949db9f3b93225/pydantic_core-2.46.4-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:d51026d73fcfd93610abc7b27789c26b313920fcfb20e27462d74a7f8b06e983", size = 2193071, upload-time = "2026-05-06T13:38:08.682Z" }, ] +[[package]] +name = "pygments" +version = "2.21.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/49/2e/ced460408999b33da6b31b0021b0f37d329e202d4169aeb164493778f25b/pygments-2.21.0.tar.gz", hash = "sha256:610ca751c9bc2492b38eb9a38a7fbc93edbbb2d7182edaf34e66ae493dee5c8c", size = 5005329, upload-time = "2026-08-17T08:02:48.824Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/71/46/17f022dd3e953bf20a04a028a21ec746d942f8d2af30fa0f124fa0e6a684/pygments-2.21.0-py3-none-any.whl", hash = "sha256:2363c69b61c4a97c838da3b130dcd6468f4848992b21a82f2a63ec34377137d9", size = 1250147, upload-time = "2026-08-17T08:02:44.912Z" }, +] + +[[package]] +name = "pytest" +version = "9.1.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "colorama", marker = "sys_platform == 'win32'" }, + { name = "iniconfig" }, + { name = "packaging" }, + { name = "pluggy" }, + { name = "pygments" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/e4/47/b9efed96c114afcfa3c9d3fe98a76a1d14c74a9e266d397cf6eb64be5e01/pytest-9.1.1.tar.gz", hash = "sha256:1088fbde8f2b49d95a549a195707afa7a76a3ce9bcadc26b6d71f0ffda5fe313", size = 1636369, upload-time = "2026-06-19T10:58:32.857Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/24/25/1de2678b631f5a49215c6c96fff41ba892b0a34df68d6d80292b1b48aa7f/pytest-9.1.1-py3-none-any.whl", hash = "sha256:37a86b45efb9a47a61a36449063e8e18d0cab3161329fc099eb21783169c4f0c", size = 386536, upload-time = "2026-06-19T10:58:31.347Z" }, +] + [[package]] name = "python-dateutil" version = "2.9.0.post0" From 0402811d2fe694f23e16f1a03037d40ddf0498bf Mon Sep 17 00:00:00 2001 From: Nic Cope Date: Mon, 5 Oct 2026 16:14:26 -0700 Subject: [PATCH 3/7] Write the function unit tests in pytest's style The previous commit runs the unittest suites under pytest unchanged. This commit ports them to pytest's own style, so each case in a table is a test of its own that -k can select. Async tests call RunFunction with asyncio.run rather than needing a plugin. pytest captures a test's output and shows it only when the test fails, so the tests no longer disable the functions' logging. pytest diffs two dicts in insertion order, where unittest sorted their keys. The Structs a function returns rarely list their keys in the same order as the test's expected ones, so the tests compare dicts with sorted keys, which keeps a diff to the fields that differ. No case's request or expected response changes. Towards #473. Signed-off-by: Nic Cope --- CONTRIBUTING.md | 104 +- .../compose-aks-cluster/tests/test_fn.py | 540 +- .../compose-civo-cluster/tests/test_fn.py | 568 +- .../compose-eks-cluster/tests/__init__.py | 14 - .../compose-eks-cluster/tests/test_fn.py | 898 ++- .../compose-gke-cluster/tests/__init__.py | 15 - .../compose-gke-cluster/tests/test_fn.py | 637 +- .../compose-inference-class/tests/__init__.py | 14 - .../compose-inference-class/tests/test_fn.py | 118 +- .../tests/__init__.py | 15 - .../tests/test_fn.py | 5864 ++++++++--------- .../tests/__init__.py | 15 - .../tests/test_fn.py | 1932 +++--- .../compose-metric-mapping/tests/__init__.py | 13 - .../compose-metric-mapping/tests/test_fn.py | 179 +- .../compose-model-cache/tests/test_fn.py | 1110 ++-- .../tests/__init__.py | 14 - .../tests/test_cel.py | 461 +- .../compose-model-deployment/tests/test_fn.py | 2391 ++++--- .../tests/test_quantity.py | 329 +- .../tests/test_scheduling.py | 2602 ++++---- .../tests/test_semver.py | 155 +- .../compose-model-endpoint/tests/__init__.py | 14 - .../compose-model-endpoint/tests/test_fn.py | 260 +- .../compose-model-replica/tests/__init__.py | 14 - .../tests/test_backends.py | 2090 +++--- .../compose-model-replica/tests/test_fn.py | 974 ++- .../compose-model-route/tests/__init__.py | 14 - .../compose-model-route/tests/test_fn.py | 895 ++- .../compose-model-service/tests/__init__.py | 14 - .../compose-model-service/tests/test_fn.py | 501 +- .../compose-nebius-cluster/tests/test_fn.py | 721 +- .../compose-serving-stack/tests/__init__.py | 15 - .../tests/test_collector.py | 854 +-- .../compose-serving-stack/tests/test_fn.py | 1237 ++-- .../tests/test_stacks.py | 352 +- .../tests/__init__.py | 13 - .../tests/test_fn.py | 405 +- functions/compose-usages/tests/__init__.py | 14 - functions/compose-usages/tests/test_fn.py | 241 +- .../compose-vultr-cluster/tests/test_fn.py | 516 +- nix/checks.nix | 6 +- pyproject.toml | 3 + 43 files changed, 13337 insertions(+), 13804 deletions(-) delete mode 100644 functions/compose-eks-cluster/tests/__init__.py delete mode 100644 functions/compose-gke-cluster/tests/__init__.py delete mode 100644 functions/compose-inference-class/tests/__init__.py delete mode 100644 functions/compose-inference-cluster/tests/__init__.py delete mode 100644 functions/compose-inference-gateway/tests/__init__.py delete mode 100644 functions/compose-metric-mapping/tests/__init__.py delete mode 100644 functions/compose-model-deployment/tests/__init__.py delete mode 100644 functions/compose-model-endpoint/tests/__init__.py delete mode 100644 functions/compose-model-replica/tests/__init__.py delete mode 100644 functions/compose-model-route/tests/__init__.py delete mode 100644 functions/compose-model-service/tests/__init__.py delete mode 100644 functions/compose-serving-stack/tests/__init__.py delete mode 100644 functions/compose-telemetry-destination/tests/__init__.py delete mode 100644 functions/compose-usages/tests/__init__.py diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 87e4989da..74339614d 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -245,7 +245,7 @@ functions// main.py # CLI entrypoint (boilerplate) fn.py # FunctionRunner gRPC service and Composer logic tests/ - test_fn.py # unittest-based tests for fn.py + test_fn.py # pytest tests for fn.py ``` The `Composer.compose()` method in `fn.py` reads the XR from the request, @@ -274,12 +274,15 @@ XRDs or dependencies you've removed don't linger. ### Tests -Every function has tests under `functions//tests/test_fn.py`. The -canonical form is a table of `Case`s, each running the function on a +Every function has tests under `functions//tests/`, run with +[pytest](https://docs.pytest.org/). `test_fn.py` tests the function as a whole. +A module with logic of its own, such as `compose-model-deployment`'s scheduler, +can have its own `test_.py` too. + +The canonical form is a table of `Case`s, each running the function on a `RunFunctionRequest` and comparing the whole `RunFunctionResponse` against an -expected one — not asserting on individual fields. `compose-usages` is a clean -example; `compose-model-cache` shows the same form scaled up to a multi-pass -reconcile. The skeleton: +expected one, rather than asserting on individual fields. `compose-usages` is a +small example. The skeleton: ```python @dataclasses.dataclass @@ -289,52 +292,59 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - cases = [ - Case( - name="describes what this case exercises", - req=fnv1.RunFunctionRequest(...), - want=fnv1.RunFunctionResponse(...), - ), - ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) +COMPOSE_CASES = [ + Case( + name="describes what this case exercises", + req=fnv1.RunFunctionRequest(...), + want=fnv1.RunFunctionResponse(...), + ), +] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes the resources an XR needs.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) ``` -Build the XR with -`resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json"))` from a -generated Pydantic model; build other observed, desired, and required resources -as plain dicts. Because `want` is the whole response, it must include the parts -the function always emits: `meta.ttl` (60s), an empty `context`, and any -conditions, results, and requirements. Give observed conditions a fixed -`lastTransitionTime` so the input is deterministic. Protobuf maps -(`desired.resources`, `requirements.resources`) compare order-independently, but -repeated fields (`conditions`, `results`, status arrays) must match the order -the function emits. +Name a table for the test that runs it, and put it just above that test. Each +case becomes its own test, named for the case, so `pytest -k` can select it. +With `got` on the left, pytest's diff shows the expected lines as `-` and the +actual lines as `+`, the same way round as Go's `cmp.Diff(want, got)`. Tests are +plain functions, with no classes, fixtures, or `conftest.py`. They call the +async `RunFunction` with `asyncio.run` rather than needing a plugin, and check +errors with `pytest.raises(..., match=...)`. + +Build the XR with `resource.dict_to_struct(xr.model_dump(exclude_none=True, +mode="json"))` from a generated Pydantic model; build other observed, desired, +and required resources as plain dicts. Because `want` is the whole response, it +must include the parts the function always emits: `meta.ttl` (60s), an empty +`context`, and any conditions, results, and requirements. Give observed +conditions a fixed `lastTransitionTime` so the input is deterministic. Protobuf +maps (`desired.resources`, `requirements.resources`) compare +order-independently, but repeated fields (`conditions`, `results`, status +arrays) must match the order the function emits. Some existing tests (`compose-serving-stack`, the second method in `compose-eks-cluster`) predate this form and assert on individual fields. Don't -model new tests on them. Add new cases to the function's `test_fn.py`, and run -them with `nix run .#test`. Name a function to run only its tests, and pass -pytest arguments after it, as in `nix run .#test -- compose-usages -k -namespace`. `nix flake check` runs them too. +model new tests on them. + +`nix flake check` runs every function's tests, and so does `nix run .#test`, +outside the sandbox. Name a function to run only its tests, and pass pytest +arguments after it: + +```bash +nix run .#test -- compose-usages -k namespace +``` + +Each function runs in a pytest session of its own, because every function names +its package `function`. ### Running locally diff --git a/functions/compose-aks-cluster/tests/test_fn.py b/functions/compose-aks-cluster/tests/test_fn.py index d3ea4c9fa..e5a28a466 100644 --- a/functions/compose-aks-cluster/tests/test_fn.py +++ b/functions/compose-aks-cluster/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-aks-cluster function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.akscluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -36,10 +38,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - # Names derived like the function derives them - the hash suffix depends only # on the input names. _CLUSTER_NAME = resource.child_name("modelplane-system", "test-cluster", "aks") @@ -344,285 +342,277 @@ def _observed_ready(desired: dict) -> fnv1.Resource: ) -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function composes AKS cluster infrastructure.""" - cases = [ - Case( - name="first pass composes infra; gated resources wait for the cluster", - req=_req([_GPU_POOL]), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - # The StorageClass isn't composed yet: the cluster - # isn't observed, so the ProviderConfigs can't - # reach it. - "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, +COMPOSE_CASES = [ + Case( + name="first pass composes infra; gated resources wait for the cluster", + req=_req([_GPU_POOL]), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), + "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), + # The StorageClass isn't composed yet: the cluster + # isn't observed, so the ProviderConfigs can't + # reach it. + "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, ), - context=structpb.Struct(), - ), - ), - Case( - name="zones pass through to the node pool", - req=_req( - [ - v1alpha1.NodePool( - name="gpuh100", - role="GPU", - vmSize="Standard_ND96isr_H100_v5", - diskSizeGb=200, - nodeCount=1, - minNodeCount=1, - maxNodeCount=4, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), - zones=[v1alpha1.Zone("1")], + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), ), - ] - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "nodepool-gpuh100": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu(zones=["1"])), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, + ready=fnv1.READY_TRUE, ), - context=structpb.Struct(), - ), + }, ), - Case( - name="InfiniBand pool composes the network operator once the cluster is observed", - req=_req( - [_GPU_POOL_INFINIBAND], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, + context=structpb.Struct(), + ), + ), + Case( + name="zones pass through to the node pool", + req=_req( + [ + v1alpha1.NodePool( + name="gpuh100", + role="GPU", + vmSize="Standard_ND96isr_H100_v5", + diskSizeGb=200, + nodeCount=1, + minNodeCount=1, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + zones=[v1alpha1.Zone("1")], ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), - "release-network-operator": fnv1.Resource( - resource=resource.dict_to_struct(_network_operator_release()), - ), - "storage-class-rwx-fs": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class()), - ready=fnv1.READY_TRUE, - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, + ] + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), + "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), + "nodepool-gpuh100": fnv1.Resource( + resource=resource.dict_to_struct(_nodepool_gpu(zones=["1"])), ), - context=structpb.Struct(), - ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + }, ), - Case( - name="InfiniBand pool before the cluster is observed gates the network operator", - req=_req([_GPU_POOL_INFINIBAND]), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, + context=structpb.Struct(), + ), + ), + Case( + name="InfiniBand pool composes the network operator once the cluster is observed", + req=_req( + [_GPU_POOL_INFINIBAND], + observed_resources={ + "cluster": _observed_ready(_cluster()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), + "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, ), - context=structpb.Struct(), - ), + "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), + "release-network-operator": fnv1.Resource( + resource=resource.dict_to_struct(_network_operator_release()), + ), + "storage-class-rwx-fs": fnv1.Resource( + resource=resource.dict_to_struct(_storage_class()), + ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + }, ), - Case( - name="custom credentials flow through to all cloud MRs", - req=_req( - [_GPU_POOL], - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="my-azure-account", + context=structpb.Struct(), + ), + ), + Case( + name="InfiniBand pool before the cluster is observed gates the network operator", + req=_req([_GPU_POOL_INFINIBAND]), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), + "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), + "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "resource-group": fnv1.Resource( - resource=resource.dict_to_struct(_resource_group("ProviderConfig", "my-azure-account")), - ), - "virtual-network": fnv1.Resource( - resource=resource.dict_to_struct( - _virtual_network("ProviderConfig", "my-azure-account") - ), - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet("ProviderConfig", "my-azure-account")), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster("ProviderConfig", "my-azure-account")), - ), - "nodepool-gpuh100": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu("ProviderConfig", "my-azure-account")), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, ), - context=structpb.Struct(), - ), + }, ), - Case( - name="marks managed resources ready from observed conditions", - req=_req( - [_GPU_POOL], - observed_resources={ - "resource-group": _observed_ready(_resource_group()), - "virtual-network": _observed_ready(_virtual_network()), - "subnet": _observed_ready(_subnet()), - "cluster": _observed_ready(_cluster()), - "nodepool-gpuh100": _observed_ready(_nodepool_gpu()), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "resource-group": fnv1.Resource( - resource=resource.dict_to_struct(_resource_group()), - ready=fnv1.READY_TRUE, - ), - "virtual-network": fnv1.Resource( - resource=resource.dict_to_struct(_virtual_network()), - ready=fnv1.READY_TRUE, - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ready=fnv1.READY_TRUE, - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - # The cluster is observed, so the StorageClass is - # composed too. - "storage-class-rwx-fs": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class()), - ready=fnv1.READY_TRUE, - ), - "nodepool-gpuh100": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu()), - ready=fnv1.READY_TRUE, - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, + context=structpb.Struct(), + ), + ), + Case( + name="custom credentials flow through to all cloud MRs", + req=_req( + [_GPU_POOL], + credentials=v1alpha1.Credentials( + type="ProviderConfig", + name="my-azure-account", + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "resource-group": fnv1.Resource( + resource=resource.dict_to_struct(_resource_group("ProviderConfig", "my-azure-account")), ), - context=structpb.Struct(), - ), + "virtual-network": fnv1.Resource( + resource=resource.dict_to_struct(_virtual_network("ProviderConfig", "my-azure-account")), + ), + "subnet": fnv1.Resource( + resource=resource.dict_to_struct(_subnet("ProviderConfig", "my-azure-account")), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster("ProviderConfig", "my-azure-account")), + ), + "nodepool-gpuh100": fnv1.Resource( + resource=resource.dict_to_struct(_nodepool_gpu("ProviderConfig", "my-azure-account")), + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + ), + ), + Case( + name="marks managed resources ready from observed conditions", + req=_req( + [_GPU_POOL], + observed_resources={ + "resource-group": _observed_ready(_resource_group()), + "virtual-network": _observed_ready(_virtual_network()), + "subnet": _observed_ready(_subnet()), + "cluster": _observed_ready(_cluster()), + "nodepool-gpuh100": _observed_ready(_nodepool_gpu()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "resource-group": fnv1.Resource( + resource=resource.dict_to_struct(_resource_group()), + ready=fnv1.READY_TRUE, + ), + "virtual-network": fnv1.Resource( + resource=resource.dict_to_struct(_virtual_network()), + ready=fnv1.READY_TRUE, + ), + "subnet": fnv1.Resource( + resource=resource.dict_to_struct(_subnet()), + ready=fnv1.READY_TRUE, + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, + ), + # The cluster is observed, so the StorageClass is + # composed too. + "storage-class-rwx-fs": fnv1.Resource( + resource=resource.dict_to_struct(_storage_class()), + ready=fnv1.READY_TRUE, + ), + "nodepool-gpuh100": fnv1.Resource( + resource=resource.dict_to_struct(_nodepool_gpu()), + ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + }, ), - ] - - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) + context=structpb.Struct(), + ), + ), +] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """The function composes AKS cluster infrastructure.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-civo-cluster/tests/test_fn.py b/functions/compose-civo-cluster/tests/test_fn.py index 938a23024..a6bed72dd 100644 --- a/functions/compose-civo-cluster/tests/test_fn.py +++ b/functions/compose-civo-cluster/tests/test_fn.py @@ -14,15 +14,17 @@ """Tests for the compose-civo-cluster function.""" +import asyncio import dataclasses -import unittest +import json from typing import Any -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.civocluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -37,10 +39,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - # Name of the cluster's connection secret. Derived like the function derives # it - the hash suffix depends only on the parent and child names. _KUBECONFIG_SECRET_NAME = resource.child_name("test-cluster", "kubeconfig") @@ -309,306 +307,298 @@ def _observed_unready(desired: dict, external_name: str | None = None) -> fnv1.R ) -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function composes Civo cluster infrastructure.""" - cases = [ - Case( - name="network, firewall and cluster composed first; node pools withheld until cluster Ready", - req=_req([_GPU_POOL]), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ), - "firewall": fnv1.Resource( - resource=resource.dict_to_struct(_firewall()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - }, +COMPOSE_CASES = [ + Case( + name="network, firewall and cluster composed first; node pools withheld until cluster Ready", + req=_req([_GPU_POOL]), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct(_network()), ), - context=structpb.Struct(), - ), + "firewall": fnv1.Resource( + resource=resource.dict_to_struct(_firewall()), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ), + }, ), - Case( - name="node pools and provider config composed once cluster is Ready; autoscaler has no groups until pool IDs observed", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_ready(_cluster(), external_name=_CLUSTER_ID), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ), - "firewall": fnv1.Resource( - resource=resource.dict_to_struct(_firewall()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - "release-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler([])), - ), - }, + context=structpb.Struct(), + ), + ), + Case( + name="node pools and provider config composed once cluster is Ready; autoscaler has no groups until pool IDs observed", + req=_req( + [_GPU_POOL], + observed_resources={ + "cluster": _observed_ready(_cluster(), external_name=_CLUSTER_ID), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct(_network()), ), - context=structpb.Struct(), - ), + "firewall": fnv1.Resource( + resource=resource.dict_to_struct(_firewall()), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, + ), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config_helm()), + ready=fnv1.READY_TRUE, + ), + "release-cluster-autoscaler": fnv1.Resource( + resource=resource.dict_to_struct(_autoscaler([])), + ), + }, ), - Case( - name="autoscaler release composed from observed pool IDs", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_ready(_cluster(), external_name=_CLUSTER_ID), - "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN, external_name=_GPU_POOL_ID), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ), - "firewall": fnv1.Resource( - resource=resource.dict_to_struct(_firewall()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - "release-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct( - _autoscaler( - [{"name": _GPU_POOL_ID, "minSize": 1, "maxSize": 4}], - ), - ), + context=structpb.Struct(), + ), + ), + Case( + name="autoscaler release composed from observed pool IDs", + req=_req( + [_GPU_POOL], + observed_resources={ + "cluster": _observed_ready(_cluster(), external_name=_CLUSTER_ID), + "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN, external_name=_GPU_POOL_ID), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct(_network()), + ), + "firewall": fnv1.Resource( + resource=resource.dict_to_struct(_firewall()), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, + ), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config_helm()), + ready=fnv1.READY_TRUE, + ), + "release-cluster-autoscaler": fnv1.Resource( + resource=resource.dict_to_struct( + _autoscaler( + [{"name": _GPU_POOL_ID, "minSize": 1, "maxSize": 4}], ), - }, + ), ), - context=structpb.Struct(), - ), + }, ), - Case( - name="dependents kept when the cluster Ready condition transiently regresses", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_unready(_cluster(), external_name=_CLUSTER_ID), - "provider-config-helm": _observed_ready(_provider_config_helm()), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ), - "firewall": fnv1.Resource( - resource=resource.dict_to_struct(_firewall()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - "release-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler([])), - ), - }, + context=structpb.Struct(), + ), + ), + Case( + name="dependents kept when the cluster Ready condition transiently regresses", + req=_req( + [_GPU_POOL], + observed_resources={ + "cluster": _observed_unready(_cluster(), external_name=_CLUSTER_ID), + "provider-config-helm": _observed_ready(_provider_config_helm()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct(_network()), ), - context=structpb.Struct(), - ), + "firewall": fnv1.Resource( + resource=resource.dict_to_struct(_firewall()), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config_helm()), + ready=fnv1.READY_TRUE, + ), + "release-cluster-autoscaler": fnv1.Resource( + resource=resource.dict_to_struct(_autoscaler([])), + ), + }, ), - Case( - name="fixed-size GPU pool composes the autoscaler release with no node groups", - req=_req( - [ - v1alpha1.NodePool( - name="gpu-l40s", - role="GPU", - size="an.g1.l40s.kube.x1", - nodeCount=2, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), - ), - ], - observed_resources={ - "cluster": _observed_ready(_cluster(), external_name=_CLUSTER_ID), - }, + context=structpb.Struct(), + ), + ), + Case( + name="fixed-size GPU pool composes the autoscaler release with no node groups", + req=_req( + [ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + size="an.g1.l40s.kube.x1", + nodeCount=2, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ), - "firewall": fnv1.Resource( - resource=resource.dict_to_struct(_firewall()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct( - _node_pool( - label="gpu-l40s", - size="an.g1.l40s.kube.x1", - node_count=2, - labels={ - "modelplane.ai/pool": "gpu-l40s", - "modelplane.ai/gpu": "nvidia-l40s", - }, - taint=_GPU_TAINT, - ), - ), - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - "release-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler([])), - ), - }, + ], + observed_resources={ + "cluster": _observed_ready(_cluster(), external_name=_CLUSTER_ID), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct(_network()), ), - context=structpb.Struct(), - ), - ), - Case( - name="System pool carries no taint; credentials override propagates", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.CivoCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - region="LON1", - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="team-a", - ), - nodePools=[ - v1alpha1.NodePool( - name="workers", - role="System", - size="g4p.kube.small", - nodeCount=2, - ), - ], - ), - ).model_dump(exclude_none=True, mode="json"), + "firewall": fnv1.Resource( + resource=resource.dict_to_struct(_firewall()), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, + ), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct( + _node_pool( + label="gpu-l40s", + size="an.g1.l40s.kube.x1", + node_count=2, + labels={ + "modelplane.ai/pool": "gpu-l40s", + "modelplane.ai/gpu": "nvidia-l40s", + }, + taint=_GPU_TAINT, ), ), - resources={ - "cluster": _observed_ready( - _cluster(cred_kind="ProviderConfig", cred_name="team-a"), - external_name=_CLUSTER_ID, - ), - }, ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct( - _network(cred_kind="ProviderConfig", cred_name="team-a"), - ), - ), - "firewall": fnv1.Resource( - resource=resource.dict_to_struct( - _firewall(cred_kind="ProviderConfig", cred_name="team-a"), - ), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct( - _cluster(cred_kind="ProviderConfig", cred_name="team-a"), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config_helm()), + ready=fnv1.READY_TRUE, + ), + "release-cluster-autoscaler": fnv1.Resource( + resource=resource.dict_to_struct(_autoscaler([])), + ), + }, + ), + context=structpb.Struct(), + ), + ), + Case( + name="System pool carries no taint; credentials override propagates", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.CivoCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + region="LON1", + credentials=v1alpha1.Credentials( + type="ProviderConfig", + name="team-a", ), - ready=fnv1.READY_TRUE, - ), - "node-pool-workers": fnv1.Resource( - resource=resource.dict_to_struct( - _node_pool( - label="workers", + nodePools=[ + v1alpha1.NodePool( + name="workers", + role="System", size="g4p.kube.small", - node_count=2, - labels={"modelplane.ai/pool": "workers"}, - cred_kind="ProviderConfig", - cred_name="team-a", + nodeCount=2, ), - ), - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, + ], ), - "release-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler([])), - ), - }, + ).model_dump(exclude_none=True, mode="json"), ), - context=structpb.Struct(), ), + resources={ + "cluster": _observed_ready( + _cluster(cred_kind="ProviderConfig", cred_name="team-a"), + external_name=_CLUSTER_ID, + ), + }, ), - ] - - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct( + _network(cred_kind="ProviderConfig", cred_name="team-a"), + ), + ), + "firewall": fnv1.Resource( + resource=resource.dict_to_struct( + _firewall(cred_kind="ProviderConfig", cred_name="team-a"), + ), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct( + _cluster(cred_kind="ProviderConfig", cred_name="team-a"), + ), + ready=fnv1.READY_TRUE, + ), + "node-pool-workers": fnv1.Resource( + resource=resource.dict_to_struct( + _node_pool( + label="workers", + size="g4p.kube.small", + node_count=2, + labels={"modelplane.ai/pool": "workers"}, + cred_kind="ProviderConfig", + cred_name="team-a", + ), + ), + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config_helm()), + ready=fnv1.READY_TRUE, + ), + "release-cluster-autoscaler": fnv1.Resource( + resource=resource.dict_to_struct(_autoscaler([])), + ), + }, + ), + context=structpb.Struct(), + ), + ), +] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """The function composes Civo cluster infrastructure.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-eks-cluster/tests/__init__.py b/functions/compose-eks-cluster/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-eks-cluster/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-eks-cluster/tests/test_fn.py b/functions/compose-eks-cluster/tests/test_fn.py index d0ba254ba..9658abec0 100644 --- a/functions/compose-eks-cluster/tests/test_fn.py +++ b/functions/compose-eks-cluster/tests/test_fn.py @@ -14,15 +14,17 @@ """Tests for the compose-eks-cluster function.""" +import asyncio import dataclasses -import unittest +import json from typing import Any -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.ekscluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -37,10 +39,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - _KUBECONFIG_SECRET = "test-cluster-kubeconfig-55b57" _SUBNET_A = "test-cluster-subnet-us-west-2a-952dc" _SUBNET_B = "test-cluster-subnet-us-west-2b-2b80f" @@ -1139,403 +1137,386 @@ def _expected_resources() -> dict: } -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function composes EKS cluster infrastructure.""" - # Second pass: cluster and cluster-auth observed Ready, function flips - # those two desired resources ready while still emitting everything. - ready_resources = _expected_resources() - ready_resources["cluster"] = fnv1.Resource( - resource=ready_resources["cluster"].resource, - ready=fnv1.READY_TRUE, - ) - ready_resources["cluster-auth"] = fnv1.Resource( - resource=ready_resources["cluster-auth"].resource, - ready=fnv1.READY_TRUE, - ) - ready_resources["efs-filesystem"] = fnv1.Resource( - resource=ready_resources["efs-filesystem"].resource, - ready=fnv1.READY_TRUE, - ) - # Once the EFS filesystem id is observed, the managed StorageClass Object - # is composed (and marked ready) against the cluster's own ProviderConfig. - ready_resources["storage-class-rwx-efs"] = fnv1.Resource( - resource=resource.dict_to_struct(_storage_class_object("fs-0abc123")), - ready=fnv1.READY_TRUE, - ) - # With the cluster observed, the autoscaler Helm release is composed (it's - # gated on the cluster existing so provider-helm can reach it). It carries - # no Ready condition yet, so it stays not-ready this pass. - ready_resources["release-cluster-autoscaler"] = fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler_release()), - ) +def _compose_cases() -> list[Case]: + """The cases test_compose runs.""" + # Second pass: cluster and cluster-auth observed Ready, function flips + # those two desired resources ready while still emitting everything. + ready_resources = _expected_resources() + ready_resources["cluster"] = fnv1.Resource( + resource=ready_resources["cluster"].resource, + ready=fnv1.READY_TRUE, + ) + ready_resources["cluster-auth"] = fnv1.Resource( + resource=ready_resources["cluster-auth"].resource, + ready=fnv1.READY_TRUE, + ) + ready_resources["efs-filesystem"] = fnv1.Resource( + resource=ready_resources["efs-filesystem"].resource, + ready=fnv1.READY_TRUE, + ) + # Once the EFS filesystem id is observed, the managed StorageClass Object + # is composed (and marked ready) against the cluster's own ProviderConfig. + ready_resources["storage-class-rwx-efs"] = fnv1.Resource( + resource=resource.dict_to_struct(_storage_class_object("fs-0abc123")), + ready=fnv1.READY_TRUE, + ) + # With the cluster observed, the autoscaler Helm release is composed (it's + # gated on the cluster existing so provider-helm can reach it). It carries + # no Ready condition yet, so it stays not-ready this pass. + ready_resources["release-cluster-autoscaler"] = fnv1.Resource( + resource=resource.dict_to_struct(_autoscaler_release()), + ) - cases = [ - Case( - name="first pass composes infra resources; none ready", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr().model_dump(exclude_none=True, mode="json"), - ), + return [ + Case( + name="first pass composes infra resources; none ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr().model_dump(exclude_none=True, mode="json"), ), ), ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources=_expected_resources(), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct(_expected_status()), ), - context=structpb.Struct(), + resources=_expected_resources(), ), + context=structpb.Struct(), ), - Case( - name="second pass with observed cluster ready marks cluster resources ready", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( + ), + Case( + name="second pass with observed cluster ready marks cluster resources ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr().model_dump(exclude_none=True, mode="json"), + ), + ), + resources={ + "cluster": fnv1.Resource( resource=resource.dict_to_struct( - _xr().model_dump(exclude_none=True, mode="json"), + { + **_eks_cluster(), + "status": {"conditions": [_ready_condition()]}, + }, ), ), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_eks_cluster(), - "status": {"conditions": [_ready_condition()]}, - }, - ), - ), - "cluster-auth": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_cluster_auth(), - "status": {"conditions": [_ready_condition()]}, - }, - ), + "cluster-auth": fnv1.Resource( + resource=resource.dict_to_struct( + { + **_cluster_auth(), + "status": {"conditions": [_ready_condition()]}, + }, ), - "efs-filesystem": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_efs_filesystem(), - "metadata": {"annotations": {"crossplane.io/external-name": "fs-0abc123"}}, - "status": {"conditions": [_ready_condition()]}, - }, - ), + ), + "efs-filesystem": fnv1.Resource( + resource=resource.dict_to_struct( + { + **_efs_filesystem(), + "metadata": {"annotations": {"crossplane.io/external-name": "fs-0abc123"}}, + "status": {"conditions": [_ready_condition()]}, + }, ), - }, - ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), ), - resources=ready_resources, - ), - context=structpb.Struct(), + }, ), ), - ] - - for case in cases: - with self.subTest(name=case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - ) - - async def test_compose_capacity_block(self) -> None: - """A Capacity Block pool composes a launch template and a CAPACITY_BLOCK node group. - - The GPU node group must not set instanceTypes (EKS takes the type - from the launch template), must set capacityType=CAPACITY_BLOCK, and - must reference the launch template. The launch template targets the - reservation via the capacity-block market type. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr_capacity_block().model_dump(exclude_none=True, mode="json"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct(_expected_status()), ), + resources=ready_resources, ), + context=structpb.Struct(), ), - ) + ), + ] - got = await self.runner.RunFunction(req, None) - resources = got.desired.resources - # The launch template is composed and targets the reservation. - self.assertIn("launch-template-gpu-h200", resources) - self.assertEqual( - _launch_template(), - resource.struct_to_dict(resources["launch-template-gpu-h200"].resource), - ) +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - # The GPU node group uses CAPACITY_BLOCK + the launch template and - # carries no instanceTypes. - self.assertEqual( - _gpu_node_group_capacity_block(), - resource.struct_to_dict(resources["nodegroup-gpu-h200"].resource), - ) - async def test_compose_efa(self) -> None: - """An EFA GPU pool composes EFA infrastructure end to end. - - The node group's launch template carries one EFA interface per network - card (card 0 keeps device index 0 for the node's IP traffic, the rest - device index 1 for RDMA), the cluster gets an EFA security group with - self-referencing all-traffic ingress and egress rules, and the node - group references the launch template instead of setting instanceTypes. - """ - want_resources = { - "vpc": fnv1.Resource(resource=resource.dict_to_struct(_vpc())), - "subnet-0": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_A, "us-west-2a", "10.0.0.0/20")), - ), - "subnet-1": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_B, "us-west-2b", "10.0.16.0/20")), - ), - "subnet-2": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_C, "us-west-2c", "10.0.32.0/20")), - ), - "private-subnet-0": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_A, "us-west-2a", "10.0.48.0/20")), - ), - "private-subnet-1": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_B, "us-west-2b", "10.0.64.0/20")), - ), - "private-subnet-2": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_C, "us-west-2c", "10.0.80.0/20")), - ), - "internet-gateway": fnv1.Resource(resource=resource.dict_to_struct(_internet_gateway())), - "nat-eip": fnv1.Resource(resource=resource.dict_to_struct(_nat_eip())), - "nat-gateway": fnv1.Resource(resource=resource.dict_to_struct(_nat_gateway("us-west-2a"))), - "route-table": fnv1.Resource(resource=resource.dict_to_struct(_route_table())), - "route-default": fnv1.Resource(resource=resource.dict_to_struct(_route_default())), - "private-route-table": fnv1.Resource(resource=resource.dict_to_struct(_private_route_table())), - "private-route-default": fnv1.Resource(resource=resource.dict_to_struct(_private_route_default())), - "route-table-association-0": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2a")), - ), - "route-table-association-1": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2b")), - ), - "route-table-association-2": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2c")), - ), - "private-route-table-association-0": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2a")), - ), - "private-route-table-association-1": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2b")), - ), - "private-route-table-association-2": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2c")), - ), - "iam-role-cluster": fnv1.Resource( - resource=resource.dict_to_struct(_role("cluster", _ASSUME_CLUSTER)), - ), - "iam-attach-cluster-policy": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("cluster", "arn:aws:iam::aws:policy/AmazonEKSClusterPolicy"), - ), - ), - "iam-role-node": fnv1.Resource(resource=resource.dict_to_struct(_role("node", _ASSUME_NODE))), - "iam-attach-node-worker": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy"), - ), - ), - "iam-attach-node-cni": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy"), - ), - ), - "iam-attach-node-ecr": fnv1.Resource( +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """The function composes EKS cluster infrastructure.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) + + +def test_compose_capacity_block() -> None: + """A Capacity Block pool composes a launch template and a CAPACITY_BLOCK node group.""" + # The GPU node group must not set instanceTypes (EKS takes the type + # from the launch template), must set capacityType=CAPACITY_BLOCK, and + # must reference the launch template. The launch template targets the + # reservation via the capacity-block market type. + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly"), + _xr_capacity_block().model_dump(exclude_none=True, mode="json"), ), ), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_eks_cluster())), - "cluster-auth": fnv1.Resource(resource=resource.dict_to_struct(_cluster_auth())), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_system_node_group())), - "launch-template-gpu-h200": fnv1.Resource(resource=resource.dict_to_struct(_launch_template_efa())), - "efa-security-group": fnv1.Resource(resource=resource.dict_to_struct(_efa_security_group())), - "efa-security-group-ingress": fnv1.Resource( - resource=resource.dict_to_struct(_efa_security_group_ingress()), - ), - "efa-security-group-egress": fnv1.Resource( - resource=resource.dict_to_struct(_efa_security_group_egress()), - ), - "nodegroup-gpu-h200": fnv1.Resource(resource=resource.dict_to_struct(_gpu_node_group_efa())), - "addon-vpc-cni": fnv1.Resource(resource=resource.dict_to_struct(_addon("vpc-cni"))), - "addon-kube-proxy": fnv1.Resource(resource=resource.dict_to_struct(_addon("kube-proxy"))), - "addon-coredns": fnv1.Resource(resource=resource.dict_to_struct(_addon("coredns"))), - "efs-filesystem": fnv1.Resource(resource=resource.dict_to_struct(_efs_filesystem())), - "efs-security-group": fnv1.Resource(resource=resource.dict_to_struct(_efs_security_group())), - "efs-security-group-ingress": fnv1.Resource( - resource=resource.dict_to_struct(_efs_security_group_ingress()), - ), - "efs-mount-target-0": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_A))), - "efs-mount-target-1": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_B))), - "efs-mount-target-2": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_C))), - "iam-role-efs-csi": fnv1.Resource( - resource=resource.dict_to_struct(_role("efs-csi", _ASSUME_POD_IDENTITY)), - ), - "iam-attach-efs-csi": fnv1.Resource( - resource=resource.dict_to_struct(_role_policy_attachment("efs-csi", _POLICY_EFS_CSI)), - ), - "addon-eks-pod-identity-agent": fnv1.Resource( - resource=resource.dict_to_struct(_addon("eks-pod-identity-agent")), - ), - "pod-identity-efs-csi": fnv1.Resource(resource=resource.dict_to_struct(_pod_identity_association())), - "addon-aws-efs-csi-driver": fnv1.Resource( - resource=resource.dict_to_struct(_addon("aws-efs-csi-driver")), - ), - "iam-policy-cluster-autoscaler": fnv1.Resource(resource=resource.dict_to_struct(_autoscaler_policy())), - "iam-role-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_role("cluster-autoscaler", _ASSUME_POD_IDENTITY)), - ), - "iam-attach-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler_attachment()), + ), + ) + + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + resources = got.desired.resources + + # The launch template is composed and targets the reservation. + assert "launch-template-gpu-h200" in resources + assert resource.struct_to_dict(resources["launch-template-gpu-h200"].resource) == _launch_template() + + # The GPU node group uses CAPACITY_BLOCK + the launch template and + # carries no instanceTypes. + assert resource.struct_to_dict(resources["nodegroup-gpu-h200"].resource) == _gpu_node_group_capacity_block() + + +def test_compose_efa() -> None: + """An EFA GPU pool composes EFA infrastructure end to end.""" + # The node group's launch template carries one EFA interface per network + # card (card 0 keeps device index 0 for the node's IP traffic, the rest + # device index 1 for RDMA), the cluster gets an EFA security group with + # self-referencing all-traffic ingress and egress rules, and the node + # group references the launch template instead of setting instanceTypes. + want_resources = { + "vpc": fnv1.Resource(resource=resource.dict_to_struct(_vpc())), + "subnet-0": fnv1.Resource( + resource=resource.dict_to_struct(_subnet(_SUBNET_A, "us-west-2a", "10.0.0.0/20")), + ), + "subnet-1": fnv1.Resource( + resource=resource.dict_to_struct(_subnet(_SUBNET_B, "us-west-2b", "10.0.16.0/20")), + ), + "subnet-2": fnv1.Resource( + resource=resource.dict_to_struct(_subnet(_SUBNET_C, "us-west-2c", "10.0.32.0/20")), + ), + "private-subnet-0": fnv1.Resource( + resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_A, "us-west-2a", "10.0.48.0/20")), + ), + "private-subnet-1": fnv1.Resource( + resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_B, "us-west-2b", "10.0.64.0/20")), + ), + "private-subnet-2": fnv1.Resource( + resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_C, "us-west-2c", "10.0.80.0/20")), + ), + "internet-gateway": fnv1.Resource(resource=resource.dict_to_struct(_internet_gateway())), + "nat-eip": fnv1.Resource(resource=resource.dict_to_struct(_nat_eip())), + "nat-gateway": fnv1.Resource(resource=resource.dict_to_struct(_nat_gateway("us-west-2a"))), + "route-table": fnv1.Resource(resource=resource.dict_to_struct(_route_table())), + "route-default": fnv1.Resource(resource=resource.dict_to_struct(_route_default())), + "private-route-table": fnv1.Resource(resource=resource.dict_to_struct(_private_route_table())), + "private-route-default": fnv1.Resource(resource=resource.dict_to_struct(_private_route_default())), + "route-table-association-0": fnv1.Resource( + resource=resource.dict_to_struct(_route_table_association("us-west-2a")), + ), + "route-table-association-1": fnv1.Resource( + resource=resource.dict_to_struct(_route_table_association("us-west-2b")), + ), + "route-table-association-2": fnv1.Resource( + resource=resource.dict_to_struct(_route_table_association("us-west-2c")), + ), + "private-route-table-association-0": fnv1.Resource( + resource=resource.dict_to_struct(_private_route_table_association("us-west-2a")), + ), + "private-route-table-association-1": fnv1.Resource( + resource=resource.dict_to_struct(_private_route_table_association("us-west-2b")), + ), + "private-route-table-association-2": fnv1.Resource( + resource=resource.dict_to_struct(_private_route_table_association("us-west-2c")), + ), + "iam-role-cluster": fnv1.Resource( + resource=resource.dict_to_struct(_role("cluster", _ASSUME_CLUSTER)), + ), + "iam-attach-cluster-policy": fnv1.Resource( + resource=resource.dict_to_struct( + _role_policy_attachment("cluster", "arn:aws:iam::aws:policy/AmazonEKSClusterPolicy"), ), - "pod-identity-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler_pod_identity()), + ), + "iam-role-node": fnv1.Resource(resource=resource.dict_to_struct(_role("node", _ASSUME_NODE))), + "iam-attach-node-worker": fnv1.Resource( + resource=resource.dict_to_struct( + _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy"), ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config("kubernetes.m.crossplane.io/v1alpha1")), - ready=fnv1.READY_TRUE, + ), + "iam-attach-node-cni": fnv1.Resource( + resource=resource.dict_to_struct( + _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy"), ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config("helm.m.crossplane.io/v1beta1")), - ready=fnv1.READY_TRUE, + ), + "iam-attach-node-ecr": fnv1.Resource( + resource=resource.dict_to_struct( + _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly"), ), - } + ), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_eks_cluster())), + "cluster-auth": fnv1.Resource(resource=resource.dict_to_struct(_cluster_auth())), + "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_system_node_group())), + "launch-template-gpu-h200": fnv1.Resource(resource=resource.dict_to_struct(_launch_template_efa())), + "efa-security-group": fnv1.Resource(resource=resource.dict_to_struct(_efa_security_group())), + "efa-security-group-ingress": fnv1.Resource( + resource=resource.dict_to_struct(_efa_security_group_ingress()), + ), + "efa-security-group-egress": fnv1.Resource( + resource=resource.dict_to_struct(_efa_security_group_egress()), + ), + "nodegroup-gpu-h200": fnv1.Resource(resource=resource.dict_to_struct(_gpu_node_group_efa())), + "addon-vpc-cni": fnv1.Resource(resource=resource.dict_to_struct(_addon("vpc-cni"))), + "addon-kube-proxy": fnv1.Resource(resource=resource.dict_to_struct(_addon("kube-proxy"))), + "addon-coredns": fnv1.Resource(resource=resource.dict_to_struct(_addon("coredns"))), + "efs-filesystem": fnv1.Resource(resource=resource.dict_to_struct(_efs_filesystem())), + "efs-security-group": fnv1.Resource(resource=resource.dict_to_struct(_efs_security_group())), + "efs-security-group-ingress": fnv1.Resource( + resource=resource.dict_to_struct(_efs_security_group_ingress()), + ), + "efs-mount-target-0": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_A))), + "efs-mount-target-1": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_B))), + "efs-mount-target-2": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_C))), + "iam-role-efs-csi": fnv1.Resource( + resource=resource.dict_to_struct(_role("efs-csi", _ASSUME_POD_IDENTITY)), + ), + "iam-attach-efs-csi": fnv1.Resource( + resource=resource.dict_to_struct(_role_policy_attachment("efs-csi", _POLICY_EFS_CSI)), + ), + "addon-eks-pod-identity-agent": fnv1.Resource( + resource=resource.dict_to_struct(_addon("eks-pod-identity-agent")), + ), + "pod-identity-efs-csi": fnv1.Resource(resource=resource.dict_to_struct(_pod_identity_association())), + "addon-aws-efs-csi-driver": fnv1.Resource( + resource=resource.dict_to_struct(_addon("aws-efs-csi-driver")), + ), + "iam-policy-cluster-autoscaler": fnv1.Resource(resource=resource.dict_to_struct(_autoscaler_policy())), + "iam-role-cluster-autoscaler": fnv1.Resource( + resource=resource.dict_to_struct(_role("cluster-autoscaler", _ASSUME_POD_IDENTITY)), + ), + "iam-attach-cluster-autoscaler": fnv1.Resource( + resource=resource.dict_to_struct(_autoscaler_attachment()), + ), + "pod-identity-cluster-autoscaler": fnv1.Resource( + resource=resource.dict_to_struct(_autoscaler_pod_identity()), + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config("kubernetes.m.crossplane.io/v1alpha1")), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config("helm.m.crossplane.io/v1beta1")), + ready=fnv1.READY_TRUE, + ), + } - case = Case( - name="an EFA pool composes EFA launch template, security group, and rules", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr_efa().model_dump(exclude_none=True, mode="json"), - ), + case = Case( + name="an EFA pool composes EFA launch template, security group, and rules", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr_efa().model_dump(exclude_none=True, mode="json"), ), ), ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources=want_resources, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct(_expected_status()), ), - context=structpb.Struct(), + resources=want_resources, ), - ) - - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - ) + context=structpb.Struct(), + ), + ) - async def test_compose_efa_cluster_security_group(self) -> None: - """Once both security groups are observed, every interface carries them. - - A launch template with networkInterfaces makes its security groups - authoritative, so the interfaces must carry both the EFA security group - and the EKS cluster security group or the node never joins. Both are set - as raw IDs in securityGroups (not securityGroupRefs): the provider's - reference resolver no-ops once that field is populated, so a ref mixed - with a literal would be dropped. The EFA group's ID comes from its - observed external name, the cluster group's from the observed cluster's - status, so both appear only once their resources report them. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr_efa().model_dump(exclude_none=True, mode="json"), - ), + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) + + +def test_compose_efa_cluster_security_group() -> None: + """Once both security groups are observed, every interface carries them.""" + # A launch template with networkInterfaces makes its security groups + # authoritative, so the interfaces must carry both the EFA security group + # and the EKS cluster security group or the node never joins. Both are set + # as raw IDs in securityGroups (not securityGroupRefs): the provider's + # reference resolver no-ops once that field is populated, so a ref mixed + # with a literal would be dropped. The EFA group's ID comes from its + # observed external name, the cluster group's from the observed cluster's + # status, so both appear only once their resources report them. + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr_efa().model_dump(exclude_none=True, mode="json"), ), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_eks_cluster(), - "status": { - "atProvider": { - "vpcConfig": {"clusterSecurityGroupId": "sg-0cluster"}, - }, + ), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + **_eks_cluster(), + "status": { + "atProvider": { + "vpcConfig": {"clusterSecurityGroupId": "sg-0cluster"}, }, }, - ), + }, ), - "efa-security-group": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_efa_security_group(), - "metadata": { - **_efa_security_group()["metadata"], - "annotations": {"crossplane.io/external-name": "sg-0efa"}, - }, + ), + "efa-security-group": fnv1.Resource( + resource=resource.dict_to_struct( + { + **_efa_security_group(), + "metadata": { + **_efa_security_group()["metadata"], + "annotations": {"crossplane.io/external-name": "sg-0efa"}, }, - ), + }, ), - }, - ), - ) - - got = await self.runner.RunFunction(req, None) - lt = resource.struct_to_dict(got.desired.resources["launch-template-gpu-h200"].resource) - interfaces = lt["spec"]["forProvider"]["networkInterfaces"] - - # Every interface carries both SGs as raw IDs (EFA first, then cluster) - # and no securityGroupRefs; no interface requests a public IP (nodes are - # in private subnets). - self.assertEqual("efa", interfaces[0]["interfaceType"]) - for ni in interfaces: - self.assertNotIn("securityGroupRefs", ni) - self.assertEqual(["sg-0efa", "sg-0cluster"], ni["securityGroups"]) - self.assertNotIn("associatePublicIpAddress", ni) - for ni in interfaces[1:]: - self.assertEqual("efa-only", ni["interfaceType"]) - - async def test_compose_efa_dra_driver(self) -> None: - """An EFA pool installs the EFA DRA driver Helm release. - - Like the autoscaler, the release is gated on the cluster being observed - so provider-helm can reach it. A pool without the EFA fabric installs no - driver even once the cluster is observed. - """ - observed_cluster = { - "cluster": fnv1.Resource( - resource=resource.dict_to_struct( - {**_eks_cluster(), "status": {"conditions": [_ready_condition()]}}, ), + }, + ), + ) + + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + lt = resource.struct_to_dict(got.desired.resources["launch-template-gpu-h200"].resource) + interfaces = lt["spec"]["forProvider"]["networkInterfaces"] + + # Every interface carries both SGs as raw IDs (EFA first, then cluster) + # and no securityGroupRefs; no interface requests a public IP (nodes are + # in private subnets). + assert interfaces[0]["interfaceType"] == "efa" + for ni in interfaces: + assert "securityGroupRefs" not in ni + assert ni["securityGroups"] == ["sg-0efa", "sg-0cluster"] + assert "associatePublicIpAddress" not in ni + for ni in interfaces[1:]: + assert ni["interfaceType"] == "efa-only" + + +def test_compose_efa_dra_driver() -> None: + """An EFA pool installs the EFA DRA driver Helm release, and a pool without EFA doesn't.""" + # Like the autoscaler, the release is gated on the cluster being observed + # so provider-helm can reach it. A pool without the EFA fabric installs no + # driver even once the cluster is observed. + observed_cluster = { + "cluster": fnv1.Resource( + resource=resource.dict_to_struct( + {**_eks_cluster(), "status": {"conditions": [_ready_condition()]}}, ), - } + ), + } - got_efa = await self.runner.RunFunction( + got_efa = asyncio.run( + fn.FunctionRunner().RunFunction( fnv1.RunFunctionRequest( observed=fnv1.State( composite=fnv1.Resource( @@ -1546,13 +1527,15 @@ async def test_compose_efa_dra_driver(self) -> None: ), None, ) - self.assertIn("release-efa-dra-driver", got_efa.desired.resources) - self.assertEqual( - _efa_dra_driver_release(), - resource.struct_to_dict(got_efa.desired.resources["release-efa-dra-driver"].resource), - ) + ) + assert "release-efa-dra-driver" in got_efa.desired.resources + assert ( + resource.struct_to_dict(got_efa.desired.resources["release-efa-dra-driver"].resource) + == _efa_dra_driver_release() + ) - got_none = await self.runner.RunFunction( + got_none = asyncio.run( + fn.FunctionRunner().RunFunction( fnv1.RunFunctionRequest( observed=fnv1.State( composite=fnv1.Resource( @@ -1563,100 +1546,97 @@ async def test_compose_efa_dra_driver(self) -> None: ), None, ) - self.assertNotIn("release-efa-dra-driver", got_none.desired.resources) - - async def test_custom_credentials(self) -> None: - """Custom credentials flow through to all cloud MRs. - - When spec.credentials is set with a custom type and name, every cloud - provider MR (VPC, subnets, IAM roles, EKS cluster, node groups, addons, - EFS resources, autoscaler IAM resources) carries the corresponding - providerConfigRef. The kubeconfig-based resources (provider-config-kubernetes, - provider-config-helm, release-*, storage-class-*) are unaffected. - """ - ck = "ProviderConfig" - cn = "my-aws-account" - creds = v1alpha1.Credentials(type=ck, name=cn) - - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr(credentials=creds).model_dump(exclude_none=True, mode="json"), - ), + ) + assert "release-efa-dra-driver" not in got_none.desired.resources + + +def test_custom_credentials() -> None: + """Custom credentials flow through to all cloud MRs.""" + # When spec.credentials is set with a custom type and name, every cloud + # provider MR (VPC, subnets, IAM roles, EKS cluster, node groups, addons, + # EFS resources, autoscaler IAM resources) carries the corresponding + # providerConfigRef. The kubeconfig-based resources (provider-config-kubernetes, + # provider-config-helm, release-*, storage-class-*) are unaffected. + ck = "ProviderConfig" + cn = "my-aws-account" + creds = v1alpha1.Credentials(type=ck, name=cn) + + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr(credentials=creds).model_dump(exclude_none=True, mode="json"), ), ), - ) + ), + ) - got = await self.runner.RunFunction(req, None) - rs = got.desired.resources - - cloud_checks = { - "vpc": _vpc(ck, cn), - "subnet-0": _subnet(_SUBNET_A, "us-west-2a", "10.0.0.0/20", ck, cn), - "subnet-1": _subnet(_SUBNET_B, "us-west-2b", "10.0.16.0/20", ck, cn), - "subnet-2": _subnet(_SUBNET_C, "us-west-2c", "10.0.32.0/20", ck, cn), - "private-subnet-0": _private_subnet(_PRIVATE_SUBNET_A, "us-west-2a", "10.0.48.0/20", ck, cn), - "private-subnet-1": _private_subnet(_PRIVATE_SUBNET_B, "us-west-2b", "10.0.64.0/20", ck, cn), - "private-subnet-2": _private_subnet(_PRIVATE_SUBNET_C, "us-west-2c", "10.0.80.0/20", ck, cn), - "internet-gateway": _internet_gateway(ck, cn), - "nat-eip": _nat_eip(ck, cn), - "nat-gateway": _nat_gateway("us-west-2a", ck, cn), - "route-table": _route_table(ck, cn), - "route-default": _route_default(ck, cn), - "private-route-table": _private_route_table(ck, cn), - "private-route-default": _private_route_default(ck, cn), - "route-table-association-0": _route_table_association("us-west-2a", ck, cn), - "route-table-association-1": _route_table_association("us-west-2b", ck, cn), - "route-table-association-2": _route_table_association("us-west-2c", ck, cn), - "private-route-table-association-0": _private_route_table_association("us-west-2a", ck, cn), - "private-route-table-association-1": _private_route_table_association("us-west-2b", ck, cn), - "private-route-table-association-2": _private_route_table_association("us-west-2c", ck, cn), - "iam-role-cluster": _role("cluster", _ASSUME_CLUSTER, ck, cn), - "iam-attach-cluster-policy": _role_policy_attachment( - "cluster", "arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", ck, cn - ), - "iam-role-node": _role("node", _ASSUME_NODE, ck, cn), - "iam-attach-node-worker": _role_policy_attachment( - "node", "arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", ck, cn - ), - "iam-attach-node-cni": _role_policy_attachment( - "node", "arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", ck, cn - ), - "iam-attach-node-ecr": _role_policy_attachment( - "node", "arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", ck, cn - ), - "cluster": _eks_cluster(ck, cn), - "cluster-auth": _cluster_auth(ck, cn), - "nodegroup-system": _system_node_group(ck, cn), - "nodegroup-gpu-l4": _gpu_node_group(ck, cn), - "addon-vpc-cni": _addon("vpc-cni", ck, cn), - "addon-kube-proxy": _addon("kube-proxy", ck, cn), - "addon-coredns": _addon("coredns", ck, cn), - "efs-filesystem": _efs_filesystem(ck, cn), - "efs-security-group": _efs_security_group(ck, cn), - "efs-security-group-ingress": _efs_security_group_ingress(ck, cn), - "efs-mount-target-0": _efs_mount_target(_PRIVATE_SUBNET_A, ck, cn), - "efs-mount-target-1": _efs_mount_target(_PRIVATE_SUBNET_B, ck, cn), - "efs-mount-target-2": _efs_mount_target(_PRIVATE_SUBNET_C, ck, cn), - "iam-role-efs-csi": _role("efs-csi", _ASSUME_POD_IDENTITY, ck, cn), - "iam-attach-efs-csi": _role_policy_attachment("efs-csi", _POLICY_EFS_CSI, ck, cn), - "addon-eks-pod-identity-agent": _addon("eks-pod-identity-agent", ck, cn), - "pod-identity-efs-csi": _pod_identity_association(ck, cn), - "addon-aws-efs-csi-driver": _addon("aws-efs-csi-driver", ck, cn), - "iam-policy-cluster-autoscaler": _autoscaler_policy(ck, cn), - "iam-role-cluster-autoscaler": _role("cluster-autoscaler", _ASSUME_POD_IDENTITY, ck, cn), - "iam-attach-cluster-autoscaler": _autoscaler_attachment(ck, cn), - "pod-identity-cluster-autoscaler": _autoscaler_pod_identity(ck, cn), - } - - for key, want in cloud_checks.items(): - with self.subTest(resource=key): - self.assertIn(key, rs, f"resource {key!r} not found in desired") - got_dict = resource.struct_to_dict(rs[key].resource) - self.assertEqual(want, got_dict, f"resource {key!r} mismatch") - - # kubeconfig-based resources must NOT carry the cloud providerConfigRef - for key in ("provider-config-kubernetes", "provider-config-helm"): - got_dict = resource.struct_to_dict(rs[key].resource) - self.assertNotIn("providerConfigRef", got_dict.get("spec", {}), f"{key} should not have providerConfigRef") + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + rs = got.desired.resources + + cloud_checks = { + "vpc": _vpc(ck, cn), + "subnet-0": _subnet(_SUBNET_A, "us-west-2a", "10.0.0.0/20", ck, cn), + "subnet-1": _subnet(_SUBNET_B, "us-west-2b", "10.0.16.0/20", ck, cn), + "subnet-2": _subnet(_SUBNET_C, "us-west-2c", "10.0.32.0/20", ck, cn), + "private-subnet-0": _private_subnet(_PRIVATE_SUBNET_A, "us-west-2a", "10.0.48.0/20", ck, cn), + "private-subnet-1": _private_subnet(_PRIVATE_SUBNET_B, "us-west-2b", "10.0.64.0/20", ck, cn), + "private-subnet-2": _private_subnet(_PRIVATE_SUBNET_C, "us-west-2c", "10.0.80.0/20", ck, cn), + "internet-gateway": _internet_gateway(ck, cn), + "nat-eip": _nat_eip(ck, cn), + "nat-gateway": _nat_gateway("us-west-2a", ck, cn), + "route-table": _route_table(ck, cn), + "route-default": _route_default(ck, cn), + "private-route-table": _private_route_table(ck, cn), + "private-route-default": _private_route_default(ck, cn), + "route-table-association-0": _route_table_association("us-west-2a", ck, cn), + "route-table-association-1": _route_table_association("us-west-2b", ck, cn), + "route-table-association-2": _route_table_association("us-west-2c", ck, cn), + "private-route-table-association-0": _private_route_table_association("us-west-2a", ck, cn), + "private-route-table-association-1": _private_route_table_association("us-west-2b", ck, cn), + "private-route-table-association-2": _private_route_table_association("us-west-2c", ck, cn), + "iam-role-cluster": _role("cluster", _ASSUME_CLUSTER, ck, cn), + "iam-attach-cluster-policy": _role_policy_attachment( + "cluster", "arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", ck, cn + ), + "iam-role-node": _role("node", _ASSUME_NODE, ck, cn), + "iam-attach-node-worker": _role_policy_attachment( + "node", "arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", ck, cn + ), + "iam-attach-node-cni": _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", ck, cn), + "iam-attach-node-ecr": _role_policy_attachment( + "node", "arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", ck, cn + ), + "cluster": _eks_cluster(ck, cn), + "cluster-auth": _cluster_auth(ck, cn), + "nodegroup-system": _system_node_group(ck, cn), + "nodegroup-gpu-l4": _gpu_node_group(ck, cn), + "addon-vpc-cni": _addon("vpc-cni", ck, cn), + "addon-kube-proxy": _addon("kube-proxy", ck, cn), + "addon-coredns": _addon("coredns", ck, cn), + "efs-filesystem": _efs_filesystem(ck, cn), + "efs-security-group": _efs_security_group(ck, cn), + "efs-security-group-ingress": _efs_security_group_ingress(ck, cn), + "efs-mount-target-0": _efs_mount_target(_PRIVATE_SUBNET_A, ck, cn), + "efs-mount-target-1": _efs_mount_target(_PRIVATE_SUBNET_B, ck, cn), + "efs-mount-target-2": _efs_mount_target(_PRIVATE_SUBNET_C, ck, cn), + "iam-role-efs-csi": _role("efs-csi", _ASSUME_POD_IDENTITY, ck, cn), + "iam-attach-efs-csi": _role_policy_attachment("efs-csi", _POLICY_EFS_CSI, ck, cn), + "addon-eks-pod-identity-agent": _addon("eks-pod-identity-agent", ck, cn), + "pod-identity-efs-csi": _pod_identity_association(ck, cn), + "addon-aws-efs-csi-driver": _addon("aws-efs-csi-driver", ck, cn), + "iam-policy-cluster-autoscaler": _autoscaler_policy(ck, cn), + "iam-role-cluster-autoscaler": _role("cluster-autoscaler", _ASSUME_POD_IDENTITY, ck, cn), + "iam-attach-cluster-autoscaler": _autoscaler_attachment(ck, cn), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity(ck, cn), + } + + for key, want in cloud_checks.items(): + assert key in rs, f"resource {key!r} not found in desired" + got_dict = resource.struct_to_dict(rs[key].resource) + assert got_dict == want, f"resource {key!r} mismatch" + + # kubeconfig-based resources must NOT carry the cloud providerConfigRef + for key in ("provider-config-kubernetes", "provider-config-helm"): + got_dict = resource.struct_to_dict(rs[key].resource) + assert "providerConfigRef" not in got_dict.get("spec", {}), f"{key} should not have providerConfigRef" diff --git a/functions/compose-gke-cluster/tests/__init__.py b/functions/compose-gke-cluster/tests/__init__.py deleted file mode 100644 index 5d373016d..000000000 --- a/functions/compose-gke-cluster/tests/__init__.py +++ /dev/null @@ -1,15 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - - diff --git a/functions/compose-gke-cluster/tests/test_fn.py b/functions/compose-gke-cluster/tests/test_fn.py index 2f556d9de..07898ac6b 100644 --- a/functions/compose-gke-cluster/tests/test_fn.py +++ b/functions/compose-gke-cluster/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-gke-cluster function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.gkecluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -36,10 +38,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - _DEFAULT_CRED_KIND = "ClusterProviderConfig" _DEFAULT_CRED_NAME = "default" @@ -384,345 +382,330 @@ def _expected_status() -> dict: } -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function composes GKE cluster infrastructure.""" - req1 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr().model_dump(exclude_none=True, mode="json"), - ), +def _compose_cases() -> list[Case]: + """The cases for test_compose.""" + req1 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _gke_xr().model_dump(exclude_none=True, mode="json"), ), ), - ) - req1.required_resources["gcp-provider-config"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(_GCP_PROVIDER_CONFIG)) - ) - - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ), - "projectservice-filestore": fnv1.Resource( - resource=resource.dict_to_struct(_projectservice_filestore()), - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "nodepool-system": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_system()), - ), - "nodepool-gpu-pool": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu()), - ), - "service-account": fnv1.Resource( - resource=resource.dict_to_struct(_service_account()), - ), - "service-account-key": fnv1.Resource( - resource=resource.dict_to_struct(_service_account_key()), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_kubernetes()), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - }, + ), + ) + req1.required_resources["gcp-provider-config"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(_GCP_PROVIDER_CONFIG)) + ) + + want1 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct(_expected_status()), ), - context=structpb.Struct(), - ) - want1.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct(_network()), + ), + "projectservice-filestore": fnv1.Resource( + resource=resource.dict_to_struct(_projectservice_filestore()), + ), + "subnet": fnv1.Resource( + resource=resource.dict_to_struct(_subnet()), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ), + "nodepool-system": fnv1.Resource( + resource=resource.dict_to_struct(_nodepool_system()), + ), + "nodepool-gpu-pool": fnv1.Resource( + resource=resource.dict_to_struct(_nodepool_gpu()), + ), + "service-account": fnv1.Resource( + resource=resource.dict_to_struct(_service_account()), + ), + "service-account-key": fnv1.Resource( + resource=resource.dict_to_struct(_service_account_key()), + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config_kubernetes()), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config_helm()), + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + ) + want1.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) - req2 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr().model_dump(exclude_none=True, mode="json"), - ), + req2 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _gke_xr().model_dump(exclude_none=True, mode="json"), ), - resources={ - "service-account": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ServiceAccount", - "spec": { - "forProvider": {}, + ), + resources={ + "service-account": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ServiceAccount", + "spec": { + "forProvider": {}, + }, + "status": { + "atProvider": { + "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", }, - "status": { - "atProvider": { - "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", }, - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - }, - } - ), + ], + }, + } ), - "network": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "compute.gcp.m.upbound.io/v1beta1", - "kind": "Network", - # The external-name annotation carries the - # provider-generated VPC name, which the - # function pins the Filestore StorageClass to. - "metadata": { - "annotations": {"crossplane.io/external-name": "test-cluster-abc12"}, + ), + "network": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.gcp.m.upbound.io/v1beta1", + "kind": "Network", + # The external-name annotation carries the + # provider-generated VPC name, which the + # function pins the Filestore StorageClass to. + "metadata": { + "annotations": {"crossplane.io/external-name": "test-cluster-abc12"}, + }, + "spec": { + "forProvider": { + "autoCreateSubnetworks": False, }, - "spec": { - "forProvider": { - "autoCreateSubnetworks": False, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", }, - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - }, - } - ), + ], + }, + } ), - }, - ), - ) - req2.required_resources["gcp-provider-config"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(_GCP_PROVIDER_CONFIG)) - ) - - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), ), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ready=fnv1.READY_TRUE, - ), - "projectservice-filestore": fnv1.Resource( - resource=resource.dict_to_struct(_projectservice_filestore()), - ), - # With the network name known, the managed Filestore - # StorageClass is composed against the cluster's own - # provider-kubernetes ProviderConfig, pinned to the - # observed VPC. StorageClass has no Ready condition, - # so readiness is SuccessfulCreate. It's orphaned (no - # Delete policy) so it dies with the cluster instead of - # wedging on a deleted kubeconfig Secret during teardown. - "storage-class-rwx": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class_rwx("test-cluster-abc12")), - ready=fnv1.READY_TRUE, - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "nodepool-system": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_system()), - ), - "nodepool-gpu-pool": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu()), - ), - "service-account": fnv1.Resource( - resource=resource.dict_to_struct(_service_account()), - ready=fnv1.READY_TRUE, - ), - "service-account-key": fnv1.Resource( - resource=resource.dict_to_struct(_service_account_key()), - ), - "iam-binding": fnv1.Resource( - resource=resource.dict_to_struct(_iam_binding()), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_kubernetes()), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - }, + }, + ), + ) + req2.required_resources["gcp-provider-config"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(_GCP_PROVIDER_CONFIG)) + ) + + want2 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct(_expected_status()), ), - context=structpb.Struct(), - ) - want2.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) - - # The ProviderConfig resolved to nothing and no cluster is observed to - # take the project from, so nothing can be composed. The XR is marked - # not ready rather than left to aggregate to trivially ready. - req3 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr().model_dump(exclude_none=True, mode="json"), - ), + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct(_network()), + ready=fnv1.READY_TRUE, ), - ), - ) - req3.required_resources["gcp-provider-config"].SetInParent() - - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for GCP ClusterProviderConfig default", + "projectservice-filestore": fnv1.Resource( + resource=resource.dict_to_struct(_projectservice_filestore()), + ), + # With the network name known, the managed Filestore + # StorageClass is composed against the cluster's own + # provider-kubernetes ProviderConfig, pinned to the + # observed VPC. StorageClass has no Ready condition, + # so readiness is SuccessfulCreate. It's orphaned (no + # Delete policy) so it dies with the cluster instead of + # wedging on a deleted kubeconfig Secret during teardown. + "storage-class-rwx": fnv1.Resource( + resource=resource.dict_to_struct(_storage_class_rwx("test-cluster-abc12")), + ready=fnv1.READY_TRUE, + ), + "subnet": fnv1.Resource( + resource=resource.dict_to_struct(_subnet()), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ), + "nodepool-system": fnv1.Resource( + resource=resource.dict_to_struct(_nodepool_system()), + ), + "nodepool-gpu-pool": fnv1.Resource( + resource=resource.dict_to_struct(_nodepool_gpu()), + ), + "service-account": fnv1.Resource( + resource=resource.dict_to_struct(_service_account()), + ready=fnv1.READY_TRUE, + ), + "service-account-key": fnv1.Resource( + resource=resource.dict_to_struct(_service_account_key()), + ), + "iam-binding": fnv1.Resource( + resource=resource.dict_to_struct(_iam_binding()), + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config_kubernetes()), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config_helm()), + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + ) + want2.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) + + # The ProviderConfig resolved to nothing and no cluster is observed to + # take the project from, so nothing can be composed. The XR is marked + # not ready rather than left to aggregate to trivially ready. + req3 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _gke_xr().model_dump(exclude_none=True, mode="json"), ), - ], - context=structpb.Struct(), - ) - want3.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) - - cases = [ - Case(name="first pass composes infra resources; IAM binding gated", req=req1, want=want1), - Case(name="a missing ProviderConfig composes nothing and isn't ready", req=req3, want=want3), - Case( - name="second pass with observed SA email composes IAM binding and marks ready resources", - req=req2, - want=want2, ), - ] - - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) - - async def test_custom_credentials(self) -> None: - """Custom credentials flow through to all cloud MRs. - - When spec.credentials is set with a custom type and name, every cloud - provider MR (Network, ProjectService, Subnetwork, Cluster, NodePools, - ServiceAccount, ServiceAccountKey, IAM binding) carries the corresponding - providerConfigRef. The kubeconfig-based resources (provider-config-kubernetes, - provider-config-helm, storage-class-rwx) are unaffected. - """ - ck = "ProviderConfig" - cn = "my-gcp-account" - creds = v1alpha1.Credentials(type=ck, name=cn) - custom_pc = { - "apiVersion": "gcp.m.upbound.io/v1beta1", - "kind": "ProviderConfig", - "metadata": {"name": cn, "namespace": "crossplane-system"}, - "spec": { - "projectID": "my-gcp-project", - "credentials": { - "source": "Secret", - "secretRef": { - "name": "gcp-credentials", - "namespace": "crossplane-system", - "key": "credentials", - }, + ), + ) + req3.required_resources["gcp-provider-config"].SetInParent() + + want3 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for GCP ClusterProviderConfig default", + ), + ], + context=structpb.Struct(), + ) + want3.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) + + return [ + Case(name="first pass composes infra resources; IAM binding gated", req=req1, want=want1), + Case(name="a missing ProviderConfig composes nothing and isn't ready", req=req3, want=want3), + Case( + name="second pass with observed SA email composes IAM binding and marks ready resources", + req=req2, + want=want2, + ), + ] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes GKE cluster infrastructure.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) + + +def test_custom_credentials() -> None: + """Custom credentials flow through to all cloud MRs.""" + # When spec.credentials is set with a custom type and name, every cloud + # provider MR (Network, ProjectService, Subnetwork, Cluster, NodePools, + # ServiceAccount, ServiceAccountKey, IAM binding) carries the corresponding + # providerConfigRef. The kubeconfig-based resources (provider-config-kubernetes, + # provider-config-helm, storage-class-rwx) are unaffected. + ck = "ProviderConfig" + cn = "my-gcp-account" + creds = v1alpha1.Credentials(type=ck, name=cn) + custom_pc = { + "apiVersion": "gcp.m.upbound.io/v1beta1", + "kind": "ProviderConfig", + "metadata": {"name": cn, "namespace": "crossplane-system"}, + "spec": { + "projectID": "my-gcp-project", + "credentials": { + "source": "Secret", + "secretRef": { + "name": "gcp-credentials", + "namespace": "crossplane-system", + "key": "credentials", }, }, - } + }, + } - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr(credentials=creds).model_dump(exclude_none=True, mode="json"), - ), + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _gke_xr(credentials=creds).model_dump(exclude_none=True, mode="json"), ), - resources={ - "service-account": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ServiceAccount", - "spec": { - "forProvider": {}, - }, - "status": { - "atProvider": { - "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", - }, + ), + resources={ + "service-account": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ServiceAccount", + "spec": { + "forProvider": {}, + }, + "status": { + "atProvider": { + "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", }, - } - ), + }, + } ), - }, - ), - ) - req.required_resources["gcp-provider-config"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(custom_pc)) - ) - - got = await self.runner.RunFunction(req, None) - rs = got.desired.resources - - cloud_checks = { - "network": _network(ck, cn), - "projectservice-filestore": _projectservice_filestore(ck, cn), - "subnet": _subnet(ck, cn), - "cluster": _cluster(ck, cn), - "nodepool-system": _nodepool_system(ck, cn), - "nodepool-gpu-pool": _nodepool_gpu(ck, cn), - "service-account": _service_account(ck, cn), - "service-account-key": _service_account_key(ck, cn), - "iam-binding": _iam_binding("test-sa@my-gcp-project.iam.gserviceaccount.com", ck, cn), - } - - for key, want in cloud_checks.items(): - with self.subTest(resource=key): - self.assertIn(key, rs, f"resource {key!r} not found in desired") - got_dict = resource.struct_to_dict(rs[key].resource) - self.assertEqual(want, got_dict, f"resource {key!r} mismatch") - - # kubeconfig-based resources must NOT carry the cloud providerConfigRef - for key in ("provider-config-kubernetes", "provider-config-helm"): - got_dict = resource.struct_to_dict(rs[key].resource) - self.assertNotIn( - "providerConfigRef", - got_dict.get("spec", {}), - f"{key} should not have providerConfigRef", - ) - - custom_selector = fnv1.ResourceSelector( - api_version="gcp.m.upbound.io/v1beta1", - kind="ProviderConfig", - match_name=cn, - namespace="modelplane-system", - ) - self.assertEqual( - custom_selector, - got.requirements.resources["gcp-provider-config"], - ) + ), + }, + ), + ) + req.required_resources["gcp-provider-config"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(custom_pc)) + ) + + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + rs = got.desired.resources + + cloud_checks = { + "network": _network(ck, cn), + "projectservice-filestore": _projectservice_filestore(ck, cn), + "subnet": _subnet(ck, cn), + "cluster": _cluster(ck, cn), + "nodepool-system": _nodepool_system(ck, cn), + "nodepool-gpu-pool": _nodepool_gpu(ck, cn), + "service-account": _service_account(ck, cn), + "service-account-key": _service_account_key(ck, cn), + "iam-binding": _iam_binding("test-sa@my-gcp-project.iam.gserviceaccount.com", ck, cn), + } + + got_cloud = {key: resource.struct_to_dict(r.resource) for key, r in rs.items() if key in cloud_checks} + assert got_cloud == cloud_checks + + # kubeconfig-based resources must NOT carry the cloud providerConfigRef + for key in ("provider-config-kubernetes", "provider-config-helm"): + got_dict = resource.struct_to_dict(rs[key].resource) + assert "providerConfigRef" not in got_dict.get("spec", {}), f"{key} should not have providerConfigRef" + + custom_selector = fnv1.ResourceSelector( + api_version="gcp.m.upbound.io/v1beta1", + kind="ProviderConfig", + match_name=cn, + namespace="modelplane-system", + ) + assert _to_dict(got.requirements.resources["gcp-provider-config"]) == _to_dict(custom_selector) diff --git a/functions/compose-inference-class/tests/__init__.py b/functions/compose-inference-class/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-inference-class/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-inference-class/tests/test_fn.py b/functions/compose-inference-class/tests/test_fn.py index d4c5c99c4..327807286 100644 --- a/functions/compose-inference-class/tests/test_fn.py +++ b/functions/compose-inference-class/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-inference-class function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.inferenceclass import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -36,70 +38,60 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function marks the InferenceClass as ready.""" - cases = [ - Case( - name="marks XR ready with Accepted condition and empty status", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceClass( - metadata=metav1.ObjectMeta(name="gpu-l4"), - spec=v1alpha1.Spec( - devices=[ - v1alpha1.Device( - name="gpu", - claim="DRA", - driver="gpu.nvidia.com", - deviceClassName="gpu.nvidia.com", - count=1, - capacity={"memory": v1alpha1.Capacity(value="24Gi")}, - ), - ], +COMPOSE_CASES = [ + Case( + name="marks XR ready with Accepted condition and empty status", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceClass( + metadata=metav1.ObjectMeta(name="gpu-l4"), + spec=v1alpha1.Spec( + devices=[ + v1alpha1.Device( + name="gpu", + claim="DRA", + driver="gpu.nvidia.com", + deviceClassName="gpu.nvidia.com", + count=1, + capacity={"memory": v1alpha1.Capacity(value="24Gi")}, ), - ).model_dump(exclude_none=True, mode="json") + ], ), - ), + ).model_dump(exclude_none=True, mode="json") ), ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {}}), - ready=fnv1.READY_TRUE, - ), - ), - conditions=[ - fnv1.Condition( - type="Accepted", - status=fnv1.STATUS_CONDITION_TRUE, - reason="Available", - ), - ], - context=structpb.Struct(), + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {}}), + ready=fnv1.READY_TRUE, ), ), - ] + conditions=[ + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_TRUE, + reason="Available", + ), + ], + context=structpb.Struct(), + ), + ), +] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction marks the InferenceClass ready.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-inference-cluster/tests/__init__.py b/functions/compose-inference-cluster/tests/__init__.py deleted file mode 100644 index 5d373016d..000000000 --- a/functions/compose-inference-cluster/tests/__init__.py +++ /dev/null @@ -1,15 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - - diff --git a/functions/compose-inference-cluster/tests/test_fn.py b/functions/compose-inference-cluster/tests/test_fn.py index 69ed95122..4120990da 100644 --- a/functions/compose-inference-cluster/tests/test_fn.py +++ b/functions/compose-inference-cluster/tests/test_fn.py @@ -14,15 +14,17 @@ """Tests for the compose-inference-cluster function.""" +import asyncio import copy import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.inferencecluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -42,10 +44,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - _ACTIVATION_API_VERSION = "apiextensions.crossplane.io/v1alpha1" @@ -420,2114 +418,1992 @@ def _early_return_guard_case() -> tuple[fnv1.RunFunctionRequest, fnv1.RunFunctio return req, want -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() +def _compose_cases() -> list[Case]: # noqa: PLR0915 + """The RunFunction cases, built from shared bases. - async def test_compose(self) -> None: # noqa: PLR0915 - """The function composes an InferenceCluster. - - Many table entries, each exercising a distinct compose path across the - GKE, EKS, and Existing sources, push this over the statement limit. - """ - # Shared InferenceClass resource for required_resources. - inference_class_l4 = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-l4"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", + Many table entries, each exercising a distinct compose path across the + GKE, EKS, and Existing sources, push this over the statement limit. + """ + # Shared InferenceClass resource for required_resources. + inference_class_l4 = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceClass", + "metadata": {"name": "gpu-l4"}, + "spec": { + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + "provisioning": { + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": { + "type": "nvidia-l4", "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], - "provisioning": { - "provider": "GKE", - "gke": { - "machineType": "g2-standard-48", - "diskSizeGb": 100, - "accelerator": { - "type": "nvidia-l4", - "count": 1, - }, }, }, }, - } - - # Shared resource selector for class requirement. - class_selector = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-l4", - ) + }, + } + + # Shared resource selector for class requirement. + class_selector = fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceClass", + match_name="gpu-l4", + ) - # --- Case 1: Existing cluster with secrets composes backend and CPC. --- - req1 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing( - secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), - ), + # --- Case 1: Existing cluster with secrets composes backend and CPC. --- + req1 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing( + secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - ), - ], ), - ).model_dump(exclude_none=True, mode="json") - ), + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + ), + ], + ), + ).model_dump(exclude_none=True, mode="json") ), ), - ) - req1.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) + ), + ) + req1.required_resources["class-gpu-l4"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) + ) - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + want1 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + }, + } + ), + ), + resources={ + "cluster-provider-config-kubernetes": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "my-kubeconfig", + "key": "kubeconfig", + }, }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "serving-stack": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "cloud": "Existing", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "type": "Kubeconfig", + "name": "my-kubeconfig", + "key": "kubeconfig", }, ], }, } ), ), - resources={ - "cluster-provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "my-kubeconfig", - "key": "kubeconfig", - }, - }, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Existing", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "my-kubeconfig", - "key": "kubeconfig", - }, - ], - }, - } - ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ], - context=structpb.Struct(), - ) - want1.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Installing", + ), + ], + context=structpb.Struct(), + ) + want1.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - # --- Case 1b: Existing cluster with a non-GCP identity threads the - # declared identity type into the CPC and the ServingStack. --- - req1b = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing( - secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), - identitySecretRef=v1alpha1.IdentitySecretRef( - name="nebius-creds", - key="credentials.json", - type="NebiusServiceAccountCredentials", - ), + # --- Case 1b: Existing cluster with a non-GCP identity threads the + # declared identity type into the CPC and the ServingStack. --- + req1b = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing( + secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), + identitySecretRef=v1alpha1.IdentitySecretRef( + name="nebius-creds", + key="credentials.json", + type="NebiusServiceAccountCredentials", ), ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - ), - ], ), - ).model_dump(exclude_none=True, mode="json") - ), + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + ), + ], + ), + ).model_dump(exclude_none=True, mode="json") ), ), - ) - req1b.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) + ), + ) + req1b.required_resources["class-gpu-l4"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) + ) - # want1b mirrors want1 but with the Nebius identity on the CPC and an - # extra ServingStack identity secret of the same type. - want1b = fnv1.RunFunctionResponse() - want1b.CopyFrom(want1) - cpc1b = want1b.desired.resources["cluster-provider-config-kubernetes"] - cpc1b_dict = resource.struct_to_dict(cpc1b.resource) - cpc1b_dict["spec"]["identity"] = { + # want1b mirrors want1 but with the Nebius identity on the CPC and an + # extra ServingStack identity secret of the same type. + want1b = fnv1.RunFunctionResponse() + want1b.CopyFrom(want1) + cpc1b = want1b.desired.resources["cluster-provider-config-kubernetes"] + cpc1b_dict = resource.struct_to_dict(cpc1b.resource) + cpc1b_dict["spec"]["identity"] = { + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "nebius-creds", + "key": "credentials.json", + }, + } + cpc1b.resource.CopyFrom(resource.dict_to_struct(cpc1b_dict)) + backend1b = want1b.desired.resources["serving-stack"] + backend1b_dict = resource.struct_to_dict(backend1b.resource) + backend1b_dict["spec"]["secrets"].append( + { "type": "NebiusServiceAccountCredentials", - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "nebius-creds", - "key": "credentials.json", - }, + "name": "nebius-creds", + "key": "credentials.json", } - cpc1b.resource.CopyFrom(resource.dict_to_struct(cpc1b_dict)) - backend1b = want1b.desired.resources["serving-stack"] - backend1b_dict = resource.struct_to_dict(backend1b.resource) - backend1b_dict["spec"]["secrets"].append( - { - "type": "NebiusServiceAccountCredentials", - "name": "nebius-creds", - "key": "credentials.json", - } - ) - backend1b.resource.CopyFrom(resource.dict_to_struct(backend1b_dict)) - # want1 gains the replica, route, cache and gateway requirements in place - # from the guard cases below, after this snapshot; add them here so want1b - # matches on its own. - want1b.requirements.resources["gateways"].CopyFrom(_gateways_selector()) - want1b.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) - want1b.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) - want1b.requirements.resources["model-caches"].CopyFrom(_caches_selector()) - - # --- Case 2: GKE cluster first pass - no observed GKE, classes resolved. --- - req2 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="GKE", - gke=v1alpha1.Gke( - region="us-central1", - ), + ) + backend1b.resource.CopyFrom(resource.dict_to_struct(backend1b_dict)) + # want1 gains the replica, route, cache and gateway requirements in place + # from the guard cases below, after this snapshot; add them here so want1b + # matches on its own. + want1b.requirements.resources["gateways"].CopyFrom(_gateways_selector()) + want1b.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) + want1b.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) + want1b.requirements.resources["model-caches"].CopyFrom(_caches_selector()) + + # --- Case 2: GKE cluster first pass - no observed GKE, classes resolved. --- + req2 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="GKE", + gke=v1alpha1.Gke( + region="us-central1", ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - zones=["us-central1-a"], - ), - ], ), - ).model_dump(exclude_none=True, mode="json") - ), + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], + ), + ], + ), + ).model_dump(exclude_none=True, mode="json") ), ), - ) - req2.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) + ), + ) + req2.required_resources["class-gpu-l4"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) + ) - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + want2 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + }, + } + ), + ), + resources={ + "gke-cluster": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "metadata": { + "name": "test-cluster", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "region": "us-central1", + "kubernetesVersion": "1.35", + "nodePools": [ { "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "role": "GPU", + "machineType": "g2-standard-48", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": { + "acceleratorType": "nvidia-l4", + "acceleratorCount": 1, + }, + "zones": ["us-central1-a"], }, ], }, } ), ), - resources={ - "gke-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-central1", - "kubernetesVersion": "1.35", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "machineType": "g2-standard-48", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - "acceleratorCount": 1, - }, - "zones": ["us-central1-a"], - }, - ], - }, - } - ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want2.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want2.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - # --- Case 3: Existing cluster second pass - backend observed ready. --- - req3 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing( - secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), - ), + # --- Case 3: Existing cluster second pass - backend observed ready. --- + req3 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing( + secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - ), - ], ), - ).model_dump(exclude_none=True, mode="json") - ), - ), - resources={ - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": {"name": "test-cluster-serving-stack-fd00b"}, - "status": { - "conditions": [{"type": "Ready", "status": "True"}], - "gateway": {"address": "34.55.100.10"}, - }, - } + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + ), + ], ), - ), - }, + ).model_dump(exclude_none=True, mode="json") + ), ), - ) - req3.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + resources={ + "serving-stack": fnv1.Resource( resource=resource.dict_to_struct( { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": {"name": "test-cluster-serving-stack-fd00b"}, "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ - { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], - }, - ], + "conditions": [{"type": "Ready", "status": "True"}], "gateway": {"address": "34.55.100.10"}, }, } ), ), - resources={ - "cluster-provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "my-kubeconfig", - "key": "kubeconfig", - }, - }, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Existing", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ + }, + ), + ) + req3.required_resources["class-gpu-l4"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) + ) + + want3 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "type": "Kubeconfig", - "name": "my-kubeconfig", - "key": "kubeconfig", + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - } - ), - ready=fnv1.READY_TRUE, - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="BackendHealthy", + ], + "gateway": {"address": "34.55.100.10"}, + }, + } ), - ], - context=structpb.Struct(), - ) - want3.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - - # --- Case 4: EKS cluster first pass - no observed EKS, classes resolved. --- - inference_class_l4_eks = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-l4-eks"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], - "provisioning": { - "provider": "EKS", - "eks": { - "instanceType": "g6.xlarge", - "diskSizeGb": 100, - "accelerator": {"type": "nvidia-l4", "count": 1}, - }, - }, - }, - } - class_selector_eks = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-l4-eks", - ) - - req4 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( + ), + resources={ + "cluster-provider-config-kubernetes": fnv1.Resource( resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="EKS", - eks=v1alpha1.Eks(region="us-west-2"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4-eks", - nodeCount=2, - maxNodeCount=4, - zones=["us-west-2a", "us-west-2b"], - ), - ], - ), - ).model_dump(exclude_none=True, mode="json"), + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "my-kubeconfig", + "key": "kubeconfig", + }, + }, + }, + } ), + ready=fnv1.READY_TRUE, ), - ), - ) - req4.required_resources["class-gpu-l4-eks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), - ) - - want4 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + "serving-stack": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "cloud": "Existing", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "type": "Kubeconfig", + "name": "my-kubeconfig", + "key": "kubeconfig", }, ], }, - }, + } ), + ready=fnv1.READY_TRUE, ), - resources={ - "eks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-west-2", - "kubernetesVersion": "1.36", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - }, - "zones": ["us-west-2a", "us-west-2b"], - }, - ], - }, - }, - ), - ), + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="BackendHealthy", + ), + ], + context=structpb.Struct(), + ) + want3.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) + + # --- Case 4: EKS cluster first pass - no observed EKS, classes resolved. --- + inference_class_l4_eks = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceClass", + "metadata": {"name": "gpu-l4-eks"}, + "spec": { + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), ], - context=structpb.Struct(), - ) - want4.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) - - # --- Case 8: EKS first pass with a node pool backed by a Capacity - # Block. The reservation ID flows through to the EKSCluster node pool's - # capacityBlock, which compose-eks-cluster turns into a CAPACITY_BLOCK - # node group. --- - req8 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", + "provisioning": { + "provider": "EKS", + "eks": { + "instanceType": "g6.xlarge", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + }, + } + class_selector_eks = fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceClass", + match_name="gpu-l4-eks", + ) + + req4 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="EKS", + eks=v1alpha1.Eks(region="us-west-2"), ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="EKS", - eks=v1alpha1.Eks(region="us-west-2"), + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a", "us-west-2b"], ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4-eks", - nodeCount=2, - maxNodeCount=4, - zones=["us-west-2a"], - capacityBlock=v1alpha1.CapacityBlock( - capacityReservationId="cr-0123456789abcdef0", - ), - ), - ], - ), - ).model_dump(exclude_none=True, mode="json"), - ), + ], + ), + ).model_dump(exclude_none=True, mode="json"), ), ), - ) - req8.required_resources["class-gpu-l4-eks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), - ) + ), + ) + req4.required_resources["class-gpu-l4-eks"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), + ) - want8 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + want4 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + }, + }, + ), + ), + resources={ + "eks-cluster": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "EKSCluster", + "metadata": { + "name": "test-cluster", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "region": "us-west-2", + "kubernetesVersion": "1.36", + "nodePools": [ { "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "role": "GPU", + "instanceType": "g6.xlarge", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": { + "acceleratorType": "nvidia-l4", + }, + "zones": ["us-west-2a", "us-west-2b"], }, ], }, }, ), ), - resources={ - "eks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-west-2", - "kubernetesVersion": "1.36", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - }, - "zones": ["us-west-2a"], - "capacityBlock": { - "capacityReservationId": "cr-0123456789abcdef0", - }, - }, - ], - }, - }, + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want4.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) + + # --- Case 8: EKS first pass with a node pool backed by a Capacity + # Block. The reservation ID flows through to the EKSCluster node pool's + # capacityBlock, which compose-eks-cluster turns into a CAPACITY_BLOCK + # node group. --- + req8 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want8.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) - - # --- Case 9: EKS first pass with a node pool that opts into the EFA - # fabric. fabric.type flows through to the EKSCluster node pool, which - # compose-eks-cluster turns into EFA launch-template interfaces. --- - req9 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="EKS", + eks=v1alpha1.Eks(region="us-west-2"), ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="EKS", - eks=v1alpha1.Eks(region="us-west-2"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4-eks", - nodeCount=2, - maxNodeCount=4, - zones=["us-west-2a"], - fabric=v1alpha1.Fabric(type="EFA"), + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a"], + capacityBlock=v1alpha1.CapacityBlock( + capacityReservationId="cr-0123456789abcdef0", ), - ], - ), - ).model_dump(exclude_none=True, mode="json"), - ), + ), + ], + ), + ).model_dump(exclude_none=True, mode="json"), ), ), - ) - req9.required_resources["class-gpu-l4-eks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), - ) + ), + ) + req8.required_resources["class-gpu-l4-eks"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), + ) - want9 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + want8 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + }, + }, + ), + ), + resources={ + "eks-cluster": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "EKSCluster", + "metadata": { + "name": "test-cluster", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "region": "us-west-2", + "kubernetesVersion": "1.36", + "nodePools": [ { "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "role": "GPU", + "instanceType": "g6.xlarge", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": { + "acceleratorType": "nvidia-l4", + }, + "zones": ["us-west-2a"], + "capacityBlock": { + "capacityReservationId": "cr-0123456789abcdef0", + }, }, ], }, }, ), ), - resources={ - "eks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-west-2", - "kubernetesVersion": "1.36", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - }, - "zones": ["us-west-2a"], - "fabric": "EFA", - }, - ], - }, - }, - ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want9.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) - - # --- Case 5: EKS cluster not yet ready (no kubeconfig observed) but a - # ClusterProviderConfig already exists from a prior reconcile. The CPC - # is built only from the kubeconfig, so without one it's simply omitted - # from desired state this reconcile (and recreated once the kubeconfig - # is observed again) - it is never emitted with an empty secretRef. - observed_cpc = { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, }, - } - req5 = fnv1.RunFunctionRequest() - req5.CopyFrom(req4) - req5.observed.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource(resource=resource.dict_to_struct(observed_cpc)), - ) + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want8.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) - # Desired state is identical to case 4: no ClusterProviderConfig. - want5 = fnv1.RunFunctionResponse() - want5.CopyFrom(want4) - - # --- Case 6: GKE cluster ready - composes CPC, backend, usage, and the - # VPC-pinned modelplane-rwx Filestore StorageClass on the workload - # cluster (default cache storage class). --- - observed_gke_ready = { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "region": "us-central1", - "nodePools": [{"name": "system", "role": "System", "machineType": "e2-standard-4"}], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2026-06-08T00:00:00Z", - }, - ], - # The backing GKECluster reports its effective RWX StorageClass; - # the InferenceCluster relays it up to its own status.cache. - "cache": {"storageClassName": "modelplane-rwx"}, - "secrets": [ - {"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}, - { - "type": "GoogleApplicationCredentials", - "name": "test-cluster-sa-key-fghij", - "key": "credentials.json", - }, - ], - }, - } - req6 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", + # --- Case 9: EKS first pass with a node pool that opts into the EFA + # fabric. fabric.type flows through to the EKSCluster node pool, which + # compose-eks-cluster turns into EFA launch-template interfaces. --- + req9 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="EKS", + eks=v1alpha1.Eks(region="us-west-2"), ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="GKE", - gke=v1alpha1.Gke( - region="us-central1", - ), + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a"], + fabric=v1alpha1.Fabric(type="EFA"), ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - zones=["us-central1-a"], - ), - ], - ), - ).model_dump(exclude_none=True, mode="json") - ), + ], + ), + ).model_dump(exclude_none=True, mode="json"), ), - resources={ - "gke-cluster": fnv1.Resource(resource=resource.dict_to_struct(observed_gke_ready)), - }, ), - ) - req6.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) + ), + ) + req9.required_resources["class-gpu-l4-eks"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), + ) - want6 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + want9 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + }, + }, + ), + ), + resources={ + "eks-cluster": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "EKSCluster", + "metadata": { + "name": "test-cluster", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "region": "us-west-2", + "kubernetesVersion": "1.36", + "nodePools": [ { "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - } - ], + "role": "GPU", + "instanceType": "g6.xlarge", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": { + "acceleratorType": "nvidia-l4", + }, + "zones": ["us-west-2a"], + "fabric": "EFA", }, ], - # Relayed from the backing GKECluster's status.cache. - "cache": {"storageClassName": "modelplane-rwx"}, }, - } + }, ), ), - resources={ - "gke-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-central1", - "kubernetesVersion": "1.35", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "machineType": "g2-standard-48", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - "acceleratorCount": 1, - }, - "zones": ["us-central1-a"], - }, - ], - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "cluster-provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, - "identity": { - "type": "GoogleApplicationCredentials", - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-sa-key-fghij", - "key": "credentials.json", - }, - }, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "GKE", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - { - "type": "GoogleApplicationCredentials", - "name": "test-cluster-sa-key-fghij", - "key": "credentials.json", - }, - ], - }, - } - ), - ), - "usage-gke-by-backend": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } - ), - ready=fnv1.READY_TRUE, - ), + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want9.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) + + # --- Case 5: EKS cluster not yet ready (no kubeconfig observed) but a + # ClusterProviderConfig already exists from a prior reconcile. The CPC + # is built only from the kubeconfig, so without one it's simply omitted + # from desired state this reconcile (and recreated once the kubeconfig + # is observed again) - it is never emitted with an empty secretRef. + observed_cpc = { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + }, + }, + } + req5 = fnv1.RunFunctionRequest() + req5.CopyFrom(req4) + req5.observed.resources["cluster-provider-config-kubernetes"].CopyFrom( + fnv1.Resource(resource=resource.dict_to_struct(observed_cpc)), + ) + + # Desired state is identical to case 4: no ClusterProviderConfig. + want5 = fnv1.RunFunctionResponse() + want5.CopyFrom(want4) + + # --- Case 6: GKE cluster ready - composes CPC, backend, usage, and the + # VPC-pinned modelplane-rwx Filestore StorageClass on the workload + # cluster (default cache storage class). --- + observed_gke_ready = { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "us-central1", + "nodePools": [{"name": "system", "role": "System", "machineType": "e2-standard-4"}], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="GKE cluster ready, composing backend", - ), + # The backing GKECluster reports its effective RWX StorageClass; + # the InferenceCluster relays it up to its own status.cache. + "cache": {"storageClassName": "modelplane-rwx"}, + "secrets": [ + {"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}, + { + "type": "GoogleApplicationCredentials", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", + }, ], - context=structpb.Struct(), - ) - want6.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - - # --- Case 7: EKS cluster ready - kubeconfig observed on the EKSCluster - # status. The function wires the ClusterProviderConfig, composes the - # ServingStack backend, and emits the Usage that blocks EKSCluster - # deletion until the ServingStack is gone. --- - req7 = fnv1.RunFunctionRequest() - req7.CopyFrom(req4) - req7.observed.resources["eks-cluster"].CopyFrom( - fnv1.Resource( + }, + } + req6 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "region": "us-west-2", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - }, - ], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="GKE", + gke=v1alpha1.Gke( + region="us-central1", + ), + ), + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], + ), ], - # The backing EKSCluster reports its effective RWX - # StorageClass; the InferenceCluster relays it up. - "cache": {"storageClassName": "modelplane-rwx-efs"}, - }, - } + ), + ).model_dump(exclude_none=True, mode="json") ), ), - ) + resources={ + "gke-cluster": fnv1.Resource(resource=resource.dict_to_struct(observed_gke_ready)), + }, + ), + ) + req6.required_resources["class-gpu-l4"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) + ) - want7 = fnv1.RunFunctionResponse() - want7.CopyFrom(want4) - # Mark the EKSCluster ready and relay its status.cache up to status.cache. - _eks_ready_extras(want7, "modelplane-rwx-efs") - want7.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( + want6 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", }, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - ) - want7.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", "namespace": "modelplane-system", - }, - "spec": { - "cloud": "EKS", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ + "gpuPools": [ { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + } + ], }, ], + # Relayed from the backing GKECluster's status.cache. + "cache": {"storageClassName": "modelplane-rwx"}, }, } ), ), - ) - want7.desired.resources["usage-eks-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - ) - del want7.conditions[:] - want7.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want7.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="EKS cluster ready, composing backend", - ) - ) - - # --- Case 10: Nebius first pass composes the NebiusCluster XR only. - # The pool's InfiniBand fabric flows through to the NebiusCluster - # pool's fabric, and minNodeCount stays unset so the pool's - # autoscaling floor defaults to its node count downstream. --- - inference_class_h100_nebius = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-h100-nebius"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, - }, - ], - "provisioning": { - "provider": "Nebius", - "nebius": { - "platform": "gpu-h100-sxm", - "preset": "8gpu-128vcpu-1600gb", - "diskSizeGb": 200, - "accelerator": {"type": "nvidia-h100", "count": 8}, - }, - }, - }, - } - class_selector_nebius = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-h100-nebius", - ) - - req10 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( + resources={ + "gke-cluster": fnv1.Resource( resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Nebius", - nebius=v1alpha1.Nebius(), - ), - nodePools=[ - v1alpha1.NodePool( - name="h100-pool", - className="gpu-h100-nebius", - nodeCount=2, - maxNodeCount=4, - fabric=v1alpha1.Fabric( - type="InfiniBand", - infiniband=v1alpha1.Infiniband(fabric="fabric-2"), - ), - ), + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "metadata": { + "name": "test-cluster", + "namespace": "modelplane-system", + }, + "spec": { + "region": "us-central1", + "kubernetesVersion": "1.35", + "nodePools": [ + { + "name": "l4-pool", + "role": "GPU", + "machineType": "g2-standard-48", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": { + "acceleratorType": "nvidia-l4", + "acceleratorCount": 1, + }, + "zones": ["us-central1-a"], + }, ], - ), - ).model_dump(exclude_none=True, mode="json"), + }, + } ), + ready=fnv1.READY_TRUE, ), - ), - ) - req10.required_resources["class-gpu-h100-nebius"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_nebius)), - ) - - want10 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + "cluster-provider-config-kubernetes": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + }, + "identity": { + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", + }, }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "serving-stack": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "cloud": "GKE", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ { - "name": "h100-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, - }, - ], + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + { + "type": "GoogleApplicationCredentials", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", }, ], }, - }, + } ), ), - resources={ - "nebius-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "NebiusCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "kubernetesVersion": "1.34", - "nodePools": [ - { - "name": "h100-pool", - "role": "GPU", - "platform": "gpu-h100-sxm", - "preset": "8gpu-128vcpu-1600gb", - "diskSizeGb": 200, - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-h100", - "driversPreset": "cuda13.0", - }, - "fabric": { - "type": "InfiniBand", - "infiniband": {"fabric": "fabric-2"}, - }, - }, - ], - }, + "usage-gke-by-backend": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "of": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "resourceSelector": {"matchControllerRef": True}, + }, + "by": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "resourceSelector": {"matchControllerRef": True}, + }, + "replayDeletion": True, }, - ), + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want10.requirements.resources["class-gpu-h100-nebius"].CopyFrom(class_selector_nebius) - - # --- Case 11: Nebius cluster ready - kubeconfig and service account - # credentials observed on the NebiusCluster status. The function wires - # the ClusterProviderConfig with the Nebius identity (the mk8s - # kubeconfig has no embedded credentials), composes the ServingStack - # backend with both secrets, and emits the Usage that blocks - # NebiusCluster deletion until the ServingStack is gone. --- - req11 = fnv1.RunFunctionRequest() - req11.CopyFrom(req10) - req11.observed.resources["nebius-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "NebiusCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "nodePools": [ - { - "name": "h100-pool", - "role": "GPU", - "platform": "gpu-h100-sxm", - "preset": "8gpu-128vcpu-1600gb", - "nodeCount": 2, - }, - ], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - # The credential entry carries a namespace: it - # is the Nebius ClusterProviderConfig's Secret, - # which lives outside modelplane-system. - { - "type": "NebiusServiceAccountCredentials", - "name": "nebius-credentials", - "key": "credentials.json", - "namespace": "crossplane-system", - }, - ], - }, - } + ready=fnv1.READY_TRUE, ), + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Installing", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="GKE cluster ready, composing backend", ), - ) - - want11 = fnv1.RunFunctionResponse() - want11.CopyFrom(want10) - want11.desired.resources["nebius-cluster"].ready = fnv1.READY_TRUE - want11.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, + ], + context=structpb.Struct(), + ) + want6.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) + + # --- Case 7: EKS cluster ready - kubeconfig observed on the EKSCluster + # status. The function wires the ClusterProviderConfig, composes the + # ServingStack backend, and emits the Usage that blocks EKSCluster + # deletion until the ServingStack is gone. --- + req7 = fnv1.RunFunctionRequest() + req7.CopyFrom(req4) + req7.observed.resources["eks-cluster"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "EKSCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "us-west-2", + "nodePools": [ + { + "name": "l4-pool", + "role": "GPU", + "instanceType": "g6.xlarge", + "nodeCount": 2, }, - "identity": { - "type": "NebiusServiceAccountCredentials", - "source": "Secret", - "secretRef": { - "namespace": "crossplane-system", - "name": "nebius-credentials", - "key": "credentials.json", - }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + # The backing EKSCluster reports its effective RWX + # StorageClass; the InferenceCluster relays it up. + "cache": {"storageClassName": "modelplane-rwx-efs"}, + }, + } + ), + ), + ) + + want7 = fnv1.RunFunctionResponse() + want7.CopyFrom(want4) + # Mark the EKSCluster ready and relay its status.cache up to status.cache. + _eks_ready_extras(want7, "modelplane-rwx-efs") + want7.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", }, }, - } - ), - ready=fnv1.READY_TRUE, + }, + } ), - ) - want11.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", + ready=fnv1.READY_TRUE, + ), + ) + want7.desired.resources["serving-stack"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", + "namespace": "modelplane-system", + }, + "spec": { + "cloud": "EKS", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + }, + } + ), + ), + ) + want7.desired.resources["usage-eks-by-backend"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "of": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "EKSCluster", + "resourceSelector": {"matchControllerRef": True}, }, - "spec": { - "cloud": "Nebius", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - { - "type": "NebiusServiceAccountCredentials", - "name": "nebius-credentials", - "key": "credentials.json", - "namespace": "crossplane-system", - }, - ], + "by": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "resourceSelector": {"matchControllerRef": True}, }, - } - ), + "replayDeletion": True, + }, + } ), - ) - want11.desired.resources["usage-nebius-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "NebiusCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } - ), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_TRUE, + ), + ) + del want7.conditions[:] + want7.conditions.extend( + [ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Installing", ), + ] + ) + want7.results.append( + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="EKS cluster ready, composing backend", ) - del want11.conditions[:] - want11.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want11.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Nebius cluster ready, composing backend", - ) - ) + ) - # --- Case 12: AKS first pass composes the AKSCluster XR only. The - # pool's InfiniBand fabric flows through to the AKSCluster pool as the - # plain fabric string - Azure has no user-selectable fabric ID. --- - inference_class_h100_aks = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-h100-aks"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, - }, - ], - "provisioning": { - "provider": "AKS", - "aks": { - "vmSize": "Standard_ND96isr_H100_v5", - "diskSizeGb": 200, - "accelerator": {"type": "nvidia-h100", "count": 8}, - }, + # --- Case 10: Nebius first pass composes the NebiusCluster XR only. + # The pool's InfiniBand fabric flows through to the NebiusCluster + # pool's fabric, and minNodeCount stays unset so the pool's + # autoscaling floor defaults to its node count downstream. --- + inference_class_h100_nebius = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceClass", + "metadata": {"name": "gpu-h100-nebius"}, + "spec": { + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], + "provisioning": { + "provider": "Nebius", + "nebius": { + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "diskSizeGb": 200, + "accelerator": {"type": "nvidia-h100", "count": 8}, }, }, - } - class_selector_aks = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-h100-aks", - ) + }, + } + class_selector_nebius = fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceClass", + match_name="gpu-h100-nebius", + ) - req12 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", + req10 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Nebius", + nebius=v1alpha1.Nebius(), ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="AKS", - aks=v1alpha1.Aks(location="westeurope"), - ), - nodePools=[ - v1alpha1.NodePool( - name="h100pool", - className="gpu-h100-aks", - nodeCount=2, - # AKS GPU pools keep minNodeCount at - # 1: the AKS autoscaler can't scale a - # DRA pool up from zero nodes. - minNodeCount=1, - maxNodeCount=4, - fabric=v1alpha1.Fabric(type="InfiniBand"), + nodePools=[ + v1alpha1.NodePool( + name="h100-pool", + className="gpu-h100-nebius", + nodeCount=2, + maxNodeCount=4, + fabric=v1alpha1.Fabric( + type="InfiniBand", + infiniband=v1alpha1.Infiniband(fabric="fabric-2"), ), - ], - ), - ).model_dump(exclude_none=True, mode="json"), - ), + ), + ], + ), + ).model_dump(exclude_none=True, mode="json"), ), ), - ) - req12.required_resources["class-gpu-h100-aks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_aks)), - ) + ), + ) + req10.required_resources["class-gpu-h100-nebius"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_nebius)), + ) - want12 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + want10 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "h100-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], + }, + ], + }, + }, + ), + ), + resources={ + "nebius-cluster": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "NebiusCluster", + "metadata": { + "name": "test-cluster", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "kubernetesVersion": "1.34", + "nodePools": [ { - "name": "h100pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, - }, - ], + "name": "h100-pool", + "role": "GPU", + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "diskSizeGb": 200, + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": { + "acceleratorType": "nvidia-h100", + "driversPreset": "cuda13.0", + }, + "fabric": { + "type": "InfiniBand", + "infiniband": {"fabric": "fabric-2"}, + }, }, ], }, }, ), ), - resources={ - "aks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want10.requirements.resources["class-gpu-h100-nebius"].CopyFrom(class_selector_nebius) + + # --- Case 11: Nebius cluster ready - kubeconfig and service account + # credentials observed on the NebiusCluster status. The function wires + # the ClusterProviderConfig with the Nebius identity (the mk8s + # kubeconfig has no embedded credentials), composes the ServingStack + # backend with both secrets, and emits the Usage that blocks + # NebiusCluster deletion until the ServingStack is gone. --- + req11 = fnv1.RunFunctionRequest() + req11.CopyFrom(req10) + req11.observed.resources["nebius-cluster"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "NebiusCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "nodePools": [ { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "AKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "location": "westeurope", - "kubernetesVersion": "1.34", - "nodePools": [ - { - "name": "h100pool", - "role": "GPU", - "vmSize": "Standard_ND96isr_H100_v5", - "diskSizeGb": 200, - "nodeCount": 2, - "minNodeCount": 1, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-h100", - }, - "fabric": "InfiniBand", - }, - ], - }, + "name": "h100-pool", + "role": "GPU", + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "nodeCount": 2, }, - ), - ), - }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + # The credential entry carries a namespace: it + # is the Nebius ClusterProviderConfig's Secret, + # which lives outside modelplane-system. + { + "type": "NebiusServiceAccountCredentials", + "name": "nebius-credentials", + "key": "credentials.json", + "namespace": "crossplane-system", + }, + ], + }, + } ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want12.requirements.resources["class-gpu-h100-aks"].CopyFrom(class_selector_aks) - - # --- Case 13: AKS cluster ready - kubeconfig observed on the - # AKSCluster status. The kubeconfig embeds a client certificate, so - # the ClusterProviderConfig carries no identity (unlike GKE/Nebius). - # The function composes the ServingStack backend and the Usage that - # blocks AKSCluster deletion until the ServingStack is gone, and - # relays the AKSCluster's status.cache up to status.cache. --- - req13 = fnv1.RunFunctionRequest() - req13.CopyFrom(req12) - req13.observed.resources["aks-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "AKSCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "location": "westeurope", - "nodePools": [ - { - "name": "h100pool", - "role": "GPU", - "vmSize": "Standard_ND96isr_H100_v5", - "nodeCount": 2, - }, - ], + ), + ) + + want11 = fnv1.RunFunctionResponse() + want11.CopyFrom(want10) + want11.desired.resources["nebius-cluster"].ready = fnv1.READY_TRUE + want11.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - ], - # The backing AKSCluster reports its effective RWX - # StorageClass; the InferenceCluster relays it up. - "cache": {"storageClassName": "modelplane-rwx-fs"}, + "identity": { + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": { + "namespace": "crossplane-system", + "name": "nebius-credentials", + "key": "credentials.json", + }, }, - } - ), + }, + } ), + ready=fnv1.READY_TRUE, + ), + ) + want11.desired.resources["serving-stack"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", + "namespace": "modelplane-system", + }, + "spec": { + "cloud": "Nebius", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + { + "type": "NebiusServiceAccountCredentials", + "name": "nebius-credentials", + "key": "credentials.json", + "namespace": "crossplane-system", + }, + ], + }, + } + ), + ), + ) + want11.desired.resources["usage-nebius-by-backend"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "of": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "NebiusCluster", + "resourceSelector": {"matchControllerRef": True}, + }, + "by": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "resourceSelector": {"matchControllerRef": True}, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + ) + del want11.conditions[:] + want11.conditions.extend( + [ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Installing", + ), + ] + ) + want11.results.append( + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Nebius cluster ready, composing backend", ) + ) - want13 = fnv1.RunFunctionResponse() - want13.CopyFrom(want12) - want13.desired.resources["aks-cluster"].ready = fnv1.READY_TRUE - status13 = want13.desired.composite.resource.fields["status"].struct_value - status13.fields["cache"].struct_value.fields["storageClassName"].string_value = "modelplane-rwx-fs" - want13.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( + # --- Case 12: AKS first pass composes the AKSCluster XR only. The + # pool's InfiniBand fabric flows through to the AKSCluster pool as the + # plain fabric string - Azure has no user-selectable fabric ID. --- + inference_class_h100_aks = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceClass", + "metadata": {"name": "gpu-h100-aks"}, + "spec": { + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], + "provisioning": { + "provider": "AKS", + "aks": { + "vmSize": "Standard_ND96isr_H100_v5", + "diskSizeGb": 200, + "accelerator": {"type": "nvidia-h100", "count": 8}, + }, + }, + }, + } + class_selector_aks = fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceClass", + match_name="gpu-h100-aks", + ) + + req12 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, - }, - } + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="AKS", + aks=v1alpha1.Aks(location="westeurope"), + ), + nodePools=[ + v1alpha1.NodePool( + name="h100pool", + className="gpu-h100-aks", + nodeCount=2, + # AKS GPU pools keep minNodeCount at + # 1: the AKS autoscaler can't scale a + # DRA pool up from zero nodes. + minNodeCount=1, + maxNodeCount=4, + fabric=v1alpha1.Fabric(type="InfiniBand"), + ), + ], + ), + ).model_dump(exclude_none=True, mode="json"), ), - ready=fnv1.READY_TRUE, ), - ) - want13.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( + ), + ) + req12.required_resources["class-gpu-h100-aks"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_aks)), + ) + + want12 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, "namespace": "modelplane-system", - }, - "spec": { - "cloud": "AKS", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ + "gpuPools": [ { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + "name": "h100pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], }, ], }, - } - ), - ), - ) - want13.desired.resources["usage-aks-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "AKSCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - ) - del want13.conditions[:] - want13.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want13.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="AKS cluster ready, composing backend", - ) - ) - - # --- Case 14: Vultr first pass composes the VultrCluster XR only. - # minNodeCount stays unset so the pool's autoscaling floor defaults - # to its node count downstream. --- - inference_class_l40s_vultr = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-l40s-vultr"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, - }, - ], - "provisioning": { - "provider": "Vultr", - "vultr": { - "plan": "vcg-l40s-16c-180g-48vram", - "accelerator": {"type": "nvidia-l40s", "count": 1}, }, - }, - }, - } - class_selector_vultr = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-l40s-vultr", - ) - - req14 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Vultr", - vultr=v1alpha1.Vultr(region="ewr"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l40s-pool", - className="gpu-l40s-vultr", - nodeCount=2, - maxNodeCount=4, - ), - ], - ), - ).model_dump(exclude_none=True, mode="json"), - ), ), ), - ) - req14.required_resources["class-gpu-l40s-vultr"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l40s_vultr)), - ) - - want14 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + resources={ + "aks-cluster": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "AKSCluster", + "metadata": { + "name": "test-cluster", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "location": "westeurope", + "kubernetesVersion": "1.34", + "nodePools": [ { - "name": "l40s-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, - }, - ], + "name": "h100pool", + "role": "GPU", + "vmSize": "Standard_ND96isr_H100_v5", + "diskSizeGb": 200, + "nodeCount": 2, + "minNodeCount": 1, + "maxNodeCount": 4, + "gpu": { + "acceleratorType": "nvidia-h100", + }, + "fabric": "InfiniBand", }, ], }, }, ), ), - resources={ - "vultr-cluster": fnv1.Resource( - resource=resource.dict_to_struct( + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want12.requirements.resources["class-gpu-h100-aks"].CopyFrom(class_selector_aks) + + # --- Case 13: AKS cluster ready - kubeconfig observed on the + # AKSCluster status. The kubeconfig embeds a client certificate, so + # the ClusterProviderConfig carries no identity (unlike GKE/Nebius). + # The function composes the ServingStack backend and the Usage that + # blocks AKSCluster deletion until the ServingStack is gone, and + # relays the AKSCluster's status.cache up to status.cache. --- + req13 = fnv1.RunFunctionRequest() + req13.CopyFrom(req12) + req13.observed.resources["aks-cluster"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "AKSCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "location": "westeurope", + "nodePools": [ { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "ewr", - "kubernetesVersion": "v1.36.2+1", - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "plan": "vcg-l40s-16c-180g-48vram", - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-l40s", - }, - }, - ], - }, + "name": "h100pool", + "role": "GPU", + "vmSize": "Standard_ND96isr_H100_v5", + "nodeCount": 2, }, - ), - ), - }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + # The backing AKSCluster reports its effective RWX + # StorageClass; the InferenceCluster relays it up. + "cache": {"storageClassName": "modelplane-rwx-fs"}, + }, + } ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), + ), + ) + + want13 = fnv1.RunFunctionResponse() + want13.CopyFrom(want12) + want13.desired.resources["aks-cluster"].ready = fnv1.READY_TRUE + status13 = want13.desired.composite.resource.fields["status"].struct_value + status13.fields["cache"].struct_value.fields["storageClassName"].string_value = "modelplane-rwx-fs" + want13.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + ) + want13.desired.resources["serving-stack"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", + "namespace": "modelplane-system", + }, + "spec": { + "cloud": "AKS", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + }, + } + ), + ), + ) + want13.desired.resources["usage-aks-by-backend"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "of": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "AKSCluster", + "resourceSelector": {"matchControllerRef": True}, + }, + "by": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "resourceSelector": {"matchControllerRef": True}, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + ) + del want13.conditions[:] + want13.conditions.extend( + [ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Installing", + ), + ] + ) + want13.results.append( + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="AKS cluster ready, composing backend", ) - want14.requirements.resources["class-gpu-l40s-vultr"].CopyFrom(class_selector_vultr) + ) - # --- Case 14b: Vultr credentials pass through to the VultrCluster - # spec, mirroring the GKE/EKS/AKS passthrough. --- - req_creds_vultr = fnv1.RunFunctionRequest() - req_creds_vultr.CopyFrom(req14) - req_creds_vultr.observed.composite.CopyFrom( - fnv1.Resource( + # --- Case 14: Vultr first pass composes the VultrCluster XR only. + # minNodeCount stays unset so the pool's autoscaling floor defaults + # to its node count downstream. --- + inference_class_l40s_vultr = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceClass", + "metadata": {"name": "gpu-l40s-vultr"}, + "spec": { + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, + }, + ], + "provisioning": { + "provider": "Vultr", + "vultr": { + "plan": "vcg-l40s-16c-180g-48vram", + "accelerator": {"type": "nvidia-l40s", "count": 1}, + }, + }, + }, + } + class_selector_vultr = fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceClass", + match_name="gpu-l40s-vultr", + ) + + req14 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( v1alpha1.InferenceCluster( metadata=metav1.ObjectMeta( @@ -2537,13 +2413,7 @@ async def test_compose(self) -> None: # noqa: PLR0915 spec=v1alpha1.Spec( cluster=v1alpha1.Cluster( source="Vultr", - vultr=v1alpha1.Vultr( - region="ewr", - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="my-vultr-account", - ), - ), + vultr=v1alpha1.Vultr(region="ewr"), ), nodePools=[ v1alpha1.NodePool( @@ -2557,99 +2427,16 @@ async def test_compose(self) -> None: # noqa: PLR0915 ).model_dump(exclude_none=True, mode="json"), ), ), - ) - - want_creds_vultr = fnv1.RunFunctionResponse() - want_creds_vultr.CopyFrom(want14) - want_creds_vultr.desired.resources["vultr-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "ewr", - "kubernetesVersion": "v1.36.2+1", - "credentials": { - "type": "ProviderConfig", - "name": "my-vultr-account", - }, - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "plan": "vcg-l40s-16c-180g-48vram", - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-l40s", - }, - }, - ], - }, - }, - ), - ), - ) - - # --- Case 15: Vultr cluster ready - kubeconfig observed on the - # VultrCluster status. The VKE kubeconfig embeds static client - # certificates, so the ClusterProviderConfig carries no identity - # (unlike Nebius). The function composes the ServingStack backend - # with the kubeconfig and emits the Usage that blocks VultrCluster - # deletion until the ServingStack is gone. VultrCluster reports no - # cache StorageClass, so status.cache stays unset. --- - req15 = fnv1.RunFunctionRequest() - req15.CopyFrom(req14) - req15.observed.resources["vultr-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "region": "ewr", - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "plan": "vcg-l40s-16c-180g-48vram", - "nodeCount": 2, - }, - ], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - ], - }, - } - ), - ), - ) + ), + ) + req14.required_resources["class-gpu-l40s-vultr"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l40s_vultr)), + ) - want15 = fnv1.RunFunctionResponse() - want15.CopyFrom(want14) - want15.desired.resources["vultr-cluster"].ready = fnv1.READY_TRUE - want15.desired.composite.CopyFrom( - fnv1.Resource( + want14 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( { "status": { @@ -2677,250 +2464,340 @@ async def test_compose(self) -> None: # noqa: PLR0915 }, ), ), - ) - want15.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, + resources={ + "vultr-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "VultrCluster", + "metadata": { + "name": "test-cluster", + "namespace": "modelplane-system", + }, + "spec": { + "region": "ewr", + "kubernetesVersion": "v1.36.2+1", + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "plan": "vcg-l40s-16c-180g-48vram", + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": { + "acceleratorType": "nvidia-l40s", + }, + }, + ], }, }, - } + ), ), - ready=fnv1.READY_TRUE, + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want14.requirements.resources["class-gpu-l40s-vultr"].CopyFrom(class_selector_vultr) + + # --- Case 14b: Vultr credentials pass through to the VultrCluster + # spec, mirroring the GKE/EKS/AKS passthrough. --- + req_creds_vultr = fnv1.RunFunctionRequest() + req_creds_vultr.CopyFrom(req14) + req_creds_vultr.observed.composite.CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Vultr", + vultr=v1alpha1.Vultr( + region="ewr", + credentials=v1alpha1.Credentials( + type="ProviderConfig", + name="my-vultr-account", + ), + ), + ), + nodePools=[ + v1alpha1.NodePool( + name="l40s-pool", + className="gpu-l40s-vultr", + nodeCount=2, + maxNodeCount=4, + ), + ], + ), + ).model_dump(exclude_none=True, mode="json"), ), - ) - want15.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", + ), + ) + + want_creds_vultr = fnv1.RunFunctionResponse() + want_creds_vultr.CopyFrom(want14) + want_creds_vultr.desired.resources["vultr-cluster"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "VultrCluster", + "metadata": { + "name": "test-cluster", + "namespace": "modelplane-system", + }, + "spec": { + "region": "ewr", + "kubernetesVersion": "v1.36.2+1", + "credentials": { + "type": "ProviderConfig", + "name": "my-vultr-account", }, - "spec": { - "cloud": "Vultr", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "plan": "vcg-l40s-16c-180g-48vram", + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": { + "acceleratorType": "nvidia-l40s", }, - ], - }, - } - ), - ), - ) - want15.desired.resources["usage-vultr-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } - ), - ready=fnv1.READY_TRUE, + }, + ], + }, + }, ), - ) - del want15.conditions[:] - want15.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want15.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Vultr cluster ready, composing backend", - ) - ) + ), + ) - # --- Case 16: Civo first pass composes the CivoCluster XR only. - # minNodeCount stays unset so the pool's autoscaling floor defaults - # to its node count downstream. --- - inference_class_l40s_civo = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-l40s-civo"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, + # --- Case 15: Vultr cluster ready - kubeconfig observed on the + # VultrCluster status. The VKE kubeconfig embeds static client + # certificates, so the ClusterProviderConfig carries no identity + # (unlike Nebius). The function composes the ServingStack backend + # with the kubeconfig and emits the Usage that blocks VultrCluster + # deletion until the ServingStack is gone. VultrCluster reports no + # cache StorageClass, so status.cache stays unset. --- + req15 = fnv1.RunFunctionRequest() + req15.CopyFrom(req14) + req15.observed.resources["vultr-cluster"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "VultrCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "ewr", + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "plan": "vcg-l40s-16c-180g-48vram", + "nodeCount": 2, + }, + ], }, - ], - "provisioning": { - "provider": "Civo", - "civo": { - "size": "an.g1.l40s.kube.x1", - "accelerator": {"type": "nvidia-l40s", "count": 1}, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], }, - }, - }, - } - class_selector_civo = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-l40s-civo", - ) - - req16 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Civo", - civo=v1alpha1.Civo(region="LON1"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l40s-pool", - className="gpu-l40s-civo", - nodeCount=2, - maxNodeCount=4, - ), - ], - ), - ).model_dump(exclude_none=True, mode="json"), - ), - ), + } ), - ) - req16.required_resources["class-gpu-l40s-civo"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l40s_civo)), - ) + ), + ) - want16 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ + want15 = fnv1.RunFunctionResponse() + want15.CopyFrom(want14) + want15.desired.resources["vultr-cluster"].ready = fnv1.READY_TRUE + want15.desired.composite.CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l40s-pool", + "nodes": 4, + "devices": [ { - "name": "l40s-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, }, ], }, + ], + }, + }, + ), + ), + ) + want15.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, }, - ), - ), - resources={ - "civo-cluster": fnv1.Resource( - resource=resource.dict_to_struct( + }, + } + ), + ready=fnv1.READY_TRUE, + ), + ) + want15.desired.resources["serving-stack"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", + "namespace": "modelplane-system", + }, + "spec": { + "cloud": "Vultr", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "CivoCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "LON1", - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "size": "an.g1.l40s.kube.x1", - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-l40s", - }, - }, - ], - }, + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", }, - ), - ), - }, + ], + }, + } ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), + ), + ) + want15.desired.resources["usage-vultr-by-backend"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "of": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "VultrCluster", + "resourceSelector": {"matchControllerRef": True}, + }, + "by": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "resourceSelector": {"matchControllerRef": True}, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + ) + del want15.conditions[:] + want15.conditions.extend( + [ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Installing", + ), + ] + ) + want15.results.append( + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Vultr cluster ready, composing backend", ) - want16.requirements.resources["class-gpu-l40s-civo"].CopyFrom(class_selector_civo) + ) - # --- Case 16b: Civo credentials pass through to the CivoCluster - # spec, mirroring the Vultr passthrough. --- - req_creds_civo = fnv1.RunFunctionRequest() - req_creds_civo.CopyFrom(req16) - req_creds_civo.observed.composite.CopyFrom( - fnv1.Resource( + # --- Case 16: Civo first pass composes the CivoCluster XR only. + # minNodeCount stays unset so the pool's autoscaling floor defaults + # to its node count downstream. --- + inference_class_l40s_civo = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceClass", + "metadata": {"name": "gpu-l40s-civo"}, + "spec": { + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, + }, + ], + "provisioning": { + "provider": "Civo", + "civo": { + "size": "an.g1.l40s.kube.x1", + "accelerator": {"type": "nvidia-l40s", "count": 1}, + }, + }, + }, + } + class_selector_civo = fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceClass", + match_name="gpu-l40s-civo", + ) + + req16 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( v1alpha1.InferenceCluster( metadata=metav1.ObjectMeta( @@ -2930,13 +2807,7 @@ async def test_compose(self) -> None: # noqa: PLR0915 spec=v1alpha1.Spec( cluster=v1alpha1.Cluster( source="Civo", - civo=v1alpha1.Civo( - region="LON1", - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="my-civo-account", - ), - ), + civo=v1alpha1.Civo(region="LON1"), ), nodePools=[ v1alpha1.NodePool( @@ -2950,98 +2821,16 @@ async def test_compose(self) -> None: # noqa: PLR0915 ).model_dump(exclude_none=True, mode="json"), ), ), - ) - - want_creds_civo = fnv1.RunFunctionResponse() - want_creds_civo.CopyFrom(want16) - want_creds_civo.desired.resources["civo-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "CivoCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "LON1", - "credentials": { - "type": "ProviderConfig", - "name": "my-civo-account", - }, - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "size": "an.g1.l40s.kube.x1", - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-l40s", - }, - }, - ], - }, - }, - ), - ), - ) - - # --- Case 17: Civo cluster ready - kubeconfig observed on the - # CivoCluster status. The Civo kubeconfig embeds static client - # certificates, so the ClusterProviderConfig carries no identity - # (unlike Nebius). The function composes the ServingStack backend - # with the kubeconfig and emits the Usage that blocks CivoCluster - # deletion until the ServingStack is gone. CivoCluster reports no - # cache StorageClass, so status.cache stays unset. --- - req17 = fnv1.RunFunctionRequest() - req17.CopyFrom(req16) - req17.observed.resources["civo-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "CivoCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "region": "LON1", - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "size": "an.g1.l40s.kube.x1", - "nodeCount": 2, - }, - ], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - ], - }, - } - ), - ), - ) + ), + ) + req16.required_resources["class-gpu-l40s-civo"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l40s_civo)), + ) - want17 = fnv1.RunFunctionResponse() - want17.CopyFrom(want16) - want17.desired.resources["civo-cluster"].ready = fnv1.READY_TRUE - want17.desired.composite.CopyFrom( - fnv1.Resource( + want16 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( { "status": { @@ -3069,136 +2858,509 @@ async def test_compose(self) -> None: # noqa: PLR0915 }, ), ), - ) - want17.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, + resources={ + "civo-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "CivoCluster", + "metadata": { + "name": "test-cluster", + "namespace": "modelplane-system", + }, + "spec": { + "region": "LON1", + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "size": "an.g1.l40s.kube.x1", + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": { + "acceleratorType": "nvidia-l40s", + }, + }, + ], }, }, - } + ), ), - ready=fnv1.READY_TRUE, + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want16.requirements.resources["class-gpu-l40s-civo"].CopyFrom(class_selector_civo) + + # --- Case 16b: Civo credentials pass through to the CivoCluster + # spec, mirroring the Vultr passthrough. --- + req_creds_civo = fnv1.RunFunctionRequest() + req_creds_civo.CopyFrom(req16) + req_creds_civo.observed.composite.CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Civo", + civo=v1alpha1.Civo( + region="LON1", + credentials=v1alpha1.Credentials( + type="ProviderConfig", + name="my-civo-account", + ), + ), + ), + nodePools=[ + v1alpha1.NodePool( + name="l40s-pool", + className="gpu-l40s-civo", + nodeCount=2, + maxNodeCount=4, + ), + ], + ), + ).model_dump(exclude_none=True, mode="json"), ), - ) - want17.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", + ), + ) + + want_creds_civo = fnv1.RunFunctionResponse() + want_creds_civo.CopyFrom(want16) + want_creds_civo.desired.resources["civo-cluster"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "CivoCluster", + "metadata": { + "name": "test-cluster", + "namespace": "modelplane-system", + }, + "spec": { + "region": "LON1", + "credentials": { + "type": "ProviderConfig", + "name": "my-civo-account", + }, + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "size": "an.g1.l40s.kube.x1", + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": { + "acceleratorType": "nvidia-l40s", + }, + }, + ], + }, + }, + ), + ), + ) + + # --- Case 17: Civo cluster ready - kubeconfig observed on the + # CivoCluster status. The Civo kubeconfig embeds static client + # certificates, so the ClusterProviderConfig carries no identity + # (unlike Nebius). The function composes the ServingStack backend + # with the kubeconfig and emits the Usage that blocks CivoCluster + # deletion until the ServingStack is gone. CivoCluster reports no + # cache StorageClass, so status.cache stays unset. --- + req17 = fnv1.RunFunctionRequest() + req17.CopyFrom(req16) + req17.observed.resources["civo-cluster"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "CivoCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "LON1", + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "size": "an.g1.l40s.kube.x1", + "nodeCount": 2, + }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + }, + } + ), + ), + ) + + want17 = fnv1.RunFunctionResponse() + want17.CopyFrom(want16) + want17.desired.resources["civo-cluster"].ready = fnv1.READY_TRUE + want17.desired.composite.CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l40s-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, + }, + ], + }, + ], + }, + }, + ), + ), + ) + want17.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + ) + want17.desired.resources["serving-stack"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", + "namespace": "modelplane-system", + }, + "spec": { + "cloud": "Civo", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + }, + } + ), + ), + ) + want17.desired.resources["usage-civo-by-backend"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "of": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "CivoCluster", + "resourceSelector": {"matchControllerRef": True}, }, - "spec": { - "cloud": "Civo", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - ], + "by": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "resourceSelector": {"matchControllerRef": True}, }, - } - ), + "replayDeletion": True, + }, + } ), - ) - want17.desired.resources["usage-civo-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "CivoCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } - ), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_TRUE, + ), + ) + del want17.conditions[:] + want17.conditions.extend( + [ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Installing", ), + ] + ) + want17.results.append( + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Civo cluster ready, composing backend", ) - del want17.conditions[:] - want17.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want17.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Civo cluster ready, composing backend", - ) - ) + ) - # --- Case 17b: a Civo pool of single-H100 nodes (accelerator - # nvidia-h100, count 1) projects per-pool NVLink disable into the - # backend's spec.gpu: a lone H100 SXM has NVLink links but no peer, - # so the serving stack must load that pool's driver with - # NVreg_NvLinkDisable=1. The L40S case above projects no spec.gpu. --- - inference_class_h100_civo = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-h100-civo"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "81559Mi"}}, + # --- Case 17b: a Civo pool of single-H100 nodes (accelerator + # nvidia-h100, count 1) projects per-pool NVLink disable into the + # backend's spec.gpu: a lone H100 SXM has NVLink links but no peer, + # so the serving stack must load that pool's driver with + # NVreg_NvLinkDisable=1. The L40S case above projects no spec.gpu. --- + inference_class_h100_civo = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceClass", + "metadata": {"name": "gpu-h100-civo"}, + "spec": { + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], + "provisioning": { + "provider": "Civo", + "civo": { + "size": "an.g1.h100.kube.x1", + "accelerator": {"type": "nvidia-h100", "count": 1}, + }, + }, + }, + } + + req17b = fnv1.RunFunctionRequest() + req17b.CopyFrom(req17) + req17b.observed.composite.CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Civo", + civo=v1alpha1.Civo(region="LON1"), + ), + nodePools=[ + v1alpha1.NodePool( + name="h100-pool", + className="gpu-h100-civo", + nodeCount=1, + ), + ], + ), + ).model_dump(exclude_none=True, mode="json"), + ), + ), + ) + req17b.required_resources["class-gpu-h100-civo"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_civo)), + ) + + want17b = fnv1.RunFunctionResponse() + want17b.CopyFrom(want17) + del want17b.requirements.resources["class-gpu-l40s-civo"] + want17b.requirements.resources["class-gpu-h100-civo"].CopyFrom( + fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceClass", + match_name="gpu-h100-civo", + ), + ) + want17b.desired.composite.CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "h100-pool", + "nodes": 1, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], + }, + ], + }, + }, + ), + ), + ) + want17b.desired.resources["civo-cluster"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "CivoCluster", + "metadata": { + "name": "test-cluster", + "namespace": "modelplane-system", }, - ], - "provisioning": { - "provider": "Civo", - "civo": { - "size": "an.g1.h100.kube.x1", - "accelerator": {"type": "nvidia-h100", "count": 1}, + "spec": { + "region": "LON1", + "nodePools": [ + { + "name": "h100-pool", + "role": "GPU", + "size": "an.g1.h100.kube.x1", + "nodeCount": 1, + "gpu": { + "acceleratorType": "nvidia-h100", + }, + }, + ], }, }, - }, - } + ), + ready=fnv1.READY_TRUE, + ), + ) + want17b.desired.resources["serving-stack"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", + "namespace": "modelplane-system", + }, + "spec": { + "cloud": "Civo", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "gpu": { + "pools": [ + {"name": "h100-pool", "disableNvLink": True}, + ], + }, + "stack": "Standard", + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + }, + } + ), + ), + ) - req17b = fnv1.RunFunctionRequest() - req17b.CopyFrom(req17) - req17b.observed.composite.CopyFrom( - fnv1.Resource( + # Every compose path emits the ModelReplica guard requirement. + for want in ( + want1, + want2, + want3, + want4, + want5, + want6, + want7, + want8, + want9, + want10, + want11, + want12, + want13, + want14, + want_creds_vultr, + want15, + want16, + want_creds_civo, + want17, + want17b, + ): + want.requirements.resources["gateways"].CopyFrom(_gateways_selector()) + want.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) + want.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) + want.requirements.resources["model-caches"].CopyFrom(_caches_selector()) + + # The guard cases reuse case 1's request and response. + guard_cases = [ + Case( + "ModelReplicas, ModelRoutes and ModelCaches compose the guard and the mirrored namespaces", + *_guard_case(req1, want1), + ), + Case("a ModelRoute on the cluster composes the guard", *_route_guard_case(req1, want1)), + Case("a ModelCache staging onto the cluster composes the guard", *_cache_guard_case(req1, want1)), + Case("an InferenceGateway on the cluster composes the guard", *_gateway_guard_case(req1, want1)), + Case("nothing on the cluster leaves it deletable", *_unused_case(req1, want1)), + Case("guard is composed even when compose returns early", *_early_return_guard_case()), + ] + + # --- Case credentials: GKE with custom credentials passes them through to GKECluster. --- + req_creds = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( v1alpha1.InferenceCluster( metadata=metav1.ObjectMeta( @@ -3207,37 +3369,38 @@ async def test_compose(self) -> None: # noqa: PLR0915 ), spec=v1alpha1.Spec( cluster=v1alpha1.Cluster( - source="Civo", - civo=v1alpha1.Civo(region="LON1"), + source="GKE", + gke=v1alpha1.Gke( + region="us-central1", + credentials=v1alpha1.Credentials( + type="ProviderConfig", + name="my-gcp-account", + ), + ), ), nodePools=[ v1alpha1.NodePool( - name="h100-pool", - className="gpu-h100-civo", - nodeCount=1, + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], ), ], ), - ).model_dump(exclude_none=True, mode="json"), + ).model_dump(exclude_none=True, mode="json") ), ), - ) - req17b.required_resources["class-gpu-h100-civo"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_civo)), - ) + ), + ) + req_creds.required_resources["class-gpu-l4"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) + ) - want17b = fnv1.RunFunctionResponse() - want17b.CopyFrom(want17) - del want17b.requirements.resources["class-gpu-l40s-civo"] - want17b.requirements.resources["class-gpu-h100-civo"].CopyFrom( - fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-h100-civo", - ), - ) - want17b.desired.composite.CopyFrom( - fnv1.Resource( + want_creds = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( { "status": { @@ -3247,8 +3410,8 @@ async def test_compose(self) -> None: # noqa: PLR0915 "namespace": "modelplane-system", "gpuPools": [ { - "name": "h100-pool", - "nodes": 1, + "name": "l4-pool", + "nodes": 4, "devices": [ { "name": "gpu", @@ -3256,525 +3419,352 @@ async def test_compose(self) -> None: # noqa: PLR0915 "driver": "gpu.nvidia.com", "deviceClassName": "gpu.nvidia.com", "count": 1, - "capacity": {"memory": {"value": "81559Mi"}}, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, ], }, - }, - ), - ), - ) - want17b.desired.resources["civo-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "CivoCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "LON1", - "nodePools": [ - { - "name": "h100-pool", - "role": "GPU", - "size": "an.g1.h100.kube.x1", - "nodeCount": 1, - "gpu": { - "acceleratorType": "nvidia-h100", - }, - }, - ], - }, - }, - ), - ready=fnv1.READY_TRUE, - ), - ) - want17b.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Civo", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "gpu": { - "pools": [ - {"name": "h100-pool", "disableNvLink": True}, - ], - }, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - ], - }, } ), ), - ) - - # Every compose path emits the ModelReplica guard requirement. - for want in ( - want1, - want2, - want3, - want4, - want5, - want6, - want7, - want8, - want9, - want10, - want11, - want12, - want13, - want14, - want_creds_vultr, - want15, - want16, - want_creds_civo, - want17, - want17b, - ): - want.requirements.resources["gateways"].CopyFrom(_gateways_selector()) - want.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) - want.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) - want.requirements.resources["model-caches"].CopyFrom(_caches_selector()) - - # The guard cases reuse case 1's request and response. - guard_cases = [ - Case( - "ModelReplicas, ModelRoutes and ModelCaches compose the guard and the mirrored namespaces", - *_guard_case(req1, want1), - ), - Case("a ModelRoute on the cluster composes the guard", *_route_guard_case(req1, want1)), - Case("a ModelCache staging onto the cluster composes the guard", *_cache_guard_case(req1, want1)), - Case("an InferenceGateway on the cluster composes the guard", *_gateway_guard_case(req1, want1)), - Case("nothing on the cluster leaves it deletable", *_unused_case(req1, want1)), - Case("guard is composed even when compose returns early", *_early_return_guard_case()), - ] - - # --- Case credentials: GKE with custom credentials passes them through to GKECluster. --- - req_creds = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="GKE", - gke=v1alpha1.Gke( - region="us-central1", - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="my-gcp-account", - ), - ), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - zones=["us-central1-a"], - ), - ], - ), - ).model_dump(exclude_none=True, mode="json") - ), - ), - ), - ) - req_creds.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - want_creds = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + resources={ + "gke-cluster": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "metadata": { + "name": "test-cluster", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "region": "us-central1", + "kubernetesVersion": "1.35", + "credentials": { + "type": "ProviderConfig", + "name": "my-gcp-account", + }, + "nodePools": [ { "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "role": "GPU", + "machineType": "g2-standard-48", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": { + "acceleratorType": "nvidia-l4", + "acceleratorCount": 1, + }, + "zones": ["us-central1-a"], }, ], }, } ), ), - resources={ - "gke-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-central1", - "kubernetesVersion": "1.35", - "credentials": { - "type": "ProviderConfig", - "name": "my-gcp-account", - }, - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "machineType": "g2-standard-48", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - "acceleratorCount": 1, - }, - "zones": ["us-central1-a"], - }, - ], - }, - } - ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want_creds.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - want_creds.requirements.resources["gateways"].CopyFrom(_gateways_selector()) - want_creds.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) - want_creds.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) - want_creds.requirements.resources["model-caches"].CopyFrom(_caches_selector()) - - # Every cloud cluster composes an activation policy; with the policy - # observed Healthy the cluster XR is composed. - for req, want, kinds in [ - (req2, want2, fn._ACTIVATE_GCP), - (req_creds, want_creds, fn._ACTIVATE_GCP), - (req4, want4, fn._ACTIVATE_AWS), - (req5, want5, fn._ACTIVATE_AWS), - (req6, want6, fn._ACTIVATE_GCP), - (req7, want7, fn._ACTIVATE_AWS), - (req8, want8, fn._ACTIVATE_AWS), - (req9, want9, fn._ACTIVATE_AWS), - (req10, want10, fn._ACTIVATE_NEBIUS), - (req11, want11, fn._ACTIVATE_NEBIUS), - (req12, want12, fn._ACTIVATE_AZURE), - (req13, want13, fn._ACTIVATE_AZURE), - (req14, want14, fn._ACTIVATE_VULTR), - (req_creds_vultr, want_creds_vultr, fn._ACTIVATE_VULTR), - (req15, want15, fn._ACTIVATE_VULTR), - (req16, want16, fn._ACTIVATE_CIVO), - (req_creds_civo, want_creds_civo, fn._ACTIVATE_CIVO), - (req17, want17, fn._ACTIVATE_CIVO), - (req17b, want17b, fn._ACTIVATE_CIVO), - ]: - _observe_activated(req, kinds) - _want_activation(want, kinds) - - # While the policy is missing even one of the kinds from status.activated - # (e.g. a provider still installing), and with no cluster observed, the - # function composes only the activation policy (not marked ready, so the - # composite doesn't report ready), not the cluster XR. Observing the - # policy with a kind missing exercises the all-kinds check rather than - # the policy-absent branch. - req_unactivated = copy.deepcopy(req2) - _observe_activated(req_unactivated, fn._ACTIVATE_GCP[:-1]) - want_unactivated = copy.deepcopy(want2) - del want_unactivated.desired.resources["gke-cluster"] - want_unactivated.desired.resources["activation"].ClearField("ready") - - # Once the cluster is observed, the function keeps composing it even - # when the policy momentarily stops reporting the kinds active, so an - # activation blip never drops a provisioned cluster from desired state. - req_blip = copy.deepcopy(req6) - del req_blip.observed.resources["activation"] - want_blip = copy.deepcopy(want6) - - cases = [ - Case(name="existing cluster with secrets composes backend and CPC", req=req1, want=want1), - Case(name="existing cluster with a non-GCP identity threads the identity type", req=req1b, want=want1b), - Case(name="GKE cluster first pass composes GKECluster XR only", req=req2, want=want2), - Case(name="GKE credentials pass through to GKECluster spec", req=req_creds, want=want_creds), - Case(name="existing cluster second pass with backend ready", req=req3, want=want3), - Case(name="EKS cluster first pass composes EKSCluster XR only", req=req4, want=want4), - Case(name="EKS cluster not ready re-emits existing CPC unchanged", req=req5, want=want5), - Case(name="GKE cluster ready composes CPC, backend, usage, and RWX StorageClass", req=req6, want=want6), - Case(name="EKS cluster ready composes ServingStack and Usage", req=req7, want=want7), - Case( - name="EKS node pool with a Capacity Block sets capacityBlock on the EKSCluster pool", - req=req8, - want=want8, - ), - Case( - name="EKS node pool with fabric EFA sets fabric on the EKSCluster pool", - req=req9, - want=want9, - ), - Case(name="Nebius cluster first pass composes NebiusCluster XR only", req=req10, want=want10), - Case( - name="Nebius cluster ready composes CPC with Nebius identity, ServingStack, and Usage", - req=req11, - want=want11, - ), - Case(name="AKS cluster first pass composes AKSCluster XR only", req=req12, want=want12), - Case( - name="AKS cluster ready composes CPC without identity, ServingStack, and Usage", - req=req13, - want=want13, - ), - Case( - name="cloud cluster not activated composes only the policy", req=req_unactivated, want=want_unactivated - ), - Case(name="observed cluster keeps composing through an activation blip", req=req_blip, want=want_blip), - Case(name="Vultr cluster first pass composes VultrCluster XR only", req=req14, want=want14), - Case( - name="Vultr credentials pass through to VultrCluster spec", - req=req_creds_vultr, - want=want_creds_vultr, - ), - Case( - name="Vultr cluster ready composes CPC without identity, ServingStack, and Usage", - req=req15, - want=want15, - ), - Case(name="Civo cluster first pass composes CivoCluster XR only", req=req16, want=want16), - Case( - name="Civo credentials pass through to CivoCluster spec", - req=req_creds_civo, - want=want_creds_civo, - ), - Case( - name="Civo cluster ready composes CPC without identity, ServingStack, and Usage", - req=req17, - want=want17, - ), - Case( - name="Civo single-H100 pool projects per-pool NVLink disable to the backend", - req=req17b, - want=want17b, - ), - *guard_cases, - ] + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want_creds.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) + want_creds.requirements.resources["gateways"].CopyFrom(_gateways_selector()) + want_creds.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) + want_creds.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) + want_creds.requirements.resources["model-caches"].CopyFrom(_caches_selector()) + + # Every cloud cluster composes an activation policy; with the policy + # observed Healthy the cluster XR is composed. + for req, want, kinds in [ + (req2, want2, fn._ACTIVATE_GCP), + (req_creds, want_creds, fn._ACTIVATE_GCP), + (req4, want4, fn._ACTIVATE_AWS), + (req5, want5, fn._ACTIVATE_AWS), + (req6, want6, fn._ACTIVATE_GCP), + (req7, want7, fn._ACTIVATE_AWS), + (req8, want8, fn._ACTIVATE_AWS), + (req9, want9, fn._ACTIVATE_AWS), + (req10, want10, fn._ACTIVATE_NEBIUS), + (req11, want11, fn._ACTIVATE_NEBIUS), + (req12, want12, fn._ACTIVATE_AZURE), + (req13, want13, fn._ACTIVATE_AZURE), + (req14, want14, fn._ACTIVATE_VULTR), + (req_creds_vultr, want_creds_vultr, fn._ACTIVATE_VULTR), + (req15, want15, fn._ACTIVATE_VULTR), + (req16, want16, fn._ACTIVATE_CIVO), + (req_creds_civo, want_creds_civo, fn._ACTIVATE_CIVO), + (req17, want17, fn._ACTIVATE_CIVO), + (req17b, want17b, fn._ACTIVATE_CIVO), + ]: + _observe_activated(req, kinds) + _want_activation(want, kinds) + + # While the policy is missing even one of the kinds from status.activated + # (e.g. a provider still installing), and with no cluster observed, the + # function composes only the activation policy (not marked ready, so the + # composite doesn't report ready), not the cluster XR. Observing the + # policy with a kind missing exercises the all-kinds check rather than + # the policy-absent branch. + req_unactivated = copy.deepcopy(req2) + _observe_activated(req_unactivated, fn._ACTIVATE_GCP[:-1]) + want_unactivated = copy.deepcopy(want2) + del want_unactivated.desired.resources["gke-cluster"] + want_unactivated.desired.resources["activation"].ClearField("ready") + + # Once the cluster is observed, the function keeps composing it even + # when the policy momentarily stops reporting the kinds active, so an + # activation blip never drops a provisioned cluster from desired state. + req_blip = copy.deepcopy(req6) + del req_blip.observed.resources["activation"] + want_blip = copy.deepcopy(want6) + + return [ + Case(name="existing cluster with secrets composes backend and CPC", req=req1, want=want1), + Case(name="existing cluster with a non-GCP identity threads the identity type", req=req1b, want=want1b), + Case(name="GKE cluster first pass composes GKECluster XR only", req=req2, want=want2), + Case(name="GKE credentials pass through to GKECluster spec", req=req_creds, want=want_creds), + Case(name="existing cluster second pass with backend ready", req=req3, want=want3), + Case(name="EKS cluster first pass composes EKSCluster XR only", req=req4, want=want4), + Case(name="EKS cluster not ready re-emits existing CPC unchanged", req=req5, want=want5), + Case(name="GKE cluster ready composes CPC, backend, usage, and RWX StorageClass", req=req6, want=want6), + Case(name="EKS cluster ready composes ServingStack and Usage", req=req7, want=want7), + Case( + name="EKS node pool with a Capacity Block sets capacityBlock on the EKSCluster pool", + req=req8, + want=want8, + ), + Case( + name="EKS node pool with fabric EFA sets fabric on the EKSCluster pool", + req=req9, + want=want9, + ), + Case(name="Nebius cluster first pass composes NebiusCluster XR only", req=req10, want=want10), + Case( + name="Nebius cluster ready composes CPC with Nebius identity, ServingStack, and Usage", + req=req11, + want=want11, + ), + Case(name="AKS cluster first pass composes AKSCluster XR only", req=req12, want=want12), + Case( + name="AKS cluster ready composes CPC without identity, ServingStack, and Usage", + req=req13, + want=want13, + ), + Case(name="cloud cluster not activated composes only the policy", req=req_unactivated, want=want_unactivated), + Case(name="observed cluster keeps composing through an activation blip", req=req_blip, want=want_blip), + Case(name="Vultr cluster first pass composes VultrCluster XR only", req=req14, want=want14), + Case( + name="Vultr credentials pass through to VultrCluster spec", + req=req_creds_vultr, + want=want_creds_vultr, + ), + Case( + name="Vultr cluster ready composes CPC without identity, ServingStack, and Usage", + req=req15, + want=want15, + ), + Case(name="Civo cluster first pass composes CivoCluster XR only", req=req16, want=want16), + Case( + name="Civo credentials pass through to CivoCluster spec", + req=req_creds_civo, + want=want_creds_civo, + ), + Case( + name="Civo cluster ready composes CPC without identity, ServingStack, and Usage", + req=req17, + want=want17, + ), + Case( + name="Civo single-H100 pool projects per-pool NVLink disable to the backend", + req=req17b, + want=want17b, + ), + *guard_cases, + ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) - - -class TestGatewayStatus(unittest.IsolatedAsyncioTestCase): - """The hostname gate, which is what keeps a cluster off the schedule until - traffic to it is mutually authenticated in both directions.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - @staticmethod - def _request(*, address: str | None, ca: str | None, gateway_cas: list[str]) -> fnv1.RunFunctionRequest: - """A cluster and whatever its serving stack and the fleet's gateways have - published so far. The gateway name is Modelplane's own, so nothing - configures it.""" - xr = v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), - ), - ), - ) - stack_status: dict = {"conditions": [{"type": "Ready", "status": "True"}]} - gateway: dict = {} - if address: - gateway["address"] = address - if ca: - gateway["caCertificate"] = ca - if gateway: - stack_status["gateway"] = gateway - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json")) - ), - resources={ - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": {"name": "test-cluster-serving-stack-fd00b"}, - "status": stack_status, - } - ), - ), - }, + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes the resources an InferenceCluster needs.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) + + +# The hostname gate, which is what keeps a cluster off the schedule until +# traffic to it is mutually authenticated in both directions. + + +def _gateway_status_request(*, address: str | None, ca: str | None, gateway_cas: list[str]) -> fnv1.RunFunctionRequest: + """A cluster and whatever its serving stack and the fleet's gateways have + published so far. The gateway name is Modelplane's own, so nothing + configures it.""" + xr = v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), ), - ) - for i, cert in enumerate(gateway_cas): - req.required_resources["gateways"].items.append( - fnv1.Resource( + ), + ) + stack_status: dict = {"conditions": [{"type": "Ready", "status": "True"}]} + gateway: dict = {} + if address: + gateway["address"] = address + if ca: + gateway["caCertificate"] = ca + if gateway: + stack_status["gateway"] = gateway + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json"))), + resources={ + "serving-stack": fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceGateway", - "metadata": {"name": f"fleet-{i}"}, - "spec": {"clusterName": "test-cluster"}, - "status": {"clientCACertificate": cert}, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": {"name": "test-cluster-serving-stack-fd00b"}, + "status": stack_status, } ), - ) - ) - return req - - async def _gateway_status(self, req: fnv1.RunFunctionRequest) -> dict: - got = await self.runner.RunFunction(req, None) - return resource.struct_to_dict(got.desired.composite.resource).get("status", {}).get("gateway", {}) - - async def test_hostname_published_once_both_directions_are_authenticated(self) -> None: - """An address to reach, this cluster's CA so an InferenceGateway can tell - it reached the right cluster, and an InferenceGateway CA so the cluster - gateway demands a client certificate.""" - status = await self._gateway_status( - self._request(address="34.55.100.10", ca="cluster-ca", gateway_cas=["fleet-ca"]) - ) - self.assertEqual( - status, - { - "address": "34.55.100.10", - "caCertificate": "cluster-ca", - "hostname": _GATEWAY_HOSTNAME, + ), }, + ), + ) + for i, cert in enumerate(gateway_cas): + req.required_resources["gateways"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceGateway", + "metadata": {"name": f"fleet-{i}"}, + "spec": {"clusterName": "test-cluster"}, + "status": {"clientCACertificate": cert}, + } + ), + ) ) - - async def test_no_hostname_without_an_inference_gateway_ca(self) -> None: - """The case that matters: the cluster gateway only demands a client - certificate when it has a CA to check against, and with none it serves no - Gateway at all. Publishing the hostname anyway would make the cluster - schedulable when nothing is listening on it, so every request routed - there would be stranded.""" - status = await self._gateway_status(self._request(address="34.55.100.10", ca="cluster-ca", gateway_cas=[])) - self.assertEqual(status, {"address": "34.55.100.10", "caCertificate": "cluster-ca"}) - - async def test_no_hostname_without_this_clusters_ca(self) -> None: - """Without it an InferenceGateway can't validate the cluster gateway it - reaches, so it would have to fall back to the public trust store.""" - status = await self._gateway_status(self._request(address="34.55.100.10", ca=None, gateway_cas=["fleet-ca"])) - self.assertEqual(status, {"address": "34.55.100.10"}) - - async def test_no_gateway_status_before_an_address(self) -> None: - """A hostname that resolves to nothing strands every request routed to - it, and the CA is republished from the same status.""" - status = await self._gateway_status(self._request(address=None, ca="cluster-ca", gateway_cas=["fleet-ca"])) - self.assertEqual(status, {}) - - async def test_serving_stack_accepts_every_inference_gateway_ca(self) -> None: - """Any InferenceGateway may forward to this cluster, so its gateway - accepts every published CA, whichever cluster the InferenceGateway runs - on. These CAs are what switches the cluster gateway's mTLS listener on. - One that hasn't published a CA yet is left out rather than holding the - others back, and the list is sorted so it doesn't churn.""" - req = self._request(address="34.55.100.10", ca="cluster-ca", gateway_cas=["fleet-0-ca", "fleet-1-ca"]) - for name, status in (("aaa", {"clientCACertificate": "aaa-ca"}), ("pending", {})): - req.required_resources["gateways"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceGateway", - "metadata": {"name": name}, - "spec": {"clusterName": "elsewhere"}, - "status": status, - } - ), - ) + return req + + +def _gateway_status(req: fnv1.RunFunctionRequest) -> dict: + """The status.gateway RunFunction writes to the InferenceCluster for req.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + return resource.struct_to_dict(got.desired.composite.resource).get("status", {}).get("gateway", {}) + + +def test_gateway_status_hostname_published_once_both_directions_are_authenticated() -> None: + """The hostname is published once the address and both directions' CAs are.""" + # An address to reach, this cluster's CA so an InferenceGateway can tell it + # reached the right cluster, and an InferenceGateway CA so the cluster + # gateway demands a client certificate. + status = _gateway_status(_gateway_status_request(address="34.55.100.10", ca="cluster-ca", gateway_cas=["fleet-ca"])) + assert status == { + "address": "34.55.100.10", + "caCertificate": "cluster-ca", + "hostname": _GATEWAY_HOSTNAME, + } + + +def test_gateway_status_no_hostname_without_an_inference_gateway_ca() -> None: + """No hostname is published until an InferenceGateway has published a CA.""" + # The case that matters: the cluster gateway only demands a client + # certificate when it has a CA to check against, and with none it serves no + # Gateway at all. Publishing the hostname anyway would make the cluster + # schedulable when nothing is listening on it, so every request routed + # there would be stranded. + status = _gateway_status(_gateway_status_request(address="34.55.100.10", ca="cluster-ca", gateway_cas=[])) + assert status == {"address": "34.55.100.10", "caCertificate": "cluster-ca"} + + +def test_gateway_status_no_hostname_without_this_clusters_ca() -> None: + """No hostname is published until the cluster's own gateway CA is.""" + # Without it an InferenceGateway can't validate the cluster gateway it + # reaches, so it would have to fall back to the public trust store. + status = _gateway_status(_gateway_status_request(address="34.55.100.10", ca=None, gateway_cas=["fleet-ca"])) + assert status == {"address": "34.55.100.10"} + + +def test_gateway_status_no_gateway_status_before_an_address() -> None: + """No gateway status at all is published before the gateway has an address.""" + # A hostname that resolves to nothing strands every request routed to it, + # and the CA is republished from the same status. + status = _gateway_status(_gateway_status_request(address=None, ca="cluster-ca", gateway_cas=["fleet-ca"])) + assert status == {} + + +def test_gateway_status_serving_stack_accepts_every_inference_gateway_ca() -> None: + """The ServingStack's gateway accepts every published InferenceGateway CA, sorted.""" + # Any InferenceGateway may forward to this cluster, so its gateway accepts + # every published CA, whichever cluster the InferenceGateway runs on. These + # CAs are what switches the cluster gateway's mTLS listener on. One that + # hasn't published a CA yet is left out rather than holding the others + # back, and the list is sorted so it doesn't churn. + req = _gateway_status_request(address="34.55.100.10", ca="cluster-ca", gateway_cas=["fleet-0-ca", "fleet-1-ca"]) + for name, status in (("aaa", {"clientCACertificate": "aaa-ca"}), ("pending", {})): + req.required_resources["gateways"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceGateway", + "metadata": {"name": name}, + "spec": {"clusterName": "elsewhere"}, + "status": status, + } + ), ) - got = await self.runner.RunFunction(req, None) - stack = resource.struct_to_dict(got.desired.resources[fn.BACKEND_RESOURCE_KEY].resource) - self.assertEqual( - stack["spec"]["gateway"], - { - "hostname": _GATEWAY_HOSTNAME, - "clientCAs": [ - {"name": "aaa", "certificate": "aaa-ca"}, - {"name": "fleet-0", "certificate": "fleet-0-ca"}, - {"name": "fleet-1", "certificate": "fleet-1-ca"}, - ], - }, ) - - -class TestGatewayHostname(unittest.TestCase): - """The derived gateway hostname doubles as an SNI and a certificate SAN, so - two clusters must never derive the same one.""" - - def test_dots_become_a_single_dns_label(self) -> None: - """A dotted cluster name is a DNS-1123 subdomain, but the first segment - of the hostname has to be one DNS-1035 label.""" - hostname = fn._gateway_hostname("eu.example") - label = hostname.split(".")[0] - self.assertNotIn(".", label) - self.assertTrue(label.startswith("gateway-")) - - def test_names_differing_only_in_dots_do_not_collide(self) -> None: - """The hash covers the raw cluster name, so 'eu.example' and - 'eu-example' get different hostnames. Sharing one, a cluster's Service - would shadow the other's under a certificate it accepts.""" - self.assertNotEqual(fn._gateway_hostname("eu.example"), fn._gateway_hostname("eu-example")) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + stack = resource.struct_to_dict(got.desired.resources[fn.BACKEND_RESOURCE_KEY].resource) + assert stack["spec"]["gateway"] == { + "hostname": _GATEWAY_HOSTNAME, + "clientCAs": [ + {"name": "aaa", "certificate": "aaa-ca"}, + {"name": "fleet-0", "certificate": "fleet-0-ca"}, + {"name": "fleet-1", "certificate": "fleet-1-ca"}, + ], + } + + +# The derived gateway hostname doubles as an SNI and a certificate SAN, so two +# clusters must never derive the same one. + + +def test_gateway_hostname_dots_become_a_single_dns_label() -> None: + """A dotted cluster name still gives a hostname whose first label has no dots.""" + # A dotted cluster name is a DNS-1123 subdomain, but the first segment of + # the hostname has to be one DNS-1035 label. + hostname = fn._gateway_hostname("eu.example") + label = hostname.split(".")[0] + assert "." not in label + assert label.startswith("gateway-") + + +def test_gateway_hostname_names_differing_only_in_dots_do_not_collide() -> None: + """Cluster names that differ only in dots get different hostnames.""" + # The hash covers the raw cluster name, so 'eu.example' and 'eu-example' + # get different hostnames. Sharing one, a cluster's Service would shadow + # the other's under a certificate it accepts. + assert fn._gateway_hostname("eu.example") != fn._gateway_hostname("eu-example") diff --git a/functions/compose-inference-gateway/tests/__init__.py b/functions/compose-inference-gateway/tests/__init__.py deleted file mode 100644 index 5d373016d..000000000 --- a/functions/compose-inference-gateway/tests/__init__.py +++ /dev/null @@ -1,15 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - - diff --git a/functions/compose-inference-gateway/tests/test_fn.py b/functions/compose-inference-gateway/tests/test_fn.py index 74703a0a9..22e76ec57 100644 --- a/functions/compose-inference-gateway/tests/test_fn.py +++ b/functions/compose-inference-gateway/tests/test_fn.py @@ -14,15 +14,17 @@ """Tests for the compose-inference-gateway function.""" +import asyncio import base64 import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.inferencegateway import v1alpha1 @@ -230,1066 +232,960 @@ def _not_ready(reason: str, message: str, requirements: fnv1.Requirements) -> fn ) -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) +GATES_CASES = [ + Case( + name="unresolved requirements compose nothing", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + ), + want=_not_ready( + fn.CONDITION_REASON_WAITING_FOR_CLUSTER, + "Waiting for the gateway's cluster and the other gateways to resolve", + _requirements(), + ), + ), + Case( + name="a named cluster that does not exist", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + required_resources=_required(clusters=[], gateways=[_gateway_xr("eu", _CLUSTER)]), + ), + want=_not_ready( + fn.CONDITION_REASON_WAITING_FOR_CLUSTER, + f"InferenceCluster {_CLUSTER} does not exist", + _requirements(), + ), + ), + Case( + name="a cluster that already hosts a lower-named gateway", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + required_resources=_required( + clusters=[_cluster()], + gateways=[_gateway_xr("eu", _CLUSTER), _gateway_xr("aaa", _CLUSTER)], + ), + ), + want=_not_ready( + fn.CONDITION_REASON_CLUSTER_TAKEN, + f"InferenceCluster {_CLUSTER} already hosts InferenceGateway aaa", + _requirements(), + ), + ), + Case( + name="a cluster with no providerConfigRef yet", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + required_resources=_required( + clusters=[_cluster(provider_config=None)], gateways=[_gateway_xr("eu", _CLUSTER)] + ), + ), + want=_not_ready( + fn.CONDITION_REASON_WAITING_FOR_CLUSTER, + f"InferenceCluster {_CLUSTER} has not published a providerConfigRef", + _requirements(), + ), + ), +] -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - maxDiff = None +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - async def test_gates(self) -> None: - """Passes where the gateway can't be composed compose nothing, and say - why. Asserting the whole response proves nothing is composed against a - cluster we can't reach, rather than a subset being applied.""" - cases = [ - Case( - name="unresolved requirements compose nothing", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - ), - want=_not_ready( - fn.CONDITION_REASON_WAITING_FOR_CLUSTER, - "Waiting for the gateway's cluster and the other gateways to resolve", - _requirements(), - ), - ), - Case( - name="a named cluster that does not exist", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required(clusters=[], gateways=[_gateway_xr("eu", _CLUSTER)]), - ), - want=_not_ready( - fn.CONDITION_REASON_WAITING_FOR_CLUSTER, - f"InferenceCluster {_CLUSTER} does not exist", - _requirements(), - ), - ), - Case( - name="a cluster that already hosts a lower-named gateway", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER), _gateway_xr("aaa", _CLUSTER)], - ), - ), - want=_not_ready( - fn.CONDITION_REASON_CLUSTER_TAKEN, - f"InferenceCluster {_CLUSTER} already hosts InferenceGateway aaa", - _requirements(), - ), - ), - Case( - name="a cluster with no providerConfigRef yet", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required( - clusters=[_cluster(provider_config=None)], gateways=[_gateway_xr("eu", _CLUSTER)] - ), - ), - want=_not_ready( - fn.CONDITION_REASON_WAITING_FOR_CLUSTER, - f"InferenceCluster {_CLUSTER} has not published a providerConfigRef", - _requirements(), - ), - ), - ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) +@pytest.mark.parametrize("case", GATES_CASES, ids=lambda case: case.name) +def test_gates(case: Case) -> None: + """Passes where the gateway can't be composed compose nothing, and say + why. Asserting the whole response proves nothing is composed against a + cluster we can't reach, rather than a subset being applied.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) - async def test_minimal_gateway(self) -> None: - """A gateway with no TLS or auth: the getting-started shape. - Composes the gateway objects and no auth policies, and reports no - endpoints until the Gateway has an address. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = await self.runner.RunFunction(req, None) - - self.assertEqual( - sorted(got.desired.resources), - sorted( - [ - # The CA whose client certificates a cluster gateway trusts, - # published as a ClusterIssuer for compose-model-route to issue - # per-namespace client certificates from. - "client-ca-certificate", - "client-ca-issuer", - "client-ca-bundle", - "client-ca-configmap", - "client-selfsigned-issuer", - "client-traffic-policy", - "envoy-proxy", - "failover-policy", - "gateway", - "healthz-filter", - "healthz-route", - ] - ), - "composes the gateway objects and its client PKI, and no caller auth", - ) - for key, res in got.desired.resources.items(): - d = resource.struct_to_dict(res.resource) - self.assertEqual(d["kind"], "Object", f"{key} targets the gateway's cluster") - self.assertEqual( - d["spec"]["providerConfigRef"], - {"kind": "ClusterProviderConfig", "name": _PC}, - f"{key} uses the cluster's ClusterProviderConfig", - ) - # An InferenceGateway is cluster-scoped, and Crossplane only - # defaults a composed namespaced resource's namespace from a - # namespaced composite. Without this every reconcile fails with - # "an empty namespace may not be set when a resource name is - # provided" and nothing is composed at all. - self.assertEqual( - d["metadata"]["namespace"], - fn.CONTROL_PLANE_NAMESPACE, - f"{key} sets its own namespace, which a cluster-scoped XR must", - ) - manifest = d["spec"]["forProvider"]["manifest"] - if manifest["kind"] == "ClusterIssuer": - # Cluster-scoped: compose-model-route issues client certs from it - # into team namespaces, so it has no namespace of its own. - self.assertNotIn( - "namespace", - manifest["metadata"], - f"{key} is cluster-scoped, so it sets no namespace", - ) - continue - if manifest["kind"] == "Bundle": - # A Bundle is cluster-scoped, so it has no namespace of its own. - # It picks the namespace it syncs its ConfigMap to by selector. - self.assertNotIn( - "namespace", - manifest["metadata"], - f"{key} is cluster-scoped, so it sets no namespace", - ) - self.assertEqual( - manifest["spec"]["target"]["namespaceSelector"], - {"matchLabels": {"kubernetes.io/metadata.name": fn.REMOTE_NAMESPACE}}, - f"{key} syncs only to the remote namespace", - ) - continue - self.assertEqual( - manifest["metadata"]["namespace"], - fn.REMOTE_NAMESPACE, - f"{key} lands in the remote namespace", - ) +def test_minimal_gateway() -> None: + """A gateway with no TLS or auth: the getting-started shape. - # Two attempts per priority, so a retry tries another endpoint at the - # same priority before moving down. At one, a single transient failure - # on one replica would send the request to the next priority, which may - # be a paid provider. - failover = resource.struct_to_dict(got.desired.resources["failover-policy"].resource) - self.assertEqual( - failover["spec"]["forProvider"]["manifest"]["spec"]["retry"], - { - "numAttemptsPerPriority": 2, - "numRetries": 3, - "retryOn": { - # retriable-status-codes has to be present for the status - # codes below to do anything: Envoy Gateway replaces retry_on - # wholesale with this list, and Envoy only consults - # retriable_status_codes when retry_on names it. Without it a - # provider answering 503 or 429 is never retried, which is - # the case failover exists for. - "triggers": [ - "connect-failure", - "refused-stream", - "reset", - "retriable-status-codes", - ], - # 429 so a rate-limited provider's traffic overflows to - # another endpoint rather than failing back to the caller. - "httpStatusCodes": [429, 503], - }, - }, - ) - # Panic mode defaults to 50%, above which Envoy ignores health and - # spreads traffic over every endpoint including the ejected ones. Every - # endpoint of a ModelService shares one cluster, so ejecting a whole - # priority tier usually crosses it and failover stops working. - # - # Asserted on the whole healthCheck, because panicThreshold is a sibling - # of passive rather than a field inside it, and nested wrongly the API - # server prunes it while the policy still applies. - self.assertEqual( - failover["spec"]["forProvider"]["manifest"]["spec"]["healthCheck"], - { - "passive": { - "baseEjectionTime": "30s", - "consecutive5XxErrors": 5, - "interval": "5s", - "maxEjectionPercent": 100, - }, - "panicThreshold": 0, - }, + Composes the gateway objects and no auth policies, and reports no + endpoints until the Gateway has an address. + """ + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + assert sorted(got.desired.resources) == sorted( + [ + # The CA whose client certificates a cluster gateway trusts, + # published as a ClusterIssuer for compose-model-route to issue + # per-namespace client certificates from. + "client-ca-certificate", + "client-ca-issuer", + "client-ca-bundle", + "client-ca-configmap", + "client-selfsigned-issuer", + "client-traffic-policy", + "envoy-proxy", + "failover-policy", + "gateway", + "healthz-filter", + "healthz-route", + ] + ), "composes the gateway objects and its client PKI, and no caller auth" + for key, res in got.desired.resources.items(): + d = resource.struct_to_dict(res.resource) + assert d["kind"] == "Object", f"{key} targets the gateway's cluster" + assert d["spec"]["providerConfigRef"] == {"kind": "ClusterProviderConfig", "name": _PC}, ( + f"{key} uses the cluster's ClusterProviderConfig" ) - self.assertEqual( - failover["spec"]["forProvider"]["manifest"]["spec"]["targetRefs"], - [{"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": fn._GATEWAY_NAME}], - "targets the Gateway, so it covers every ModelService's route", + # An InferenceGateway is cluster-scoped, and Crossplane only + # defaults a composed namespaced resource's namespace from a + # namespaced composite. Without this every reconcile fails with + # "an empty namespace may not be set when a resource name is + # provided" and nothing is composed at all. + assert d["metadata"]["namespace"] == fn.CONTROL_PLANE_NAMESPACE, ( + f"{key} sets its own namespace, which a cluster-scoped XR must" ) - - # AI Gateway buffers whole bodies, and Envoy Gateway's 32KiB default - # buffer limit answers 413 to a long prompt or non-streamed completion. - self.assertEqual( - resource.struct_to_dict(got.desired.resources["client-traffic-policy"].resource)["spec"]["forProvider"][ - "manifest" + manifest = d["spec"]["forProvider"]["manifest"] + if manifest["kind"] == "ClusterIssuer": + # Cluster-scoped: compose-model-route issues client certs from it + # into team namespaces, so it has no namespace of its own. + assert "namespace" not in manifest["metadata"], f"{key} is cluster-scoped, so it sets no namespace" + continue + if manifest["kind"] == "Bundle": + # A Bundle is cluster-scoped, so it has no namespace of its own. + # It picks the namespace it syncs its ConfigMap to by selector. + assert "namespace" not in manifest["metadata"], f"{key} is cluster-scoped, so it sets no namespace" + assert manifest["spec"]["target"]["namespaceSelector"] == { + "matchLabels": {"kubernetes.io/metadata.name": fn.REMOTE_NAMESPACE} + }, f"{key} syncs only to the remote namespace" + continue + assert manifest["metadata"]["namespace"] == fn.REMOTE_NAMESPACE, f"{key} lands in the remote namespace" + + # Two attempts per priority, so a retry tries another endpoint at the + # same priority before moving down. At one, a single transient failure + # on one replica would send the request to the next priority, which may + # be a paid provider. + failover = resource.struct_to_dict(got.desired.resources["failover-policy"].resource) + assert failover["spec"]["forProvider"]["manifest"]["spec"]["retry"] == { + "numAttemptsPerPriority": 2, + "numRetries": 3, + "retryOn": { + # retriable-status-codes has to be present for the status + # codes below to do anything: Envoy Gateway replaces retry_on + # wholesale with this list, and Envoy only consults + # retriable_status_codes when retry_on names it. Without it a + # provider answering 503 or 429 is never retried, which is + # the case failover exists for. + "triggers": [ + "connect-failure", + "refused-stream", + "reset", + "retriable-status-codes", ], - { - "apiVersion": "gateway.envoyproxy.io/v1alpha1", - "kind": "ClientTrafficPolicy", - "metadata": {"name": "inference-gateway-client-traffic", "namespace": "modelplane-system"}, - "spec": { - "targetRefs": [ - {"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": "inference-gateway"} - ], - "connection": {"bufferLimit": "50Mi"}, - "http2": {"initialStreamWindowSize": "16Mi", "initialConnectionWindowSize": "24Mi"}, - }, - }, - ) + # 429 so a rate-limited provider's traffic overflows to + # another endpoint rather than failing back to the caller. + "httpStatusCodes": [429, 503], + }, + } + # Panic mode defaults to 50%, above which Envoy ignores health and + # spreads traffic over every endpoint including the ejected ones. Every + # endpoint of a ModelService shares one cluster, so ejecting a whole + # priority tier usually crosses it and failover stops working. + # + # Asserted on the whole healthCheck, because panicThreshold is a sibling + # of passive rather than a field inside it, and nested wrongly the API + # server prunes it while the policy still applies. + assert failover["spec"]["forProvider"]["manifest"]["spec"]["healthCheck"] == { + "passive": { + "baseEjectionTime": "30s", + "consecutive5XxErrors": 5, + "interval": "5s", + "maxEjectionPercent": 100, + }, + "panicThreshold": 0, + } + assert failover["spec"]["forProvider"]["manifest"]["spec"]["targetRefs"] == [ + {"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": fn._GATEWAY_NAME} + ], "targets the Gateway, so it covers every ModelService's route" + + # AI Gateway buffers whole bodies, and Envoy Gateway's 32KiB default + # buffer limit answers 413 to a long prompt or non-streamed completion. + assert resource.struct_to_dict(got.desired.resources["client-traffic-policy"].resource)["spec"]["forProvider"][ + "manifest" + ] == { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "ClientTrafficPolicy", + "metadata": {"name": "inference-gateway-client-traffic", "namespace": "modelplane-system"}, + "spec": { + "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": "inference-gateway"}], + "connection": {"bufferLimit": "50Mi"}, + "http2": {"initialStreamWindowSize": "16Mi", "initialConnectionWindowSize": "24Mi"}, + }, + } - # The token fields must read request metadata, not the response body or - # a header. The caller header is stripped before a third-party backend - # sees it, so a log reading the header loses the caller on exactly the - # records that attribute provider spend. - log = resource.struct_to_dict(got.desired.resources["envoy-proxy"].resource) - fields = log["spec"]["forProvider"]["manifest"]["spec"]["telemetry"]["accessLog"]["settings"][0]["format"][ - "json" - ] - # Two proxy pods spread softly across nodes and zones, a disruption - # budget so a drain can't evict both, and ndots:1. Without ndots:1 every - # backend hostname is resolved against each of the pod's search domains - # first, since they all have fewer than five dots. A cluster whose - # upstream resolver is slow then stalls resolution, and Envoy answers 503 - # with nothing but DNS timeouts to show for it. - proxy_labels = { - "gateway.envoyproxy.io/owning-gateway-name": "inference-gateway", - "gateway.envoyproxy.io/owning-gateway-namespace": "modelplane-system", - } - self.assertEqual( - log["spec"]["forProvider"]["manifest"]["spec"]["provider"], - { - "type": "Kubernetes", - "kubernetes": { - "envoyService": {"externalTrafficPolicy": "Cluster"}, - "envoyDeployment": { - "replicas": 2, - "patch": {"type": "StrategicMerge", "value": fn._NDOTS_PATCH}, - "pod": { - "topologySpreadConstraints": [ - { - "maxSkew": 1, - "topologyKey": "kubernetes.io/hostname", - "whenUnsatisfiable": "ScheduleAnyway", - "labelSelector": {"matchLabels": proxy_labels}, - }, - { - "maxSkew": 1, - "topologyKey": "topology.kubernetes.io/zone", - "whenUnsatisfiable": "ScheduleAnyway", - "labelSelector": {"matchLabels": proxy_labels}, - }, - ] + # The token fields must read request metadata, not the response body or + # a header. The caller header is stripped before a third-party backend + # sees it, so a log reading the header loses the caller on exactly the + # records that attribute provider spend. + log = resource.struct_to_dict(got.desired.resources["envoy-proxy"].resource) + fields = log["spec"]["forProvider"]["manifest"]["spec"]["telemetry"]["accessLog"]["settings"][0]["format"]["json"] + # Two proxy pods spread softly across nodes and zones, a disruption + # budget so a drain can't evict both, and ndots:1. Without ndots:1 every + # backend hostname is resolved against each of the pod's search domains + # first, since they all have fewer than five dots. A cluster whose + # upstream resolver is slow then stalls resolution, and Envoy answers 503 + # with nothing but DNS timeouts to show for it. + proxy_labels = { + "gateway.envoyproxy.io/owning-gateway-name": "inference-gateway", + "gateway.envoyproxy.io/owning-gateway-namespace": "modelplane-system", + } + assert log["spec"]["forProvider"]["manifest"]["spec"]["provider"] == { + "type": "Kubernetes", + "kubernetes": { + "envoyService": {"externalTrafficPolicy": "Cluster"}, + "envoyDeployment": { + "replicas": 2, + "patch": {"type": "StrategicMerge", "value": fn._NDOTS_PATCH}, + "pod": { + "topologySpreadConstraints": [ + { + "maxSkew": 1, + "topologyKey": "kubernetes.io/hostname", + "whenUnsatisfiable": "ScheduleAnyway", + "labelSelector": {"matchLabels": proxy_labels}, + }, + { + "maxSkew": 1, + "topologyKey": "topology.kubernetes.io/zone", + "whenUnsatisfiable": "ScheduleAnyway", + "labelSelector": {"matchLabels": proxy_labels}, }, - }, - "envoyPDB": {"maxUnavailable": 1}, + ] }, }, - ) - # A stopping pod drains for as long as a request may run by default, so - # a restart doesn't cut off streams in flight. - self.assertEqual(log["spec"]["forProvider"]["manifest"]["spec"]["shutdown"], {"drainTimeout": "300s"}) - self.assertEqual( - fn._NDOTS_PATCH["spec"]["template"]["spec"]["dnsConfig"]["options"], - [{"name": "ndots", "value": "1"}], - ) - - self.assertEqual(fields["caller"], "%DYNAMIC_METADATA(io.envoy.ai_gateway:caller)%") - self.assertEqual(fields["input_tokens"], "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_input_token)%") - self.assertEqual(fields["output_tokens"], "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_output_token)%") - - gw = resource.struct_to_dict(got.desired.resources["gateway"].resource) - manifest = gw["spec"]["forProvider"]["manifest"] - self.assertEqual( - manifest["spec"]["listeners"], - [ - { - "name": "http", - "protocol": "HTTP", - "port": 80, - "allowedRoutes": { - "namespaces": { - "from": "Selector", - "selector": { - "matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}] - }, - } - }, + "envoyPDB": {"maxUnavailable": 1}, + }, + } + # A stopping pod drains for as long as a request may run by default, so + # a restart doesn't cut off streams in flight. + assert log["spec"]["forProvider"]["manifest"]["spec"]["shutdown"] == {"drainTimeout": "300s"} + assert fn._NDOTS_PATCH["spec"]["template"]["spec"]["dnsConfig"]["options"] == [{"name": "ndots", "value": "1"}] + + assert fields["caller"] == "%DYNAMIC_METADATA(io.envoy.ai_gateway:caller)%" + assert fields["input_tokens"] == "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_input_token)%" + assert fields["output_tokens"] == "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_output_token)%" + + gw = resource.struct_to_dict(got.desired.resources["gateway"].resource) + manifest = gw["spec"]["forProvider"]["manifest"] + assert manifest["spec"]["listeners"] == [ + { + "name": "http", + "protocol": "HTTP", + "port": 80, + "allowedRoutes": { + "namespaces": { + "from": "Selector", + "selector": {"matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}]}, } - ], - "one HTTP listener, no hostname, accepting routes from the mirrored namespaces", - ) - self.assertEqual( - manifest["spec"]["infrastructure"]["parametersRef"], - {"group": "gateway.envoyproxy.io", "kind": "EnvoyProxy", "name": fn._GATEWAY_NAME}, - "its own EnvoyProxy, not the GatewayClass's", - ) - self.assertEqual( - resource.struct_to_dict(got.desired.composite.resource).get("status"), - {}, - "nothing to report until the Gateway has an address", - ) + }, + } + ], "one HTTP listener, no hostname, accepting routes from the mirrored namespaces" + assert manifest["spec"]["infrastructure"]["parametersRef"] == { + "group": "gateway.envoyproxy.io", + "kind": "EnvoyProxy", + "name": fn._GATEWAY_NAME, + }, "its own EnvoyProxy, not the GatewayClass's" + assert resource.struct_to_dict(got.desired.composite.resource).get("status") == {}, ( + "nothing to report until the Gateway has an address" + ) - async def test_full_gateway(self) -> None: - """A gateway with TLS and auth, whose Gateway has an address. - - Checks the things a caller depends on: the HTTPS listener, the Secrets - copied to the cluster, the caller policy naming them, /healthz exempted - from that policy, and a status publishing no URLs, since a caller - reaches a TLS gateway on a DNS name only its owner knows. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr( - tls=v1alpha1.Tls(certificateRefs=[v1alpha1.CertificateRef(name="eu-tls-0")]), - auth=_api_key_auth(), - ) - ) - ), - resources={ - "gateway": _observed_gateway(_ADDRESS, ready=True), - "caller-auth": _observed_accepted(), - }, - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{ - "caller-secrets": [_secret("ml-team-keys", {"ml-team-assistant": "sk-mp-a1b2c3"})], - "tls-secret-0": [_secret("eu-tls-0", {"tls.crt": "cert", "tls.key": "key"})], - }, - ), - ) - got = await self.runner.RunFunction(req, None) - - self.assertEqual( - sorted(got.desired.resources), - [ - "caller-auth", - "caller-secret-ml-team-keys", - "client-ca-bundle", - "client-ca-certificate", - "client-ca-configmap", - "client-ca-issuer", - "client-selfsigned-issuer", - "client-traffic-policy", - "envoy-proxy", - "failover-policy", - "gateway", - "healthz-auth", - "healthz-filter", - "healthz-route", - "redirect-auth", - "redirect-route", - "tls-secret-eu-tls-0", - ], - ) - def manifest(key: str) -> dict: - return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] +def test_full_gateway() -> None: + """A gateway with TLS and auth, whose Gateway has an address. - self.assertEqual( - manifest("gateway")["spec"]["listeners"][1], - { - "name": "https", - "protocol": "HTTPS", - "port": 443, - "tls": {"mode": "Terminate", "certificateRefs": [{"name": "eu-tls-0"}]}, - "allowedRoutes": { - "namespaces": { - "from": "Selector", - "selector": {"matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}]}, - } - }, + Checks the things a caller depends on: the HTTPS listener, the Secrets + copied to the cluster, the caller policy naming them, /healthz exempted + from that policy, and a status publishing no URLs, since a caller + reaches a TLS gateway on a DNS name only its owner knows. + """ + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr( + tls=v1alpha1.Tls(certificateRefs=[v1alpha1.CertificateRef(name="eu-tls-0")]), + auth=_api_key_auth(), + ) + ) + ), + resources={ + "gateway": _observed_gateway(_ADDRESS, ready=True), + "caller-auth": _observed_accepted(), }, - ) - self.assertEqual(got.requirements, _requirements(auth=True, tls=1)) - self.assertEqual( - manifest("tls-secret-eu-tls-0"), - { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": "eu-tls-0", "namespace": fn.REMOTE_NAMESPACE}, - "type": "kubernetes.io/tls", - "data": { - "tls.crt": base64.b64encode(b"cert").decode(), - "tls.key": base64.b64encode(b"key").decode(), - }, + ), + required_resources=_required( + clusters=[_cluster()], + gateways=[_gateway_xr("eu", _CLUSTER)], + **{ + "caller-secrets": [_secret("ml-team-keys", {"ml-team-assistant": "sk-mp-a1b2c3"})], + "tls-secret-0": [_secret("eu-tls-0", {"tls.crt": "cert", "tls.key": "key"})], }, - "the certificate is copied verbatim, keeping the name the Gateway refers to it by", - ) - self.assertEqual( - manifest("caller-auth")["spec"]["apiKeyAuth"], + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + assert sorted(got.desired.resources) == [ + "caller-auth", + "caller-secret-ml-team-keys", + "client-ca-bundle", + "client-ca-certificate", + "client-ca-configmap", + "client-ca-issuer", + "client-selfsigned-issuer", + "client-traffic-policy", + "envoy-proxy", + "failover-policy", + "gateway", + "healthz-auth", + "healthz-filter", + "healthz-route", + "redirect-auth", + "redirect-route", + "tls-secret-eu-tls-0", + ] + + def manifest(key: str) -> dict: + return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] + + assert manifest("gateway")["spec"]["listeners"][1] == { + "name": "https", + "protocol": "HTTPS", + "port": 443, + "tls": {"mode": "Terminate", "certificateRefs": [{"name": "eu-tls-0"}]}, + "allowedRoutes": { + "namespaces": { + "from": "Selector", + "selector": {"matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}]}, + } + }, + } + assert _to_dict(got.requirements) == _to_dict(_requirements(auth=True, tls=1)) + assert manifest("tls-secret-eu-tls-0") == { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "eu-tls-0", "namespace": fn.REMOTE_NAMESPACE}, + "type": "kubernetes.io/tls", + "data": { + "tls.crt": base64.b64encode(b"cert").decode(), + "tls.key": base64.b64encode(b"key").decode(), + }, + }, "the certificate is copied verbatim, keeping the name the Gateway refers to it by" + assert manifest("caller-auth")["spec"]["apiKeyAuth"] == { + "credentialRefs": [{"name": "callers-ml-team-keys"}], + # Authorization for OpenAI clients, x-api-key for Anthropic ones. + "extractFrom": [{"headers": ["Authorization", "x-api-key"]}], + "forwardClientIDHeader": fn._CALLER_HEADER, + "sanitize": True, + } + assert manifest("healthz-auth")["spec"] == { + "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "HTTPRoute", "name": fn._HEALTHZ_NAME}], + "authorization": {"defaultAction": "Allow"}, + }, "/healthz overrides the Gateway-level policy so a health check needs no credential" + # Inference binds to the HTTPS listener alone, so :80 carries only + # /healthz and this catch-all redirect to it. /healthz is an Exact match, + # so it still answers a plain-HTTP health check. + assert manifest("healthz-route")["spec"]["parentRefs"] == [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": fn._GATEWAY_NAME, + "sectionName": "http", + } + ] + assert manifest("redirect-route")["spec"] == { + "parentRefs": [ { - "credentialRefs": [{"name": "callers-ml-team-keys"}], - # Authorization for OpenAI clients, x-api-key for Anthropic ones. - "extractFrom": [{"headers": ["Authorization", "x-api-key"]}], - "forwardClientIDHeader": fn._CALLER_HEADER, - "sanitize": True, - }, - ) - self.assertEqual( - manifest("healthz-auth")["spec"], + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": fn._GATEWAY_NAME, + "sectionName": "http", + } + ], + "rules": [ { - "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "HTTPRoute", "name": fn._HEALTHZ_NAME}], - "authorization": {"defaultAction": "Allow"}, - }, - "/healthz overrides the Gateway-level policy so a health check needs no credential", + "matches": [{"path": {"type": "PathPrefix", "value": "/"}}], + "filters": [{"type": "RequestRedirect", "requestRedirect": {"scheme": "https", "statusCode": 301}}], + } + ], + } + assert manifest("redirect-auth")["spec"] == { + "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "HTTPRoute", "name": fn._REDIRECT_NAME}], + "authorization": {"defaultAction": "Allow"}, + }, "the redirect must happen before auth, or an unauthenticated caller gets 401 instead of being sent to HTTPS" + assert resource.struct_to_dict(got.desired.composite.resource)["status"] == {"address": _ADDRESS} + assert [_to_dict(c) for c in got.conditions] == [ + _to_dict( + fnv1.Condition( + type=fn.CONDITION_TYPE_GATEWAY_READY, + status=fnv1.STATUS_CONDITION_TRUE, + reason=fn.CONDITION_REASON_GATEWAY_PROGRAMMED, + ) ) - # Inference binds to the HTTPS listener alone, so :80 carries only - # /healthz and this catch-all redirect to it. /healthz is an Exact match, - # so it still answers a plain-HTTP health check. - self.assertEqual( - manifest("healthz-route")["spec"]["parentRefs"], - [ - { - "group": "gateway.networking.k8s.io", - "kind": "Gateway", - "name": fn._GATEWAY_NAME, - "sectionName": "http", - } + ] + + +def test_endpoints_are_built_from_the_address() -> None: + """A gateway serving plain HTTP publishes URLs on its address, which is + something a caller can actually put in an SDK's base_url.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), + resources={"gateway": _observed_gateway(_ADDRESS, ready=False)}, + ), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert resource.struct_to_dict(got.desired.composite.resource)["status"] == { + "address": _ADDRESS, + "endpoints": { + "openAI": f"http://{_ADDRESS}/v1", + "anthropic": f"http://{_ADDRESS}/anthropic/v1", + }, + } + assert next(iter(got.conditions)).reason == fn.CONDITION_REASON_WAITING_FOR_GATEWAY, ( + "an address alone isn't readiness; the Gateway must be programmed" + ) + + +def test_an_ipv6_address_is_bracketed_in_the_endpoints() -> None: + """A bare IPv6 literal collides with the port separator in a URL, so an + SDK given http://2001:db8::1/v1 as a base_url can't use it.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), + resources={"gateway": _observed_gateway("2001:db8::1", ready=False)}, + ), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert resource.struct_to_dict(got.desired.composite.resource)["status"]["endpoints"] == { + "openAI": "http://[2001:db8::1]/v1", + "anthropic": "http://[2001:db8::1]/anthropic/v1", + } + + +def test_resolves_each_cluster_gateway_name() -> None: + """A Service per cluster gateway, resolving its name to its address here. + + A ModelService's backends address a cluster gateway by the name + compose-inference-cluster derived, and this gateway's Envoy resolves it, + so its cluster needs a Service of that name. An IP is served by a + headless Service and an EndpointSlice; a hostname, which is how a cloud + load balancer names itself, by an ExternalName Service. An IP also gets + the Endpoints the slice supersedes, because kube-dns reads only that and + is what GKE runs. A cluster that hasn't published both an address and a + name gets neither. + """ + ipv4 = "prod-ipv4-gateway-aaaaa.modelplane-system.svc.cluster.local" + ipv6 = "prod-ipv6-gateway-bbbbb.modelplane-system.svc.cluster.local" + dns = "prod-dns-gateway-ccccc.modelplane-system.svc.cluster.local" + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + required_resources=_required( + gateways=[_gateway_xr("eu", _CLUSTER)], + clusters=[ + _cluster(), # this gateway's own cluster, no gateway published yet + _cluster_with_gateway("prod-ipv4", address="203.0.113.7", hostname=ipv4), + _cluster_with_gateway("prod-ipv6", address="2001:db8::1", hostname=ipv6), + _cluster_with_gateway("prod-dns", address="lb-x.elb.amazonaws.com", hostname=dns), ], + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + resolvers = { + key: resource.struct_to_dict(res.resource) + for key, res in got.desired.resources.items() + if key.startswith("cluster-name") + } + for key, obj in resolvers.items(): + assert obj["spec"]["providerConfigRef"] == {"kind": "ClusterProviderConfig", "name": _PC}, ( + f"{key} is composed against this gateway's own cluster" ) - self.assertEqual( - manifest("redirect-route")["spec"], - { - "parentRefs": [ - { - "group": "gateway.networking.k8s.io", - "kind": "Gateway", - "name": fn._GATEWAY_NAME, - "sectionName": "http", - } - ], - "rules": [ - { - "matches": [{"path": {"type": "PathPrefix", "value": "/"}}], - "filters": [ - {"type": "RequestRedirect", "requestRedirect": {"scheme": "https", "statusCode": 301}} - ], - } - ], + manifests = {key: obj["spec"]["forProvider"]["manifest"] for key, obj in resolvers.items()} + assert manifests == { + "cluster-name-prod-ipv4-gateway-aaaaa": { + "apiVersion": "v1", + "kind": "Service", + "metadata": {"name": "prod-ipv4-gateway-aaaaa", "namespace": fn.REMOTE_NAMESPACE}, + "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, + }, + "cluster-name-endpoints-prod-ipv4-gateway-aaaaa": { + "apiVersion": "v1", + "kind": "Endpoints", + "metadata": { + "name": "prod-ipv4-gateway-aaaaa", + "namespace": fn.REMOTE_NAMESPACE, + # Off, or the mirroring controller writes a second + # EndpointSlice over the one composed beside this. + "labels": {"endpointslice.kubernetes.io/skip-mirror": "true"}, }, - ) - self.assertEqual( - manifest("redirect-auth")["spec"], - { - "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "HTTPRoute", "name": fn._REDIRECT_NAME}], - "authorization": {"defaultAction": "Allow"}, + "subsets": [{"addresses": [{"ip": "203.0.113.7"}], "ports": [{"name": "https", "port": 443}]}], + }, + "cluster-name-slice-prod-ipv4-gateway-aaaaa": { + "apiVersion": "discovery.k8s.io/v1", + "kind": "EndpointSlice", + "metadata": { + "name": "prod-ipv4-gateway-aaaaa", + "namespace": fn.REMOTE_NAMESPACE, + "labels": {"kubernetes.io/service-name": "prod-ipv4-gateway-aaaaa"}, }, - "the redirect must happen before auth, or an unauthenticated caller gets 401 instead of being sent to HTTPS", - ) - self.assertEqual(resource.struct_to_dict(got.desired.composite.resource)["status"], {"address": _ADDRESS}) - self.assertEqual( - list(got.conditions), - [ - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_TRUE, - reason=fn.CONDITION_REASON_GATEWAY_PROGRAMMED, - ) - ], - ) - - async def test_endpoints_are_built_from_the_address(self) -> None: - """A gateway serving plain HTTP publishes URLs on its address, which is - something a caller can actually put in an SDK's base_url.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={"gateway": _observed_gateway(_ADDRESS, ready=False)}, - ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual( - resource.struct_to_dict(got.desired.composite.resource)["status"], - { - "address": _ADDRESS, - "endpoints": { - "openAI": f"http://{_ADDRESS}/v1", - "anthropic": f"http://{_ADDRESS}/anthropic/v1", - }, + "addressType": "IPv4", + "ports": [{"name": "https", "port": 443}], + "endpoints": [{"addresses": ["203.0.113.7"], "conditions": {"ready": True}}], + }, + "cluster-name-prod-ipv6-gateway-bbbbb": { + "apiVersion": "v1", + "kind": "Service", + "metadata": {"name": "prod-ipv6-gateway-bbbbb", "namespace": fn.REMOTE_NAMESPACE}, + "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, + }, + "cluster-name-endpoints-prod-ipv6-gateway-bbbbb": { + "apiVersion": "v1", + "kind": "Endpoints", + "metadata": { + "name": "prod-ipv6-gateway-bbbbb", + "namespace": fn.REMOTE_NAMESPACE, + # Off, or the mirroring controller writes a second + # EndpointSlice over the one composed beside this. + "labels": {"endpointslice.kubernetes.io/skip-mirror": "true"}, }, - ) - self.assertEqual( - next(iter(got.conditions)).reason, - fn.CONDITION_REASON_WAITING_FOR_GATEWAY, - "an address alone isn't readiness; the Gateway must be programmed", - ) - - async def test_an_ipv6_address_is_bracketed_in_the_endpoints(self) -> None: - """A bare IPv6 literal collides with the port separator in a URL, so an - SDK given http://2001:db8::1/v1 as a base_url can't use it.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={"gateway": _observed_gateway("2001:db8::1", ready=False)}, - ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual( - resource.struct_to_dict(got.desired.composite.resource)["status"]["endpoints"], - { - "openAI": "http://[2001:db8::1]/v1", - "anthropic": "http://[2001:db8::1]/anthropic/v1", + "subsets": [{"addresses": [{"ip": "2001:db8::1"}], "ports": [{"name": "https", "port": 443}]}], + }, + "cluster-name-slice-prod-ipv6-gateway-bbbbb": { + "apiVersion": "discovery.k8s.io/v1", + "kind": "EndpointSlice", + "metadata": { + "name": "prod-ipv6-gateway-bbbbb", + "namespace": fn.REMOTE_NAMESPACE, + "labels": {"kubernetes.io/service-name": "prod-ipv6-gateway-bbbbb"}, }, - ) + "addressType": "IPv6", + "ports": [{"name": "https", "port": 443}], + "endpoints": [{"addresses": ["2001:db8::1"], "conditions": {"ready": True}}], + }, + "cluster-name-prod-dns-gateway-ccccc": { + "apiVersion": "v1", + "kind": "Service", + "metadata": {"name": "prod-dns-gateway-ccccc", "namespace": fn.REMOTE_NAMESPACE}, + "spec": {"type": "ExternalName", "externalName": "lb-x.elb.amazonaws.com"}, + }, + }, ( + "IP clusters get a headless Service + EndpointSlice, the hostname cluster an ExternalName, " + "and the own cluster with nothing published gets neither" + ) - async def test_resolves_each_cluster_gateway_name(self) -> None: - """A Service per cluster gateway, resolving its name to its address here. - - A ModelService's backends address a cluster gateway by the name - compose-inference-cluster derived, and this gateway's Envoy resolves it, - so its cluster needs a Service of that name. An IP is served by a - headless Service and an EndpointSlice; a hostname, which is how a cloud - load balancer names itself, by an ExternalName Service. An IP also gets - the Endpoints the slice supersedes, because kube-dns reads only that and - is what GKE runs. A cluster that hasn't published both an address and a - name gets neither. - """ - ipv4 = "prod-ipv4-gateway-aaaaa.modelplane-system.svc.cluster.local" - ipv6 = "prod-ipv6-gateway-bbbbb.modelplane-system.svc.cluster.local" - dns = "prod-dns-gateway-ccccc.modelplane-system.svc.cluster.local" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required( - gateways=[_gateway_xr("eu", _CLUSTER)], - clusters=[ - _cluster(), # this gateway's own cluster, no gateway published yet - _cluster_with_gateway("prod-ipv4", address="203.0.113.7", hostname=ipv4), - _cluster_with_gateway("prod-ipv6", address="2001:db8::1", hostname=ipv6), - _cluster_with_gateway("prod-dns", address="lb-x.elb.amazonaws.com", hostname=dns), - ], - ), - ) - got = await self.runner.RunFunction(req, None) - resolvers = { - key: resource.struct_to_dict(res.resource) - for key, res in got.desired.resources.items() - if key.startswith("cluster-name") - } - for key, obj in resolvers.items(): - self.assertEqual( - obj["spec"]["providerConfigRef"], - {"kind": "ClusterProviderConfig", "name": _PC}, - f"{key} is composed against this gateway's own cluster", - ) - manifests = {key: obj["spec"]["forProvider"]["manifest"] for key, obj in resolvers.items()} - self.assertEqual( - manifests, - { - "cluster-name-prod-ipv4-gateway-aaaaa": { - "apiVersion": "v1", - "kind": "Service", - "metadata": {"name": "prod-ipv4-gateway-aaaaa", "namespace": fn.REMOTE_NAMESPACE}, - "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, - }, - "cluster-name-endpoints-prod-ipv4-gateway-aaaaa": { - "apiVersion": "v1", - "kind": "Endpoints", - "metadata": { - "name": "prod-ipv4-gateway-aaaaa", - "namespace": fn.REMOTE_NAMESPACE, - # Off, or the mirroring controller writes a second - # EndpointSlice over the one composed beside this. - "labels": {"endpointslice.kubernetes.io/skip-mirror": "true"}, - }, - "subsets": [{"addresses": [{"ip": "203.0.113.7"}], "ports": [{"name": "https", "port": 443}]}], - }, - "cluster-name-slice-prod-ipv4-gateway-aaaaa": { - "apiVersion": "discovery.k8s.io/v1", - "kind": "EndpointSlice", - "metadata": { - "name": "prod-ipv4-gateway-aaaaa", - "namespace": fn.REMOTE_NAMESPACE, - "labels": {"kubernetes.io/service-name": "prod-ipv4-gateway-aaaaa"}, - }, - "addressType": "IPv4", - "ports": [{"name": "https", "port": 443}], - "endpoints": [{"addresses": ["203.0.113.7"], "conditions": {"ready": True}}], - }, - "cluster-name-prod-ipv6-gateway-bbbbb": { - "apiVersion": "v1", - "kind": "Service", - "metadata": {"name": "prod-ipv6-gateway-bbbbb", "namespace": fn.REMOTE_NAMESPACE}, - "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, - }, - "cluster-name-endpoints-prod-ipv6-gateway-bbbbb": { - "apiVersion": "v1", - "kind": "Endpoints", - "metadata": { - "name": "prod-ipv6-gateway-bbbbb", - "namespace": fn.REMOTE_NAMESPACE, - # Off, or the mirroring controller writes a second - # EndpointSlice over the one composed beside this. - "labels": {"endpointslice.kubernetes.io/skip-mirror": "true"}, - }, - "subsets": [{"addresses": [{"ip": "2001:db8::1"}], "ports": [{"name": "https", "port": 443}]}], - }, - "cluster-name-slice-prod-ipv6-gateway-bbbbb": { - "apiVersion": "discovery.k8s.io/v1", - "kind": "EndpointSlice", - "metadata": { - "name": "prod-ipv6-gateway-bbbbb", - "namespace": fn.REMOTE_NAMESPACE, - "labels": {"kubernetes.io/service-name": "prod-ipv6-gateway-bbbbb"}, - }, - "addressType": "IPv6", - "ports": [{"name": "https", "port": 443}], - "endpoints": [{"addresses": ["2001:db8::1"], "conditions": {"ready": True}}], - }, - "cluster-name-prod-dns-gateway-ccccc": { - "apiVersion": "v1", - "kind": "Service", - "metadata": {"name": "prod-dns-gateway-ccccc", "namespace": fn.REMOTE_NAMESPACE}, - "spec": {"type": "ExternalName", "externalName": "lb-x.elb.amazonaws.com"}, - }, +def test_certificate_common_names_fit_the_x509_limit() -> None: + """A long gateway name must not push a certificate commonName past the + 64-byte X.509 limit, which cert-manager's webhook rejects. A gateway name + is a cluster-scoped resource name, so it can be up to 253 characters.""" + long_name = "g" + "a" * 62 + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(name=long_name)))), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr(long_name, _CLUSTER)]), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + manifest = resource.struct_to_dict(got.desired.resources["client-ca-certificate"].resource)["spec"]["forProvider"][ + "manifest" + ] + cn = manifest["spec"]["commonName"] + assert len(cn.encode()) <= 64, "client-ca-certificate commonName exceeds the 64-byte X.509 limit" + + +def test_a_rejected_caller_policy_is_not_ready() -> None: + """A gateway whose caller policy was rejected refuses every request with + a 500 while its Gateway still has an address. Envoy Gateway rejects the + policy when two selected Secrets share a key value, so this is reachable + by writing two Secrets.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth()))), + # The Gateway is programmed; the policy is not accepted. + resources={"gateway": _observed_gateway(_ADDRESS, ready=True)}, + ), + required_resources=_required( + clusters=[_cluster()], + gateways=[_gateway_xr("eu", _CLUSTER)], + **{"caller-secrets": [_secret("ml-team-keys", {"a": "sk-1"})]}, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + cond = next(iter(got.conditions)) + assert cond.status == fnv1.STATUS_CONDITION_FALSE + assert cond.reason == fn.CONDITION_REASON_AUTH_NOT_ACCEPTED + + +SHARED_CALLER_KEY_CASES = [ + ( + "two Secrets share a key", + [_secret("team-a-keys", {"a": "sk-1"}), _secret("team-b-keys", {"b": "sk-1"})], + fnv1.Condition( + type=fn.CONDITION_TYPE_GATEWAY_READY, + status=fnv1.STATUS_CONDITION_FALSE, + reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, + message="Caller team-b-keys/b has the same key as team-a-keys/a, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ), + ), + ( + "one Secret shares a key", + [_secret("team-a-keys", {"a": "sk-1", "z": "sk-1"})], + fnv1.Condition( + type=fn.CONDITION_TYPE_GATEWAY_READY, + status=fnv1.STATUS_CONDITION_FALSE, + reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, + message="Caller team-a-keys/z has the same key as team-a-keys/a, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ), + ), + ( + # Listed out of order. Walked in name order, team-a's x is seen + # first, so team-b's x is skipped and its y repeats x's key. + # Walked as listed, team-a's x would be the skipped one. + "Secrets are walked in name order", + [_secret("team-b-keys", {"x": "sk-2", "y": "sk-1"}), _secret("team-a-keys", {"x": "sk-1"})], + fnv1.Condition( + type=fn.CONDITION_TYPE_GATEWAY_READY, + status=fnv1.STATUS_CONDITION_FALSE, + reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, + message="Caller team-b-keys/y has the same key as team-a-keys/x, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ), + ), + ( + "a repeated caller name is skipped, whatever its key", + [_secret("team-a-keys", {"a": "sk-1"}), _secret("team-b-keys", {"a": "sk-1", "b": "sk-2"})], + fnv1.Condition( + type=fn.CONDITION_TYPE_GATEWAY_READY, + status=fnv1.STATUS_CONDITION_TRUE, + reason=fn.CONDITION_REASON_GATEWAY_PROGRAMMED, + ), + ), +] + + +@pytest.mark.parametrize("case", SHARED_CALLER_KEY_CASES, ids=lambda case: case[0]) +def test_a_shared_caller_key_is_not_ready_before_the_policy_is_observed( + case: tuple[str, list[dict], fnv1.Condition], +) -> None: + """Envoy Gateway rejects the caller policy when two callers share a key, + but the policy's Object still reads as accepted until provider-kubernetes + next observes it. The gateway reports the outage from the Secrets + themselves, and skips a repeated caller name before comparing its key, + as Envoy Gateway does.""" + _, secrets, want = case + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth()))), + resources={ + "gateway": _observed_gateway(_ADDRESS, ready=True), + "caller-auth": _observed_accepted(), }, - "IP clusters get a headless Service + EndpointSlice, the hostname cluster an ExternalName, " - "and the own cluster with nothing published gets neither", - ) - - async def test_certificate_common_names_fit_the_x509_limit(self) -> None: - """A long gateway name must not push a certificate commonName past the - 64-byte X.509 limit, which cert-manager's webhook rejects. A gateway name - is a cluster-scoped resource name, so it can be up to 253 characters.""" - long_name = "g" + "a" * 62 - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(name=long_name)))), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr(long_name, _CLUSTER)]), - ) - got = await self.runner.RunFunction(req, None) - manifest = resource.struct_to_dict(got.desired.resources["client-ca-certificate"].resource)["spec"][ - "forProvider" - ]["manifest"] - cn = manifest["spec"]["commonName"] - self.assertLessEqual(len(cn.encode()), 64, "client-ca-certificate commonName exceeds the 64-byte X.509 limit") - - async def test_a_rejected_caller_policy_is_not_ready(self) -> None: - """A gateway whose caller policy was rejected refuses every request with - a 500 while its Gateway still has an address. Envoy Gateway rejects the - policy when two selected Secrets share a key value, so this is reachable - by writing two Secrets.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth()))), - # The Gateway is programmed; the policy is not accepted. - resources={"gateway": _observed_gateway(_ADDRESS, ready=True)}, - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{"caller-secrets": [_secret("ml-team-keys", {"a": "sk-1"})]}, - ), - ) - got = await self.runner.RunFunction(req, None) - cond = next(iter(got.conditions)) - self.assertEqual(cond.status, fnv1.STATUS_CONDITION_FALSE) - self.assertEqual(cond.reason, fn.CONDITION_REASON_AUTH_NOT_ACCEPTED) - - async def test_a_shared_caller_key_is_not_ready_before_the_policy_is_observed(self) -> None: - """Envoy Gateway rejects the caller policy when two callers share a key, - but the policy's Object still reads as accepted until provider-kubernetes - next observes it. The gateway reports the outage from the Secrets - themselves, and skips a repeated caller name before comparing its key, - as Envoy Gateway does.""" - cases = [ - ( - "two Secrets share a key", - [_secret("team-a-keys", {"a": "sk-1"}), _secret("team-b-keys", {"b": "sk-1"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, - message="Caller team-b-keys/b has the same key as team-a-keys/a, so Envoy Gateway rejects the " - "caller authentication policy and every request is refused", - ), - ), - ( - "one Secret shares a key", - [_secret("team-a-keys", {"a": "sk-1", "z": "sk-1"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, - message="Caller team-a-keys/z has the same key as team-a-keys/a, so Envoy Gateway rejects the " - "caller authentication policy and every request is refused", - ), - ), - ( - # Listed out of order. Walked in name order, team-a's x is seen - # first, so team-b's x is skipped and its y repeats x's key. - # Walked as listed, team-a's x would be the skipped one. - "Secrets are walked in name order", - [_secret("team-b-keys", {"x": "sk-2", "y": "sk-1"}), _secret("team-a-keys", {"x": "sk-1"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, - message="Caller team-b-keys/y has the same key as team-a-keys/x, so Envoy Gateway rejects the " - "caller authentication policy and every request is refused", - ), - ), - ( - "a repeated caller name is skipped, whatever its key", - [_secret("team-a-keys", {"a": "sk-1"}), _secret("team-b-keys", {"a": "sk-1", "b": "sk-2"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_TRUE, - reason=fn.CONDITION_REASON_GATEWAY_PROGRAMMED, - ), - ), - ] - for name, secrets, want in cases: - with self.subTest(name): - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth()))), - resources={ - "gateway": _observed_gateway(_ADDRESS, ready=True), - "caller-auth": _observed_accepted(), - }, - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{"caller-secrets": secrets}, - ), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual( - [json_format.MessageToDict(c) for c in got.conditions], - [json_format.MessageToDict(want)], - ) - - async def test_caller_secrets_are_listed_in_name_order(self) -> None: - """Envoy Gateway keeps the first Secret listed when two hold the same - caller name, so the policy lists them by name rather than in the order - they resolved in, and the winner doesn't change between reconciles.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth())))), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{ - "caller-secrets": [ - _secret("team-b-keys", {"b": "sk-2"}), - _secret("team-a-keys", {"a": "sk-1"}), - ] - }, - ), - ) - got = await self.runner.RunFunction(req, None) - policy = resource.struct_to_dict(got.desired.resources["caller-auth"].resource) - self.assertEqual( - policy["spec"]["forProvider"]["manifest"]["spec"]["apiKeyAuth"]["credentialRefs"], - [{"name": "callers-team-a-keys"}, {"name": "callers-team-b-keys"}], - ) - - async def test_a_missing_caller_secret_denies_but_keeps_the_gateway(self) -> None: - """Auth is asked for but no caller Secret has resolved. The Gateway is - still composed, so its load balancer and address survive, and its caller - policy denies every request rather than leaving the door open. Two states - reach this, the selector matching no Secret and the requirement not having - resolved yet, differing only in the reason reported.""" - for name, extra, message in [ - ( - "the selector matches no Secret", - {"caller-secrets": []}, - "spec.auth.apiKey.secretSelector matches no Secret, so no caller could authenticate", - ), - ( - "the caller Secrets have not resolved yet", - {}, - "Waiting for caller key Secrets to resolve", - ), - ]: - with self.subTest(name): - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth()))) - ), - required_resources=_required( - clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)], **extra - ), - ) - got = await self.runner.RunFunction(req, None) + ), + required_resources=_required( + clusters=[_cluster()], + gateways=[_gateway_xr("eu", _CLUSTER)], + **{"caller-secrets": secrets}, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert [_to_dict(c) for c in got.conditions] == [_to_dict(want)] + + +def test_caller_secrets_are_listed_in_name_order() -> None: + """Envoy Gateway keeps the first Secret listed when two hold the same + caller name, so the policy lists them by name rather than in the order + they resolved in, and the winner doesn't change between reconciles.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth())))), + required_resources=_required( + clusters=[_cluster()], + gateways=[_gateway_xr("eu", _CLUSTER)], + **{ + "caller-secrets": [ + _secret("team-b-keys", {"b": "sk-2"}), + _secret("team-a-keys", {"a": "sk-1"}), + ] + }, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + policy = resource.struct_to_dict(got.desired.resources["caller-auth"].resource) + assert policy["spec"]["forProvider"]["manifest"]["spec"]["apiKeyAuth"]["credentialRefs"] == [ + {"name": "callers-team-a-keys"}, + {"name": "callers-team-b-keys"}, + ] + + +MISSING_CALLER_SECRET_CASES = [ + ( + "the selector matches no Secret", + {"caller-secrets": []}, + "spec.auth.apiKey.secretSelector matches no Secret, so no caller could authenticate", + ), + ( + "the caller Secrets have not resolved yet", + {}, + "Waiting for caller key Secrets to resolve", + ), +] + + +@pytest.mark.parametrize("case", MISSING_CALLER_SECRET_CASES, ids=lambda case: case[0]) +def test_a_missing_caller_secret_denies_but_keeps_the_gateway(case: tuple[str, dict, str]) -> None: + """Auth is asked for but no caller Secret has resolved. The Gateway is + still composed, so its load balancer and address survive, and its caller + policy denies every request rather than leaving the door open. Two states + reach this, the selector matching no Secret and the requirement not having + resolved yet, differing only in the reason reported.""" + _, extra, want_message = case + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth())))), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)], **extra), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - self.assertIn("gateway", got.desired.resources, "the Gateway is kept, so its address survives") - self.assertFalse( - any(key.startswith("caller-secret-") for key in got.desired.resources), - "no caller Secret resolved, so none is copied to the cluster", + assert "gateway" in got.desired.resources, "the Gateway is kept, so its address survives" + assert not any(key.startswith("caller-secret-") for key in got.desired.resources), ( + "no caller Secret resolved, so none is copied to the cluster" + ) + spec = resource.struct_to_dict(got.desired.resources["caller-auth"].resource)["spec"]["forProvider"]["manifest"][ + "spec" + ] + assert spec == { + "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": fn._GATEWAY_NAME}], + "authorization": {"defaultAction": "Deny"}, + }, "with no caller key the policy denies every request rather than authenticating nobody by omission" + cond = next(iter(got.conditions)) + assert cond.status == fnv1.STATUS_CONDITION_FALSE + assert cond.reason == fn.CONDITION_REASON_SECRETS_MISSING + assert cond.message == want_message + + +def test_a_missing_tls_secret_keeps_the_gateway() -> None: + """A referenced TLS Secret hasn't resolved. The Gateway is still composed, + so its address survives; the HTTPS listener is left without a certificate + on the cluster until the Secret appears, rather than the whole Gateway + withdrawn and its load balancer moved.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr(tls=v1alpha1.Tls(certificateRefs=[v1alpha1.CertificateRef(name="eu-tls-0")])) ) - spec = resource.struct_to_dict(got.desired.resources["caller-auth"].resource)["spec"]["forProvider"][ - "manifest" - ]["spec"] - self.assertEqual( - spec, + ) + ), + required_resources=_required( + clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)], **{"tls-secret-0": []} + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + assert "gateway" in got.desired.resources, "the Gateway is kept, so its address survives" + assert "tls-secret-eu-tls-0" not in got.desired.resources, "the missing Secret isn't copied to the cluster" + listeners = resource.struct_to_dict(got.desired.resources["gateway"].resource)["spec"]["forProvider"]["manifest"][ + "spec" + ]["listeners"] + assert [ln["name"] for ln in listeners] == ["http", "https"], "the HTTPS listener is still declared" + cond = next(iter(got.conditions)) + assert cond.status == fnv1.STATUS_CONDITION_FALSE + assert cond.reason == fn.CONDITION_REASON_SECRETS_MISSING + assert cond.message == "Waiting for TLS Secrets: eu-tls-0" + + +def test_the_incumbent_keeps_its_cluster() -> None: + """A gateway created later must not take a cluster off one already + serving traffic. Doing so would delete the incumbent's Gateway and bring + its load balancer back on a different address.""" + # "aaa" sorts before "zzz" but "zzz" already has an address. + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( { - "targetRefs": [ - {"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": fn._GATEWAY_NAME} - ], - "authorization": {"defaultAction": "Deny"}, - }, - "with no caller key the policy denies every request rather than authenticating nobody by omission", - ) - cond = next(iter(got.conditions)) - self.assertEqual(cond.status, fnv1.STATUS_CONDITION_FALSE) - self.assertEqual(cond.reason, fn.CONDITION_REASON_SECRETS_MISSING) - self.assertEqual(cond.message, message) - - async def test_a_missing_tls_secret_keeps_the_gateway(self) -> None: - """A referenced TLS Secret hasn't resolved. The Gateway is still composed, - so its address survives; the HTTPS listener is left without a certificate - on the cluster until the Secret appears, rather than the whole Gateway - withdrawn and its load balancer moved.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr(tls=v1alpha1.Tls(certificateRefs=[v1alpha1.CertificateRef(name="eu-tls-0")])) - ) + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceGateway", + "metadata": {"name": "aaa"}, + "spec": {"clusterName": _CLUSTER}, + } ) - ), - required_resources=_required( - clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)], **{"tls-secret-0": []} - ), - ) - got = await self.runner.RunFunction(req, None) - - self.assertIn("gateway", got.desired.resources, "the Gateway is kept, so its address survives") - self.assertNotIn("tls-secret-eu-tls-0", got.desired.resources, "the missing Secret isn't copied to the cluster") - listeners = resource.struct_to_dict(got.desired.resources["gateway"].resource)["spec"]["forProvider"][ - "manifest" - ]["spec"]["listeners"] - self.assertEqual([ln["name"] for ln in listeners], ["http", "https"], "the HTTPS listener is still declared") - cond = next(iter(got.conditions)) - self.assertEqual(cond.status, fnv1.STATUS_CONDITION_FALSE) - self.assertEqual(cond.reason, fn.CONDITION_REASON_SECRETS_MISSING) - self.assertEqual(cond.message, "Waiting for TLS Secrets: eu-tls-0") - - async def test_the_incumbent_keeps_its_cluster(self) -> None: - """A gateway created later must not take a cluster off one already - serving traffic. Doing so would delete the incumbent's Gateway and bring - its load balancer back on a different address.""" - # "aaa" sorts before "zzz" but "zzz" already has an address. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceGateway", - "metadata": {"name": "aaa"}, - "spec": {"clusterName": _CLUSTER}, - } + ) + ), + required_resources=_required( + clusters=[_cluster()], + gateways=[ + _gateway_xr("aaa", _CLUSTER), + {**_gateway_xr("zzz", _CLUSTER), "status": {"address": _ADDRESS}}, + ], + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert len(got.desired.resources) == 0, "the newcomer composes nothing" + cond = next(iter(got.conditions)) + assert cond.reason == fn.CONDITION_REASON_CLUSTER_TAKEN + assert "zzz" in cond.message + + +def test_no_composed_object_observes_a_secret() -> None: + """No composed Object reads a Secret, which is what keeps this gateway's + client CA private key off the control plane. + + provider-kubernetes copies an observed object's whole manifest into the + Object's status, and its --sanitize-secrets flag defaults to false, so + observing a Secret publishes every key in it to anyone who can get + objects. This CA signs the certificate every cluster gateway in the fleet + accepts, so leaking its key means anyone can reach any engine. + + Asserted over everything composed rather than over the PKI, because the + cost of reintroducing this anywhere is the same. + + Observing is the case that matters here. The Secrets this function + *writes* also end up in status, because provider-kubernetes reports what + it observes of what it manages, so this alone doesn't keep their contents + off the control plane. Those hold caller keys and serving certificates + that came from control-plane Secrets to begin with, so the exposure is a + wider audience for data already present rather than data that would + otherwise never be there, and prerequisites.yaml runs + provider-kubernetes with --sanitize-secrets to redact it. A CA private + key is different in kind: it is generated on the workload cluster and + observing it is the only way it could ever reach the control plane. + """ + # Auth and TLS both on, so the Secret-copying path is exercised: without + # them this function composes no Secret at all and the assertion holds + # vacuously. + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr( + tls={"certificateRefs": [{"name": "eu-tls-0"}]}, + auth={"method": "APIKey", "apiKey": {"secretSelector": {"matchLabels": {"team": "ml"}}}}, ) ) ), - required_resources=_required( - clusters=[_cluster()], - gateways=[ - _gateway_xr("aaa", _CLUSTER), - {**_gateway_xr("zzz", _CLUSTER), "status": {"address": _ADDRESS}}, - ], - ), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual(len(got.desired.resources), 0, "the newcomer composes nothing") - cond = next(iter(got.conditions)) - self.assertEqual(cond.reason, fn.CONDITION_REASON_CLUSTER_TAKEN) - self.assertIn("zzz", cond.message) - - async def test_no_composed_object_observes_a_secret(self) -> None: - """No composed Object reads a Secret, which is what keeps this gateway's - client CA private key off the control plane. - - provider-kubernetes copies an observed object's whole manifest into the - Object's status, and its --sanitize-secrets flag defaults to false, so - observing a Secret publishes every key in it to anyone who can get - objects. This CA signs the certificate every cluster gateway in the fleet - accepts, so leaking its key means anyone can reach any engine. - - Asserted over everything composed rather than over the PKI, because the - cost of reintroducing this anywhere is the same. - - Observing is the case that matters here. The Secrets this function - *writes* also end up in status, because provider-kubernetes reports what - it observes of what it manages, so this alone doesn't keep their contents - off the control plane. Those hold caller keys and serving certificates - that came from control-plane Secrets to begin with, so the exposure is a - wider audience for data already present rather than data that would - otherwise never be there, and prerequisites.yaml runs - provider-kubernetes with --sanitize-secrets to redact it. A CA private - key is different in kind: it is generated on the workload cluster and - observing it is the only way it could ever reach the control plane. - """ - # Auth and TLS both on, so the Secret-copying path is exercised: without - # them this function composes no Secret at all and the assertion holds - # vacuously. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr( - tls={"certificateRefs": [{"name": "eu-tls-0"}]}, - auth={"method": "APIKey", "apiKey": {"secretSelector": {"matchLabels": {"team": "ml"}}}}, - ) - ) - ), - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{ - "caller-secrets": [_secret("ml-team-keys", {"alice": "key"})], - "tls-secret-0": [_secret("eu-tls-0", {"tls.crt": "cert", "tls.key": "key"})], - }, - ), - ) - got = await self.runner.RunFunction(req, None) - - composed_secrets = [] - observed_secrets = [] - for key, res in got.desired.resources.items(): - d = resource.struct_to_dict(res.resource) - manifest = d["spec"]["forProvider"]["manifest"] - if manifest["kind"] != "Secret": - continue - composed_secrets.append(key) - if "Observe" in d["spec"].get("managementPolicies", []): - observed_secrets.append(key) - self.assertEqual(observed_secrets, [], "these observe a Secret, so its private keys reach the control plane") - self.assertNotEqual(composed_secrets, [], "no Secret composed, so the assertion above proves nothing") - - async def test_client_pki_publishes_the_ca_without_its_key(self) -> None: - """The client CA's certificate reaches the control plane through a - trust-manager Bundle, which copies one named key into a ConfigMap, rather - than through the Secret that also holds the private key.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = await self.runner.RunFunction(req, None) + ), + required_resources=_required( + clusters=[_cluster()], + gateways=[_gateway_xr("eu", _CLUSTER)], + **{ + "caller-secrets": [_secret("ml-team-keys", {"alice": "key"})], + "tls-secret-0": [_secret("eu-tls-0", {"tls.crt": "cert", "tls.key": "key"})], + }, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + composed_secrets = [] + observed_secrets = [] + for key, res in got.desired.resources.items(): + d = resource.struct_to_dict(res.resource) + manifest = d["spec"]["forProvider"]["manifest"] + if manifest["kind"] != "Secret": + continue + composed_secrets.append(key) + if "Observe" in d["spec"].get("managementPolicies", []): + observed_secrets.append(key) + assert observed_secrets == [], "these observe a Secret, so its private keys reach the control plane" + assert composed_secrets != [], "no Secret composed, so the assertion above proves nothing" + + +def test_client_pki_publishes_the_ca_without_its_key() -> None: + """The client CA's certificate reaches the control plane through a + trust-manager Bundle, which copies one named key into a ConfigMap, rather + than through the Secret that also holds the private key.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - def manifest(key: str) -> dict: - return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] + def manifest(key: str) -> dict: + return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] - self.assertEqual( - manifest("client-ca-bundle"), - { - "apiVersion": "trust.cert-manager.io/v1alpha1", - "kind": "Bundle", - "metadata": {"name": "inference-gateway-ca"}, - "spec": { - "sources": [{"secret": {"name": "inference-gateway-ca", "key": "ca.crt"}}], - "target": { - "configMap": {"key": "ca.crt"}, - "namespaceSelector": {"matchLabels": {"kubernetes.io/metadata.name": "modelplane-system"}}, - }, - }, - }, - ) - # Named after the Bundle, because that's the ConfigMap a Bundle syncs. - self.assertEqual( - manifest("client-ca-configmap"), - { - "apiVersion": "v1", - "kind": "ConfigMap", - "metadata": {"name": "inference-gateway-ca", "namespace": "modelplane-system"}, + assert manifest("client-ca-bundle") == { + "apiVersion": "trust.cert-manager.io/v1alpha1", + "kind": "Bundle", + "metadata": {"name": "inference-gateway-ca"}, + "spec": { + "sources": [{"secret": {"name": "inference-gateway-ca", "key": "ca.crt"}}], + "target": { + "configMap": {"key": "ca.crt"}, + "namespaceSelector": {"matchLabels": {"kubernetes.io/metadata.name": "modelplane-system"}}, }, - ) - self.assertEqual( - resource.struct_to_dict(got.desired.resources["client-ca-configmap"].resource)["spec"][ - "managementPolicies" - ], - ["Observe"], - "trust-manager owns this ConfigMap; Crossplane must not write it", - ) + }, + } + # Named after the Bundle, because that's the ConfigMap a Bundle syncs. + assert manifest("client-ca-configmap") == { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "inference-gateway-ca", "namespace": "modelplane-system"}, + } + assert resource.struct_to_dict(got.desired.resources["client-ca-configmap"].resource)["spec"][ + "managementPolicies" + ] == ["Observe"], "trust-manager owns this ConfigMap; Crossplane must not write it" + - async def test_client_ca_published_from_the_observed_configmap(self) -> None: - """status.clientCACertificate comes from the ConfigMap trust-manager - syncs, as plain text rather than base64. A cluster only trusts this - gateway once it has it, so nothing reaches an engine before it appears. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={ - "gateway": _observed_gateway("gw.example.org", ready=True), - "client-ca-configmap": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "status": { - "atProvider": { - "manifest": { - "apiVersion": "v1", - "kind": "ConfigMap", - "data": {"ca.crt": "-----BEGIN CERTIFICATE-----\nclient\n"}, - } +def test_client_ca_published_from_the_observed_configmap() -> None: + """status.clientCACertificate comes from the ConfigMap trust-manager + syncs, as plain text rather than base64. A cluster only trusts this + gateway once it has it, so nothing reaches an engine before it appears. + """ + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), + resources={ + "gateway": _observed_gateway("gw.example.org", ready=True), + "client-ca-configmap": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "status": { + "atProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "data": {"ca.crt": "-----BEGIN CERTIFICATE-----\nclient\n"}, } - }, - } - ), + } + }, + } ), - }, - ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = await self.runner.RunFunction(req, None) + ), + }, + ), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - self.assertEqual( - resource.struct_to_dict(got.desired.composite.resource)["status"]["clientCACertificate"], - "-----BEGIN CERTIFICATE-----\nclient\n", - ) + assert ( + resource.struct_to_dict(got.desired.composite.resource)["status"]["clientCACertificate"] + == "-----BEGIN CERTIFICATE-----\nclient\n" + ) - async def test_no_client_ca_before_the_bundle_syncs(self) -> None: - """With no observed ConfigMap the gateway publishes no CA, so no cluster - trusts it yet and no cluster publishes a hostname on its account.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={"gateway": _observed_gateway("gw.example.org", ready=True)}, - ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = await self.runner.RunFunction(req, None) - self.assertNotIn( - "clientCACertificate", - resource.struct_to_dict(got.desired.composite.resource)["status"], - ) +def test_no_client_ca_before_the_bundle_syncs() -> None: + """With no observed ConfigMap the gateway publishes no CA, so no cluster + trusts it yet and no cluster publishes a hostname on its account.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), + resources={"gateway": _observed_gateway("gw.example.org", ready=True)}, + ), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + assert "clientCACertificate" not in resource.struct_to_dict(got.desired.composite.resource)["status"] diff --git a/functions/compose-metric-mapping/tests/__init__.py b/functions/compose-metric-mapping/tests/__init__.py deleted file mode 100644 index ebf4b2ad4..000000000 --- a/functions/compose-metric-mapping/tests/__init__.py +++ /dev/null @@ -1,13 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. diff --git a/functions/compose-metric-mapping/tests/test_fn.py b/functions/compose-metric-mapping/tests/test_fn.py index 5399e5dd1..bf5acbb72 100644 --- a/functions/compose-metric-mapping/tests/test_fn.py +++ b/functions/compose-metric-mapping/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-metric-mapping function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb @@ -34,105 +36,96 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" +def _compose_cases() -> list[Case]: + mapping = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "MetricMapping", + "metadata": {"name": "my-engine"}, + "spec": { + "metrics": [{"from": "my_engine_queued", "to": "modelplane_requests_waiting"}], + }, + } + cluster = resource.dict_to_struct( + {"apiVersion": "modelplane.ai/v1alpha1", "kind": "InferenceCluster", "metadata": {"name": "prod-us-east"}} + ) - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function reports whether a mapping reaches any cluster.""" - mapping = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "MetricMapping", - "metadata": {"name": "my-engine"}, - "spec": { - "metrics": [{"from": "my_engine_queued", "to": "modelplane_requests_waiting"}], - }, - } - cluster = resource.dict_to_struct( - {"apiVersion": "modelplane.ai/v1alpha1", "kind": "InferenceCluster", "metadata": {"name": "prod-us-east"}} + def req(xr: dict, clusters: list | None) -> fnv1.RunFunctionRequest: + r = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(xr))), + ) + if clusters is not None: + r.required_resources["clusters"].items.extend([fnv1.Resource(resource=c) for c in clusters]) + return r + + def want(ready: fnv1.Ready, status: dict | None, cond: fnv1.Condition) -> fnv1.RunFunctionResponse: + composite = fnv1.Resource(ready=ready) + if status is not None: + composite.resource.CopyFrom(resource.dict_to_struct(status)) + return fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=composite), + conditions=[cond], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster") + } + ), ) - def req(xr: dict, clusters: list | None) -> fnv1.RunFunctionRequest: - r = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(xr))), - ) - if clusters is not None: - r.required_resources["clusters"].items.extend([fnv1.Resource(resource=c) for c in clusters]) - return r - - def want(ready: fnv1.Ready, status: dict | None, cond: fnv1.Condition) -> fnv1.RunFunctionResponse: - composite = fnv1.Resource(ready=ready) - if status is not None: - composite.resource.CopyFrom(resource.dict_to_struct(status)) - return fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=composite), - conditions=[cond], - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster") - } - ), - ) - - cases = [ - Case( - name="ready, and says how many clusters took the renames", - req=req(mapping, [cluster, cluster]), - want=want( - fnv1.READY_TRUE, - {"status": {"clusters": 2}}, - fnv1.Condition( - type="Accepted", - status=fnv1.STATUS_CONDITION_TRUE, - reason="Available", - message="Renaming 1 metric(s) on 2 inference cluster(s)", - ), + return [ + Case( + name="ready, and says how many clusters took the renames", + req=req(mapping, [cluster, cluster]), + want=want( + fnv1.READY_TRUE, + {"status": {"clusters": 2}}, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_TRUE, + reason="Available", + message="Renaming 1 metric(s) on 2 inference cluster(s)", ), ), - Case( - name="not ready when no cluster exists to render into", - req=req(mapping, []), - want=want( - fnv1.READY_FALSE, - {"status": {"clusters": 0}}, - fnv1.Condition( - type="Accepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoClusters", - message="No inference cluster to render these renames into", - ), + ), + Case( + name="not ready when no cluster exists to render into", + req=req(mapping, []), + want=want( + fnv1.READY_FALSE, + {"status": {"clusters": 0}}, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoClusters", + message="No inference cluster to render these renames into", ), ), - Case( - name="waits for the clusters to resolve", - req=req(mapping, None), - want=want( - fnv1.READY_FALSE, - None, - fnv1.Condition( - type="Accepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForClusters", - message="Waiting for the inference clusters to resolve", - ), + ), + Case( + name="waits for the clusters to resolve", + req=req(mapping, None), + want=want( + fnv1.READY_FALSE, + None, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForClusters", + message="Waiting for the inference clusters to resolve", ), ), - ] + ), + ] + - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """The function reports whether a mapping reaches any cluster.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-model-cache/tests/test_fn.py b/functions/compose-model-cache/tests/test_fn.py index 33e460630..b1f607ff5 100644 --- a/functions/compose-model-cache/tests/test_fn.py +++ b/functions/compose-model-cache/tests/test_fn.py @@ -14,16 +14,18 @@ """Tests for the compose-model-cache function.""" +import asyncio import dataclasses import datetime -import unittest +import json from typing import Any -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.modelcache import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -41,10 +43,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - # The XR used across cases: a HuggingFace ModelCache in the ml-team namespace. # Both the PVC and Job derive their names from # resource.child_name("modelcache", "ml-team", "qwen", ...). @@ -292,607 +290,597 @@ def _auth_object(pc: str) -> dict: } -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: # noqa: PLR0915 - """The function composes a ModelCache.""" - # --- Case 1: GKE cluster, first pass. Composes the RWX PVC + hydration - # Job per matched cluster; nothing observed yet so phase is Pending and - # ArtifactReady is Hydrating. Emits the one-time "Staging" event. --- - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Pending"}], - }, +def _compose_cases() -> list[Case]: + """The cases for test_compose, built by code that derives them from shared parts.""" + # --- Case 1: GKE cluster, first pass. Composes the RWX PVC + hydration + # Job per matched cluster; nothing observed yet so phase is Pending and + # ArtifactReady is Hydrating. Emits the one-time "Staging" event. --- + want1 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "0/1"}, + "clusters": [{"name": "cluster-a", "phase": "Pending"}], }, - ), + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), - "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), - }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - want1.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 2: GKE cluster with a pinned revision + auth secret. The Job - # command gains --revision, and the function propagates the token to a - # workload-cluster Secret (auth-cluster-a) whose name the Job's HF_TOKEN - # env references - not the user's control-plane Secret name. --- - xr2 = _cache_xr(revision="main", authSecret=v1alpha1.AuthSecret(name="hf-token")) - env2 = [ - { - "name": "HF_TOKEN", - "valueFrom": {"secretKeyRef": {"name": _AUTH_NAME, "key": "HF_TOKEN"}}, + resources={ + "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), + "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), }, - ] - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Pending"}], - }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + ) + want1.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + + # --- Case 2: GKE cluster with a pinned revision + auth secret. The Job + # command gains --revision, and the function propagates the token to a + # workload-cluster Secret (auth-cluster-a) whose name the Job's HF_TOKEN + # env references - not the user's control-plane Secret name. --- + xr2 = _cache_xr(revision="main", authSecret=v1alpha1.AuthSecret(name="hf-token")) + env2 = [ + { + "name": "HF_TOKEN", + "valueFrom": {"secretKeyRef": {"name": _AUTH_NAME, "key": "HF_TOKEN"}}, + }, + ] + want2 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "0/1"}, + "clusters": [{"name": "cluster-a", "phase": "Pending"}], }, - ), + }, ), - resources={ - "auth-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_auth_object("cluster-a-pc"))), - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), - "hydrate-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct( - _job_object("cluster-a-pc", command=_HYDRATE_CMD_REVISION, env=env2), - ), - ), - }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - want2.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want2.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 3: EKS cluster reporting an EFS RWX class on status.cache. The - # PVC sources its storageClassName from status.cache (modelplane-rwx-efs), - # not the GKE/Filestore one. --- - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + resources={ + "auth-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_auth_object("cluster-a-pc"))), + "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), + "hydrate-cluster-a": fnv1.Resource( resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "eks-a", "phase": "Pending"}], - }, - }, + _job_object("cluster-a-pc", command=_HYDRATE_CMD_REVISION, env=env2), ), ), - resources={ - "pvc-eks-a": fnv1.Resource( - resource=resource.dict_to_struct( - _pvc_object("eks-a-pc", storage_class="modelplane-rwx-efs"), - ), - ), - "hydrate-eks-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("eks-a-pc"))), - }, + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: eks-a", - ), - ], - context=structpb.Struct(), - ) - want3.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 4: ready. The observed PVC + Job Objects each carry their own - # Ready condition (from DeriveFromCelQuery), and the wrapped manifest - # status shows PVC Bound + Job succeeded. Phase Ready, both Objects - # marked ready, summary 1/1, XR ready, ArtifactReady Staged. The - # already-composed PVC suppresses the "Staging" event; the - # not-previously-ready -> ready transition emits the "staged" event. --- - observed4 = { - "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), - } - pvc_ready = _pvc_object("cluster-a-pc") - want4 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, + ], + context=structpb.Struct(), + ) + want2.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + want2.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) + + # --- Case 3: EKS cluster reporting an EFS RWX class on status.cache. The + # PVC sources its storageClassName from status.cache (modelplane-rwx-efs), + # not the GKE/Filestore one. --- + want3 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "0/1"}, + "clusters": [{"name": "eks-a", "phase": "Pending"}], }, - ), - ready=fnv1.READY_TRUE, + }, ), - resources={ - # Job dropped once Ready; only the PVC remains composed. - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(pvc_ready), ready=fnv1.READY_TRUE), - }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - results=[ - fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), - ], - context=structpb.Struct(), - ) - want4.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 5: hydrating. PVC Bound (Object Ready) but the Job hasn't - # completed, so phase is Hydrating, only the PVC is marked ready, summary - # 0/1, and the XR is not ready. No transition event fires. --- - observed5 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} - want5 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + resources={ + "pvc-eks-a": fnv1.Resource( resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Hydrating"}], - }, - }, + _pvc_object("eks-a-pc", storage_class="modelplane-rwx-efs"), ), ), - resources={ - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE - ), - "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), - }, + "hydrate-eks-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("eks-a-pc"))), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: eks-a", ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - context=structpb.Struct(), - ) - want5.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 6: failed. The Job reports a Failed condition (and is NOT - # Ready). A Failed Job takes precedence over PVC binding, so phase is - # Failed, only the PVC is marked ready, summary 0/1, XR not ready, and - # ArtifactReady is False with reason Failed. --- - observed6 = { - "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Failed", "status": "True"}]}), - } - want6 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Failed"}], - }, + ], + context=structpb.Struct(), + ) + want3.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + + # --- Case 4: ready. The observed PVC + Job Objects each carry their own + # Ready condition (from DeriveFromCelQuery), and the wrapped manifest + # status shows PVC Bound + Job succeeded. Phase Ready, both Objects + # marked ready, summary 1/1, XR ready, ArtifactReady Staged. The + # already-composed PVC suppresses the "Staging" event; the + # not-previously-ready -> ready transition emits the "staged" event. --- + observed4 = { + "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), + "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), + } + pvc_ready = _pvc_object("cluster-a-pc") + want4 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "1/1"}, + "clusters": [{"name": "cluster-a", "phase": "Ready"}], }, - ), + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE - ), - "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), - }, + ready=fnv1.READY_TRUE, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Failed"), - ], - context=structpb.Struct(), - ) - want6.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 7: partial (1/2). Cluster a is Ready (PVC Bound + Job - # succeeded), cluster b is still Hydrating (PVC Bound only). Summary - # 1/2, ArtifactReady False with reason Partial, XR not ready. --- - observed7 = { - "pvc-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), - "pvc-b": _observed_object({"phase": "Bound"}, ready=True), - } - want7 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/2"}, - "clusters": [ - {"name": "a", "phase": "Ready"}, - {"name": "b", "phase": "Hydrating"}, - ], - }, + resources={ + # Job dropped once Ready; only the PVC remains composed. + "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(pvc_ready), ready=fnv1.READY_TRUE), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), + ], + context=structpb.Struct(), + ) + want4.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + + # --- Case 5: hydrating. PVC Bound (Object Ready) but the Job hasn't + # completed, so phase is Hydrating, only the PVC is marked ready, summary + # 0/1, and the XR is not ready. No transition event fires. --- + observed5 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} + want5 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "0/1"}, + "clusters": [{"name": "cluster-a", "phase": "Hydrating"}], }, - ), + }, ), - resources={ - # Cluster a is Ready, so its Job is dropped; b is still Hydrating. - "pvc-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("a-pc")), ready=fnv1.READY_TRUE - ), - "pvc-b": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("b-pc")), ready=fnv1.READY_TRUE - ), - "hydrate-b": fnv1.Resource(resource=resource.dict_to_struct(_job_object("b-pc"))), - }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Partial"), - ], - context=structpb.Struct(), - ) - want7.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 8: latch. A previously-Ready cluster whose hydration Job was - # dropped (and TTL-cleaned), so only the PVC is observed now. The status - # latch keeps phase Ready and the Job is not re-composed, PVC marked - # ready, summary 1/1, XR ready. Already-ready, so no transition event. --- - xr_ready = _cache_xr() - xr_ready.status = v1alpha1.Status( - summary=v1alpha1.Summary(ready="1/1"), - clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], - conditions=[ - v1alpha1.Condition( - type="Ready", - status="True", - reason="Available", - lastTransitionTime=_TRANSITION_TIME, + resources={ + "pvc-cluster-a": fnv1.Resource( + resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE ), - ], - ) - observed8 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} - want8 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, + "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + context=structpb.Struct(), + ) + want5.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + + # --- Case 6: failed. The Job reports a Failed condition (and is NOT + # Ready). A Failed Job takes precedence over PVC binding, so phase is + # Failed, only the PVC is marked ready, summary 0/1, XR not ready, and + # ArtifactReady is False with reason Failed. --- + observed6 = { + "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), + "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Failed", "status": "True"}]}), + } + want6 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "0/1"}, + "clusters": [{"name": "cluster-a", "phase": "Failed"}], }, - ), - ready=fnv1.READY_TRUE, + }, ), - resources={ - # Latched Ready with the Job already dropped: only the PVC. - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE - ), - }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - context=structpb.Struct(), - ) - want8.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 9: authSecret referenced but not yet resolved. The function - # requires both the clusters and the auth Secret, then returns early - # (no resources, status, or conditions) until Crossplane resolves the - # Secret and re-calls it. --- - xr9 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - want9 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(), - context=structpb.Struct(), - ) - want9.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want9.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 10: authSecret resolved but the Secret lacks the referenced - # key (here it carries OTHER, not HF_TOKEN). The PVC still composes - it - # doesn't depend on the token, so a cache isn't pruned for a missing one - # - but the hydration Job and token Secret are held back. ArtifactReady - # is False with reason AuthSecretMissing, and a warning names the Secret - # and key so the user can fix it instead of seeing the XR stall. The XR - # is marked not ready, since the PVC alone would make it ready once it - # binds. --- - want10 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Pending"}], - }, + resources={ + "pvc-cluster-a": fnv1.Resource( + resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE + ), + "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Failed"), + ], + context=structpb.Struct(), + ) + want6.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + + # --- Case 7: partial (1/2). Cluster a is Ready (PVC Bound + Job + # succeeded), cluster b is still Hydrating (PVC Bound only). Summary + # 1/2, ArtifactReady False with reason Partial, XR not ready. --- + observed7 = { + "pvc-a": _observed_object({"phase": "Bound"}, ready=True), + "hydrate-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), + "pvc-b": _observed_object({"phase": "Bound"}, ready=True), + } + want7 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "1/2"}, + "clusters": [ + {"name": "a", "phase": "Ready"}, + {"name": "b", "phase": "Hydrating"}, + ], }, - ), - ready=fnv1.READY_FALSE, + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), - }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="AuthSecretMissing"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_WARNING, - message="authSecret ml-team/hf-token is missing or has no key 'HF_TOKEN'", - ), - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - want10.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want10.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 11: Ready cluster with an authSecret. The token is only needed - # while hydrating, so once the cluster is Ready the auth Secret is dropped - # alongside the Job (only the PVC remains composed), even though the - # control-plane Secret still resolves. Keeps the token from lingering on - # the inference cluster after hydration. --- - xr11 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - observed11 = { - "auth-cluster-a": _observed_object({}, ready=True), - "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), - } - want11 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, + resources={ + # Cluster a is Ready, so its Job is dropped; b is still Hydrating. + "pvc-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("a-pc")), ready=fnv1.READY_TRUE), + "pvc-b": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("b-pc")), ready=fnv1.READY_TRUE), + "hydrate-b": fnv1.Resource(resource=resource.dict_to_struct(_job_object("b-pc"))), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Partial"), + ], + context=structpb.Struct(), + ) + want7.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + + # --- Case 8: latch. A previously-Ready cluster whose hydration Job was + # dropped (and TTL-cleaned), so only the PVC is observed now. The status + # latch keeps phase Ready and the Job is not re-composed, PVC marked + # ready, summary 1/1, XR ready. Already-ready, so no transition event. --- + xr_ready = _cache_xr() + xr_ready.status = v1alpha1.Status( + summary=v1alpha1.Summary(ready="1/1"), + clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], + conditions=[ + v1alpha1.Condition( + type="Ready", + status="True", + reason="Available", + lastTransitionTime=_TRANSITION_TIME, + ), + ], + ) + observed8 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} + want8 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "1/1"}, + "clusters": [{"name": "cluster-a", "phase": "Ready"}], }, - ), - ready=fnv1.READY_TRUE, + }, ), - resources={ - # Ready: the auth Secret and Job are both dropped, only the PVC remains. - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE - ), - }, + ready=fnv1.READY_TRUE, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - results=[ - fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), - ], - context=structpb.Struct(), - ) - want11.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want11.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 12: token rotated away after a cache is Ready. A latched-Ready - # cluster whose authSecret now resolves without the key. The PVC keeps - # composing (and stays Ready via the status latch) rather than being - # pruned, and because hydration is already done the missing token is - # neither reported (ArtifactReady stays Staged) nor warned. --- - xr12 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - xr12.status = v1alpha1.Status( - summary=v1alpha1.Summary(ready="1/1"), - clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], - conditions=[ - v1alpha1.Condition( - type="Ready", - status="True", - reason="Available", - lastTransitionTime=_TRANSITION_TIME, + resources={ + # Latched Ready with the Job already dropped: only the PVC. + "pvc-cluster-a": fnv1.Resource( + resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE ), - ], - ) - observed12 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} - want12 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + context=structpb.Struct(), + ) + want8.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + + # --- Case 9: authSecret referenced but not yet resolved. The function + # requires both the clusters and the auth Secret, then returns early + # (no resources, status, or conditions) until Crossplane resolves the + # Secret and re-calls it. --- + xr9 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) + want9 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(), + context=structpb.Struct(), + ) + want9.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + want9.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) + + # --- Case 10: authSecret resolved but the Secret lacks the referenced + # key (here it carries OTHER, not HF_TOKEN). The PVC still composes - it + # doesn't depend on the token, so a cache isn't pruned for a missing one + # - but the hydration Job and token Secret are held back. ArtifactReady + # is False with reason AuthSecretMissing, and a warning names the Secret + # and key so the user can fix it instead of seeing the XR stall. The XR + # is marked not ready, since the PVC alone would make it ready once it + # binds. --- + want10 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "0/1"}, + "clusters": [{"name": "cluster-a", "phase": "Pending"}], }, - ), - ready=fnv1.READY_TRUE, - ), - resources={ - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE - ), - }, - ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - context=structpb.Struct(), - ) - want12.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want12.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 13: authSecret missing AND no clusters matched. NoClusters is - # the dominant signal - the cache can't progress regardless of the token - # - so both conditions report NoClusters and the missing token is neither - # reported nor warned. --- - xr13 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - want13 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"summary": {"ready": "0/0"}, "clusters": []}}), - ready=fnv1.READY_FALSE, + }, ), + ready=fnv1.READY_FALSE, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), - ], - context=structpb.Struct(), - ) - want13.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want13.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - cases = [ - Case( - name="GKE cluster first pass composes RWX PVC and hydration Job", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")]), - want=want1, - ), - Case( - name="HuggingFace revision and auth secret wire --revision and HF_TOKEN", - req=_req(xr2, [_cluster_dict("cluster-a", "cluster-a-pc")], auth=_auth_secret()), - want=want2, - ), - Case( - name="EKS cluster PVC sources the EFS class from status.cache", - req=_req(_cache_xr(), [_cluster_dict("eks-a", "eks-a-pc", source="EKS")]), - want=want3, - ), - Case( - name="PVC bound and Job complete reports Ready", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed4), - want=want4, - ), - Case( - name="PVC bound but Job running reports Hydrating", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed5), - want=want5, - ), - Case( - name="failed Job reports Failed and takes precedence over PVC binding", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed6), - want=want6, - ), - Case( - name="one of two clusters ready reports partial", - req=_req(_cache_xr(), [_cluster_dict("a", "a-pc"), _cluster_dict("b", "b-pc")], observed7), - want=want7, - ), - Case( - name="hydrated cluster stays Ready after its Job is TTL-cleaned", - req=_req(xr_ready, [_cluster_dict("cluster-a", "cluster-a-pc")], observed8), - want=want8, + resources={ + "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="AuthSecretMissing"), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="authSecret ml-team/hf-token is missing or has no key 'HF_TOKEN'", ), - Case( - name="authSecret unresolved requires it and returns early", - req=_req(xr9, [_cluster_dict("cluster-a", "cluster-a-pc")]), - want=want9, + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", ), - Case( - name="authSecret resolved without the referenced key composes PVC and warns", - req=_req( - xr9, - [_cluster_dict("cluster-a", "cluster-a-pc")], - auth=_auth_secret(data={"OTHER": _TOKEN_B64}), + ], + context=structpb.Struct(), + ) + want10.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + want10.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) + + # --- Case 11: Ready cluster with an authSecret. The token is only needed + # while hydrating, so once the cluster is Ready the auth Secret is dropped + # alongside the Job (only the PVC remains composed), even though the + # control-plane Secret still resolves. Keeps the token from lingering on + # the inference cluster after hydration. --- + xr11 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) + observed11 = { + "auth-cluster-a": _observed_object({}, ready=True), + "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), + "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), + } + want11 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "1/1"}, + "clusters": [{"name": "cluster-a", "phase": "Ready"}], + }, + }, ), - want=want10, + ready=fnv1.READY_TRUE, ), - Case( - name="authSecret resolved with an empty token value composes PVC and warns", - req=_req( - xr9, - [_cluster_dict("cluster-a", "cluster-a-pc")], - auth=_auth_secret(data={"HF_TOKEN": ""}), + resources={ + # Ready: the auth Secret and Job are both dropped, only the PVC remains. + "pvc-cluster-a": fnv1.Resource( + resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE ), - want=want10, + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), + ], + context=structpb.Struct(), + ) + want11.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + want11.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) + + # --- Case 12: token rotated away after a cache is Ready. A latched-Ready + # cluster whose authSecret now resolves without the key. The PVC keeps + # composing (and stays Ready via the status latch) rather than being + # pruned, and because hydration is already done the missing token is + # neither reported (ArtifactReady stays Staged) nor warned. --- + xr12 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) + xr12.status = v1alpha1.Status( + summary=v1alpha1.Summary(ready="1/1"), + clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], + conditions=[ + v1alpha1.Condition( + type="Ready", + status="True", + reason="Available", + lastTransitionTime=_TRANSITION_TIME, ), - Case( - name="Ready cluster drops the auth Secret with the Job", - req=_req( - xr11, - [_cluster_dict("cluster-a", "cluster-a-pc")], - observed11, - auth=_auth_secret(), + ], + ) + observed12 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} + want12 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "1/1"}, + "clusters": [{"name": "cluster-a", "phase": "Ready"}], + }, + }, ), - want=want11, + ready=fnv1.READY_TRUE, ), - Case( - name="token rotated away after Ready keeps the PVC and stays Ready", - req=_req( - xr12, - [_cluster_dict("cluster-a", "cluster-a-pc")], - observed12, - auth=_auth_secret(data={"OTHER": _TOKEN_B64}), + resources={ + "pvc-cluster-a": fnv1.Resource( + resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE ), - want=want12, + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + context=structpb.Struct(), + ) + want12.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + want12.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) + + # --- Case 13: authSecret missing AND no clusters matched. NoClusters is + # the dominant signal - the cache can't progress regardless of the token + # - so both conditions report NoClusters and the missing token is neither + # reported nor warned. --- + xr13 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) + want13 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"summary": {"ready": "0/0"}, "clusters": []}}), + ready=fnv1.READY_FALSE, + ), + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), + ], + context=structpb.Struct(), + ) + want13.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + want13.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) + + return [ + Case( + name="GKE cluster first pass composes RWX PVC and hydration Job", + req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")]), + want=want1, + ), + Case( + name="HuggingFace revision and auth secret wire --revision and HF_TOKEN", + req=_req(xr2, [_cluster_dict("cluster-a", "cluster-a-pc")], auth=_auth_secret()), + want=want2, + ), + Case( + name="EKS cluster PVC sources the EFS class from status.cache", + req=_req(_cache_xr(), [_cluster_dict("eks-a", "eks-a-pc", source="EKS")]), + want=want3, + ), + Case( + name="PVC bound and Job complete reports Ready", + req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed4), + want=want4, + ), + Case( + name="PVC bound but Job running reports Hydrating", + req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed5), + want=want5, + ), + Case( + name="failed Job reports Failed and takes precedence over PVC binding", + req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed6), + want=want6, + ), + Case( + name="one of two clusters ready reports partial", + req=_req(_cache_xr(), [_cluster_dict("a", "a-pc"), _cluster_dict("b", "b-pc")], observed7), + want=want7, + ), + Case( + name="hydrated cluster stays Ready after its Job is TTL-cleaned", + req=_req(xr_ready, [_cluster_dict("cluster-a", "cluster-a-pc")], observed8), + want=want8, + ), + Case( + name="authSecret unresolved requires it and returns early", + req=_req(xr9, [_cluster_dict("cluster-a", "cluster-a-pc")]), + want=want9, + ), + Case( + name="authSecret resolved without the referenced key composes PVC and warns", + req=_req( + xr9, + [_cluster_dict("cluster-a", "cluster-a-pc")], + auth=_auth_secret(data={"OTHER": _TOKEN_B64}), ), - Case( - name="authSecret missing with no clusters reports NoClusters not AuthSecretMissing", - req=_req(xr13, [], auth=_auth_secret(data={"OTHER": _TOKEN_B64})), - want=want13, + want=want10, + ), + Case( + name="authSecret resolved with an empty token value composes PVC and warns", + req=_req( + xr9, + [_cluster_dict("cluster-a", "cluster-a-pc")], + auth=_auth_secret(data={"HF_TOKEN": ""}), ), - ] + want=want10, + ), + Case( + name="Ready cluster drops the auth Secret with the Job", + req=_req( + xr11, + [_cluster_dict("cluster-a", "cluster-a-pc")], + observed11, + auth=_auth_secret(), + ), + want=want11, + ), + Case( + name="token rotated away after Ready keeps the PVC and stays Ready", + req=_req( + xr12, + [_cluster_dict("cluster-a", "cluster-a-pc")], + observed12, + auth=_auth_secret(data={"OTHER": _TOKEN_B64}), + ), + want=want12, + ), + Case( + name="authSecret missing with no clusters reports NoClusters not AuthSecretMissing", + req=_req(xr13, [], auth=_auth_secret(data={"OTHER": _TOKEN_B64})), + want=want13, + ), + ] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes a ModelCache's PVC and hydration Job per cluster and reports their progress.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-model-deployment/tests/__init__.py b/functions/compose-model-deployment/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-model-deployment/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-model-deployment/tests/test_cel.py b/functions/compose-model-deployment/tests/test_cel.py index 9b4979755..7a2e3fd9b 100644 --- a/functions/compose-model-deployment/tests/test_cel.py +++ b/functions/compose-model-deployment/tests/test_cel.py @@ -21,8 +21,8 @@ """ import dataclasses -import unittest +import pytest from function import cel @@ -86,237 +86,232 @@ class Case: ) -class TestMatches(unittest.TestCase): - def test_matches(self) -> None: - cases = [ - # driver. - Case(name="driver equals", expr='device.driver == "gpu.nvidia.com"', device=_GPU, want=True), - Case(name="driver not equals", expr='device.driver == "nic.nvidia.com"', device=_GPU, want=False), - # Quantity comparison + methods. - Case( - name="quantity compareTo ge", - expr=f'{_CAP}.memory.compareTo(quantity("141Gi")) >= 0', - device=_GPU, - want=True, - ), - Case( - name="quantity compareTo too big", - expr=f'{_CAP}.memory.compareTo(quantity("200Gi")) >= 0', - device=_GPU, - want=False, - ), - Case( - name="quantity isGreaterThan", - expr=f'{_CAP}.memory.isGreaterThan(quantity("80Gi"))', - device=_GPU, - want=True, - ), - Case( - name="quantity isLessThan", expr=f'{_CAP}.memory.isLessThan(quantity("200Gi"))', device=_GPU, want=True - ), - Case(name="quantity sign", expr=f"{_CAP}.memory.sign() == 1", device=_GPU, want=True), - Case(name="quantity asInteger", expr=f"{_CAP}.memory.asInteger() == {141 * 2**30}", device=_GPU, want=True), - Case(name="quantity isInteger", expr=f"{_CAP}.memory.isInteger()", device=_GPU, want=True), - Case( - name="quantity add", - expr=f'{_CAP}.memory.add(quantity("1Gi")).compareTo(quantity("142Gi")) == 0', - device=_GPU, - want=True, - ), - Case(name="isQuantity true", expr='isQuantity("1.3Gi")', device=_GPU, want=True), - Case(name="isQuantity false", expr='isQuantity("200K")', device=_GPU, want=False), - # Semver comparison + methods. - Case( - name="semver isGreaterThan", - expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.0.0"))', - device=_GPU, - want=True, - ), - Case( - name="semver not greater", - expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.9.0"))', - device=_GPU, - want=False, - ), - Case(name="semver major", expr=f"{_ATTR}.cudaComputeCapability.major() == 9", device=_GPU, want=True), - Case(name="semver minor", expr=f"{_ATTR}.cudaComputeCapability.minor() == 5", device=_GPU, want=True), - Case(name="semver patch", expr=f"{_ATTR}.cudaComputeCapability.patch() == 3", device=_GPU, want=True), - Case( - name="semver equality", expr=f'{_ATTR}.cudaComputeCapability == semver("9.5.3")', device=_GPU, want=True - ), - Case(name="isSemver strict true", expr='isSemver("1.0.0")', device=_GPU, want=True), - Case(name="isSemver strict rejects short", expr='isSemver("1.0")', device=_GPU, want=False), - Case(name="isSemver normalize accepts short", expr='isSemver("1.0", true)', device=_GPU, want=True), - Case(name="semver normalize overload", expr='semver("v1.0", true).major() == 1', device=_GPU, want=True), - # Typed scalar attributes (resolve straight to the value, no .string). - Case(name="string attribute", expr=f'{_ATTR}.architecture == "Hopper"', device=_GPU, want=True), - Case( - name="string attribute mismatch", - expr='device.attributes["nic.nvidia.com"].linkType == "infiniband"', - device=_NIC, - want=True, - ), - Case( - name="bool attribute true", - expr=f"{_ATTR}.x", - device=_device(attributes={"x": {"bool": True}}), - want=True, - ), - Case( - name="bool attribute false", - expr=f"{_ATTR}.x", - device=_device(attributes={"x": {"bool": False}}), - want=False, - ), - Case( - name="int attribute", - expr=f"{_ATTR}.x >= 8", - device=_device(attributes={"x": {"int": 8}}), - want=True, - ), - Case( - name="int attribute below", - expr=f"{_ATTR}.x >= 8", - device=_device(attributes={"x": {"int": 4}}), - want=False, - ), - # Qualified names split into their own domain. - Case( - name="qualified name under its domain", - expr='device.attributes["resource.kubernetes.io"].pcieRoot == "pci0"', - device=_GPU, - want=True, - ), - Case( - name="bare name under driver domain", expr=f'{_ATTR}.architecture == "Hopper"', device=_GPU, want=True - ), - # Non-matches that must not raise. - Case( - name="two-component version is non-match", - expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("8.0.0"))', - device=_device(attributes={"cudaComputeCapability": {"version": "9.0"}}), - want=False, - ), - Case( - name="malformed quantity is non-match", - expr=f'{_CAP}.memory.compareTo(quantity("1Gi")) >= 0', - device=_device(capacity={"memory": {"value": "10Mo"}}), - want=False, - ), - Case(name="unknown id is non-match", expr=f'{_ATTR}.nope == "x"', device=_GPU, want=False), - # A non-bool selector must not spuriously match. Upstream rejects it - # at compile time; we treat a non-bool result as a non-match. - Case(name="non-bool string selector is non-match", expr='"5"', device=_GPU, want=False), - Case( - name="non-bool int selector is non-match", - expr=f"{_ATTR}.x", - device=_device(attributes={"x": {"int": 5}}), - want=False, - ), - # Domain presence. Upstream's domain-presence idiom is "" in - # device.attributes, not has(device.attributes[""]): cel-go's - # has() macro rejects an index argument, so the has() form is a - # compile error on a real cluster (celpy accepts it - see cel.py's - # documented divergences). An unknown domain is simply absent (False), - # not present-but-empty. - Case(name="unknown domain absent", expr='"other.com" in device.attributes', device=_GPU, want=False), - Case(name="known domain present", expr='"gpu.nvidia.com" in device.attributes', device=_GPU, want=True), - # Reading an unknown domain resolves to an empty map (not an error), - # so an id lookup under it is a non-match rather than a failure. - Case( - name="unknown domain id is non-match", - expr='device.attributes["other.com"].x == "y"', - device=_GPU, - want=False, - ), - # Guard a domain read with the in idiom before indexing it. - Case( - name="guarded known domain", - expr=f'"gpu.nvidia.com" in device.attributes && {_ATTR}.architecture == "Hopper"', - device=_GPU, - want=True, - ), - # The full design selector. - Case( - name="full design expression", - expr=( - f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.0.0")) && ' - f'{_CAP}.memory.compareTo(quantity("141Gi")) >= 0' - ), - device=_GPU, - want=True, - ), - # Verbatim selector examples from the DRA docs, each against a device - # that should and should not match. - Case( - name="docs: large-black subrequest matches", - expr=( - 'device.attributes["resource-driver.example.com"].color == "black" && ' - 'device.attributes["resource-driver.example.com"].size == "large"' - ), - device=_LARGE_BLACK, - want=True, - ), - Case( - name="docs: large-black subrequest rejects small-white", - expr=( - 'device.attributes["resource-driver.example.com"].color == "black" && ' - 'device.attributes["resource-driver.example.com"].size == "large"' - ), - device=_SMALL_WHITE, - want=False, - ), - Case( - name="docs: small-white subrequest matches", - expr=( - 'device.attributes["resource-driver.example.com"].color == "white" && ' - 'device.attributes["resource-driver.example.com"].size == "small"' - ), - device=_SMALL_WHITE, - want=True, - ), - Case( - name="docs: extended-resource DeviceClass selector matches", - expr="device.driver == 'gpu.example.com' && device.attributes['gpu.example.com'].type == 'gpu'", - device=_EXAMPLE_GPU, - want=True, - ), - Case( - name="docs: extended-resource DeviceClass selector rejects other driver", - expr="device.driver == 'gpu.example.com' && device.attributes['gpu.example.com'].type == 'gpu'", - device=_NIC, - want=False, - ), - Case( - name="docs: ResourceClaim type+memory selector matches", - expr=( - 'device.attributes["driver.example.com"].type == "gpu" && ' - 'device.capacity["driver.example.com"].memory == quantity("64Gi")' - ), - device=_GPU_64GI, - want=True, - ), - Case( - name="docs: ResourceClaim type+memory selector rejects wrong memory", - expr=( - 'device.attributes["driver.example.com"].type == "gpu" && ' - 'device.capacity["driver.example.com"].memory == quantity("64Gi")' - ), - device=_device( - driver="driver.example.com", - attributes={"type": {"string": "gpu"}}, - capacity={"memory": {"value": "32Gi"}}, - ), - want=False, - ), - ] - for case in cases: - with self.subTest(case.name): - got = cel.Program(case.expr).matches(case.device) - self.assertEqual(case.want, got, f"{case.name}: -want, +got") +MATCHES_CASES = [ + # driver. + Case(name="driver equals", expr='device.driver == "gpu.nvidia.com"', device=_GPU, want=True), + Case(name="driver not equals", expr='device.driver == "nic.nvidia.com"', device=_GPU, want=False), + # Quantity comparison + methods. + Case( + name="quantity compareTo ge", + expr=f'{_CAP}.memory.compareTo(quantity("141Gi")) >= 0', + device=_GPU, + want=True, + ), + Case( + name="quantity compareTo too big", + expr=f'{_CAP}.memory.compareTo(quantity("200Gi")) >= 0', + device=_GPU, + want=False, + ), + Case( + name="quantity isGreaterThan", + expr=f'{_CAP}.memory.isGreaterThan(quantity("80Gi"))', + device=_GPU, + want=True, + ), + Case(name="quantity isLessThan", expr=f'{_CAP}.memory.isLessThan(quantity("200Gi"))', device=_GPU, want=True), + Case(name="quantity sign", expr=f"{_CAP}.memory.sign() == 1", device=_GPU, want=True), + Case(name="quantity asInteger", expr=f"{_CAP}.memory.asInteger() == {141 * 2**30}", device=_GPU, want=True), + Case(name="quantity isInteger", expr=f"{_CAP}.memory.isInteger()", device=_GPU, want=True), + Case( + name="quantity add", + expr=f'{_CAP}.memory.add(quantity("1Gi")).compareTo(quantity("142Gi")) == 0', + device=_GPU, + want=True, + ), + Case(name="isQuantity true", expr='isQuantity("1.3Gi")', device=_GPU, want=True), + Case(name="isQuantity false", expr='isQuantity("200K")', device=_GPU, want=False), + # Semver comparison + methods. + Case( + name="semver isGreaterThan", + expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.0.0"))', + device=_GPU, + want=True, + ), + Case( + name="semver not greater", + expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.9.0"))', + device=_GPU, + want=False, + ), + Case(name="semver major", expr=f"{_ATTR}.cudaComputeCapability.major() == 9", device=_GPU, want=True), + Case(name="semver minor", expr=f"{_ATTR}.cudaComputeCapability.minor() == 5", device=_GPU, want=True), + Case(name="semver patch", expr=f"{_ATTR}.cudaComputeCapability.patch() == 3", device=_GPU, want=True), + Case(name="semver equality", expr=f'{_ATTR}.cudaComputeCapability == semver("9.5.3")', device=_GPU, want=True), + Case(name="isSemver strict true", expr='isSemver("1.0.0")', device=_GPU, want=True), + Case(name="isSemver strict rejects short", expr='isSemver("1.0")', device=_GPU, want=False), + Case(name="isSemver normalize accepts short", expr='isSemver("1.0", true)', device=_GPU, want=True), + Case(name="semver normalize overload", expr='semver("v1.0", true).major() == 1', device=_GPU, want=True), + # Typed scalar attributes (resolve straight to the value, no .string). + Case(name="string attribute", expr=f'{_ATTR}.architecture == "Hopper"', device=_GPU, want=True), + Case( + name="string attribute mismatch", + expr='device.attributes["nic.nvidia.com"].linkType == "infiniband"', + device=_NIC, + want=True, + ), + Case( + name="bool attribute true", + expr=f"{_ATTR}.x", + device=_device(attributes={"x": {"bool": True}}), + want=True, + ), + Case( + name="bool attribute false", + expr=f"{_ATTR}.x", + device=_device(attributes={"x": {"bool": False}}), + want=False, + ), + Case( + name="int attribute", + expr=f"{_ATTR}.x >= 8", + device=_device(attributes={"x": {"int": 8}}), + want=True, + ), + Case( + name="int attribute below", + expr=f"{_ATTR}.x >= 8", + device=_device(attributes={"x": {"int": 4}}), + want=False, + ), + # Qualified names split into their own domain. + Case( + name="qualified name under its domain", + expr='device.attributes["resource.kubernetes.io"].pcieRoot == "pci0"', + device=_GPU, + want=True, + ), + Case(name="bare name under driver domain", expr=f'{_ATTR}.architecture == "Hopper"', device=_GPU, want=True), + # Non-matches that must not raise. + Case( + name="two-component version is non-match", + expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("8.0.0"))', + device=_device(attributes={"cudaComputeCapability": {"version": "9.0"}}), + want=False, + ), + Case( + name="malformed quantity is non-match", + expr=f'{_CAP}.memory.compareTo(quantity("1Gi")) >= 0', + device=_device(capacity={"memory": {"value": "10Mo"}}), + want=False, + ), + Case(name="unknown id is non-match", expr=f'{_ATTR}.nope == "x"', device=_GPU, want=False), + # A non-bool selector must not spuriously match. Upstream rejects it + # at compile time; we treat a non-bool result as a non-match. + Case(name="non-bool string selector is non-match", expr='"5"', device=_GPU, want=False), + Case( + name="non-bool int selector is non-match", + expr=f"{_ATTR}.x", + device=_device(attributes={"x": {"int": 5}}), + want=False, + ), + # Domain presence. Upstream's domain-presence idiom is "" in + # device.attributes, not has(device.attributes[""]): cel-go's + # has() macro rejects an index argument, so the has() form is a + # compile error on a real cluster (celpy accepts it - see cel.py's + # documented divergences). An unknown domain is simply absent (False), + # not present-but-empty. + Case(name="unknown domain absent", expr='"other.com" in device.attributes', device=_GPU, want=False), + Case(name="known domain present", expr='"gpu.nvidia.com" in device.attributes', device=_GPU, want=True), + # Reading an unknown domain resolves to an empty map (not an error), + # so an id lookup under it is a non-match rather than a failure. + Case( + name="unknown domain id is non-match", + expr='device.attributes["other.com"].x == "y"', + device=_GPU, + want=False, + ), + # Guard a domain read with the in idiom before indexing it. + Case( + name="guarded known domain", + expr=f'"gpu.nvidia.com" in device.attributes && {_ATTR}.architecture == "Hopper"', + device=_GPU, + want=True, + ), + # The full design selector. + Case( + name="full design expression", + expr=( + f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.0.0")) && ' + f'{_CAP}.memory.compareTo(quantity("141Gi")) >= 0' + ), + device=_GPU, + want=True, + ), + # Verbatim selector examples from the DRA docs, each against a device + # that should and should not match. + Case( + name="docs: large-black subrequest matches", + expr=( + 'device.attributes["resource-driver.example.com"].color == "black" && ' + 'device.attributes["resource-driver.example.com"].size == "large"' + ), + device=_LARGE_BLACK, + want=True, + ), + Case( + name="docs: large-black subrequest rejects small-white", + expr=( + 'device.attributes["resource-driver.example.com"].color == "black" && ' + 'device.attributes["resource-driver.example.com"].size == "large"' + ), + device=_SMALL_WHITE, + want=False, + ), + Case( + name="docs: small-white subrequest matches", + expr=( + 'device.attributes["resource-driver.example.com"].color == "white" && ' + 'device.attributes["resource-driver.example.com"].size == "small"' + ), + device=_SMALL_WHITE, + want=True, + ), + Case( + name="docs: extended-resource DeviceClass selector matches", + expr="device.driver == 'gpu.example.com' && device.attributes['gpu.example.com'].type == 'gpu'", + device=_EXAMPLE_GPU, + want=True, + ), + Case( + name="docs: extended-resource DeviceClass selector rejects other driver", + expr="device.driver == 'gpu.example.com' && device.attributes['gpu.example.com'].type == 'gpu'", + device=_NIC, + want=False, + ), + Case( + name="docs: ResourceClaim type+memory selector matches", + expr=( + 'device.attributes["driver.example.com"].type == "gpu" && ' + 'device.capacity["driver.example.com"].memory == quantity("64Gi")' + ), + device=_GPU_64GI, + want=True, + ), + Case( + name="docs: ResourceClaim type+memory selector rejects wrong memory", + expr=( + 'device.attributes["driver.example.com"].type == "gpu" && ' + 'device.capacity["driver.example.com"].memory == quantity("64Gi")' + ), + device=_device( + driver="driver.example.com", + attributes={"type": {"string": "gpu"}}, + capacity={"memory": {"value": "32Gi"}}, + ), + want=False, + ), +] -class TestCompile(unittest.TestCase): - def test_invalid_expression_raises(self) -> None: - with self.assertRaises(cel.CELCompileError): - cel.Program("not ) valid (") +@pytest.mark.parametrize("case", MATCHES_CASES, ids=lambda case: case.name) +def test_matches(case: Case) -> None: + """A DRA CEL selector matches a device as it does upstream.""" + got = cel.Program(case.expr).matches(case.device) + assert got == case.want + + +def test_compile_invalid_expression_raises() -> None: + """A malformed expression fails to compile.""" + with pytest.raises(cel.CELCompileError, match=r"not \) valid \("): + cel.Program("not ) valid (") diff --git a/functions/compose-model-deployment/tests/test_fn.py b/functions/compose-model-deployment/tests/test_fn.py index 5e1e4dfc9..f937959ab 100644 --- a/functions/compose-model-deployment/tests/test_fn.py +++ b/functions/compose-model-deployment/tests/test_fn.py @@ -14,16 +14,18 @@ """Tests for the compose-model-deployment function.""" +import asyncio import dataclasses import datetime -import unittest +import json from typing import Any -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 from models.ai.modelplane.modelcache import v1alpha1 as mcv1alpha1 @@ -405,1134 +407,1121 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) +def _compose_cases() -> list[Case]: + """The cases for test_compose, and the deployments they share.""" + # A deployment that sets spec.modelCacheRef. + xr_cached = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=1, + template=v1alpha1.TemplateModel( + spec=v1alpha1.SpecModel( + modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), + engines=[_ENGINE], + ) + ), + ), + ).model_dump(exclude_none=True, mode="json") + # A cached deployment that also sets its own clusterSelector, so the + # scheduler intersects it with the cache's footprint. + xr_cached_selector = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=1, + template=v1alpha1.TemplateModel( + spec=v1alpha1.SpecModel( + clusterSelector=v1alpha1.ClusterSelector(matchLabels={"region": "us-east"}), + modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), + engines=[_ENGINE], + ) + ), + ), + ).model_dump(exclude_none=True, mode="json") -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" + # A two-replica deployment (no container args) for the co-location case. + xr_two = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=2, + template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE_NO_ARGS])), + ), + ).model_dump(exclude_none=True, mode="json") - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() + # A disaggregated (PrefillDecode) deployment: a Prefill and a Decode engine. + xr_pd = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=1, + template=v1alpha1.TemplateModel( + spec=v1alpha1.SpecModel( + serving=v1alpha1.Serving(mode="PrefillDecode"), + engines=[ + _ENGINE.model_copy(update={"name": "prefill", "phase": "Prefill"}), + _ENGINE.model_copy(update={"name": "decode", "phase": "Decode"}), + ], + ) + ), + ), + ).model_dump(exclude_none=True, mode="json") - async def test_compose(self) -> None: - """The function fans out ModelReplicas and, once they're Ready, ModelEndpoints.""" + # A deployment parked at zero replicas. + xr_zero = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=0, + template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE])), + ), + ).model_dump(exclude_none=True, mode="json") - # A deployment that sets spec.modelCacheRef. - xr_cached = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - spec=v1alpha1.SpecModel( - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), - engines=[_ENGINE], - ) - ), + return [ + Case( + # First reconcile: the replica is composed but not yet observed + # Ready, so its endpoint is withheld - routing must not advertise + # a backend whose pods are still warming up (#102). + name="freshly scheduled replica composes no endpoint until ready", + req=_req(_XR, clusters=[_CLUSTER_A]), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + }, + "spec": { + "clusterName": "cluster-a", + "engines": _REPLICA_ENGINES, + }, + } + ), + ), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + ) ), - ).model_dump(exclude_none=True, mode="json") - - # A cached deployment that also sets its own clusterSelector, so the - # scheduler intersects it with the cache's footprint. - xr_cached_selector = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - spec=v1alpha1.SpecModel( - clusterSelector=v1alpha1.ClusterSelector(matchLabels={"region": "us-east"}), - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), - engines=[_ENGINE], - ) + ), + Case( + # A replica that has gone not-Ready (e.g. a crash-loop after + # once serving) has its endpoint withdrawn: the previously + # observed endpoint is absent from desired, so Crossplane + # deletes it and traffic stops routing to the dead backend + # (#102). Omitting it from desired - not composing it - is what + # drives the deletion. + name="not-ready replica withdraws its endpoint", + req=_req( + _XR, + clusters=[_CLUSTER_A], + replicas=[_EXISTING_REPLICA], + observed={ + "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=False), + "endpoint-cluster-a-0": { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, + }, + }, + ), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + }, + "spec": { + "clusterName": "cluster-a", + "engines": _REPLICA_ENGINES, + }, + } + ), + ), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + context=structpb.Struct(), + ) + ), + ), + Case( + name="no clusters produces warning", + req=_req(_XR, clusters=[]), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoClusters", + ), + ], + results=[ + fnv1.Result(severity=fnv1.SEVERITY_WARNING, message="No InferenceClusters found"), + ], + context=structpb.Struct(), + ) + ), + ), + Case( + name="insufficient capacity produces no replicas", + req=_req(_XR, clusters=[_cluster("cluster-a", nodes=0)]), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), + ready=fnv1.READY_FALSE, + ), + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="InsufficientCapacity", + message="0 of 1 replicas scheduled (checked 1 clusters)", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReplicasScheduled", + ), + ], + context=structpb.Struct(), + ) + ), + ), + Case( + # Zero desired parks the deployment before resolve_inputs runs: + # no requirements are declared (the want carries none), nothing + # is composed, and both conditions read True with the + # NoReplicasDesired reason rather than a capacity failure. + name="scaled to zero composes nothing and reports NoReplicasDesired", + req=_req(xr_zero), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), + ready=fnv1.READY_TRUE, + ), ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + ], + context=structpb.Struct(), ), - ).model_dump(exclude_none=True, mode="json") - - # A two-replica deployment (no container args) for the co-location case. - xr_two = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=2, - template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE_NO_ARGS])), + ), + Case( + # Scaling an existing deployment to zero: the observed replica + # and endpoint are absent from desired (pruned), and the + # transition is announced while they still exist. + name="scale to zero prunes observed replicas and emits an event", + req=_req( + xr_zero, + observed={ + "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True), + "endpoint-cluster-a-0": { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, + }, + }, ), - ).model_dump(exclude_none=True, mode="json") - - # A disaggregated (PrefillDecode) deployment: a Prefill and a Decode engine. - xr_pd = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - spec=v1alpha1.SpecModel( - serving=v1alpha1.Serving(mode="PrefillDecode"), - engines=[ - _ENGINE.model_copy(update={"name": "prefill", "phase": "Prefill"}), - _ENGINE.model_copy(update={"name": "decode", "phase": "Decode"}), - ], - ) + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), + ready=fnv1.READY_TRUE, + ), ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scaled to zero: removing all replicas", + ), + ], + context=structpb.Struct(), ), - ).model_dump(exclude_none=True, mode="json") - - # A deployment parked at zero replicas. - xr_zero = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=0, - template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE])), + ), + Case( + name="ready replica is preserved and keeps its endpoint", + req=_req( + _XR, + clusters=[_CLUSTER_A], + replicas=[_EXISTING_REPLICA], + observed={ + "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True), + "endpoint-cluster-a-0": { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, + }, + }, ), - ).model_dump(exclude_none=True, mode="json") - - cases = [ - Case( - # First reconcile: the replica is composed but not yet observed - # Ready, so its endpoint is withheld - routing must not advertise - # a backend whose pods are still warming up (#102). - name="freshly scheduled replica composes no endpoint until ready", - req=_req(_XR, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 1}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + }, + "spec": { + "clusterName": "cluster-a", + "engines": _REPLICA_ENGINES, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "endpoint-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, + }, + "spec": { + "origin": "https://cluster.clusters.example.com", + "api": { + "schema": "OpenAI", + "prefix": "/ml-team/my-model-5ab63/v1", }, - } - ), + "model": "ml-team/my-model", + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="AllReplicasReady", + message="1 of 1 ready", + ), + ], + context=structpb.Struct(), + ) + ), + ), + Case( + name="offline pinned cluster keeps replica but drops endpoint", + req=_req( + _XR, + clusters=[_cluster("cluster-a", ready=False, hostname=None)], + replicas=[_EXISTING_REPLICA], + observed={"replica-cluster-a-0": _EXISTING_REPLICA}, + ), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + }, + "spec": { + "clusterName": "cluster-a", + "engines": _REPLICA_ENGINES, + }, + } + ), ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + context=structpb.Struct(), + ) + ), + ), + Case( + name="deleted pinned cluster triggers replica re-placement", + req=_req( + _XR, + clusters=[_cluster("cluster-b", hostname="cluster-b.clusters.example.com")], + replicas=[_EXISTING_REPLICA], + observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, + ), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-b-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-f0b76", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-b", + "modelplane.ai/replica-index": "0", + }, + }, + "spec": { + "clusterName": "cluster-b", + "engines": _REPLICA_ENGINES, + }, + } + ), ), - ], - context=structpb.Struct(), - ) - ), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-b", + ), + ], + context=structpb.Struct(), + ) ), - Case( - # A replica that has gone not-Ready (e.g. a crash-loop after - # once serving) has its endpoint withdrawn: the previously - # observed endpoint is absent from desired, so Crossplane - # deletes it and traffic stops routing to the dead backend - # (#102). Omitting it from desired - not composing it - is what - # drives the deletion. - name="not-ready replica withdraws its endpoint", - req=_req( - _XR, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={ - "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=False), - "endpoint-cluster-a-0": { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, + ), + Case( + name="modelCacheRef is propagated onto the composed replica", + req=_req(xr_cached, clusters=[_CLUSTER_A_CACHE], cache=_cache("qwen")), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + }, + "spec": { + "clusterName": "cluster-a", + "modelCacheRef": {"name": "qwen"}, + "engines": _REPLICA_ENGINES, + }, + } + ), + ), }, - }, + ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, + cache_name="qwen", + ), + ), + Case( + # The cache stages only to a subset of clusters; the scheduler + # intersects the cache's footprint with the deployment's own + # clusterSelector so replicas never land where the cache isn't. + name="cache clusterSelector is intersected with the deployment's", + req=_req( + xr_cached_selector, + clusters=[_CLUSTER_A_CACHE], + cache=_cache("qwen", match_labels={"tier": "gpu"}), + ), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, - } - ), + }, + "spec": { + "clusterName": "cluster-a", + "modelCacheRef": {"name": "qwen"}, + "engines": _REPLICA_ENGINES, + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", ), - ], - context=structpb.Struct(), - ) + }, + ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), ), + cluster_labels={"region": "us-east", "tier": "gpu"}, + cache_name="qwen", ), - Case( - name="no clusters produces warning", - req=_req(_XR, clusters=[]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoClusters", - ), - ], - results=[ - fnv1.Result(severity=fnv1.SEVERITY_WARNING, message="No InferenceClusters found"), - ], - context=structpb.Struct(), - ) + ), + Case( + # compose-model-cache stages only onto clusters that report cache + # storage, so cluster-a, which reports none, can't host the + # replica's PVC. The replica lands on cluster-b, though cluster-a + # would win the tiebreak by name. + name="a cached replica lands only on a cluster with cache storage", + req=_req( + xr_cached, + clusters=[_CLUSTER_A, _cluster("cluster-b", cache_storage=True)], + cache=_cache("qwen"), + ), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-b-0": fnv1.Resource(resource=resource.dict_to_struct(_CACHED_REPLICA_B)), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-b", + ), + ], + context=structpb.Struct(), ), + cache_name="qwen", ), - Case( - name="insufficient capacity produces no replicas", - req=_req(_XR, clusters=[_cluster("cluster-a", nodes=0)]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_FALSE, - ), + ), + Case( + # A replica is running on cluster-a, which has no cache storage, + # say because the bug this guards against put it there. Its PVC + # never appears, so it's dropped and re-placed on cluster-b, the + # way a replica is when the cache's selector stops matching. + name="a running cached replica on a cluster without cache storage is re-placed", + req=_req( + xr_cached, + clusters=[_CLUSTER_A, _cluster("cluster-b", cache_storage=True)], + replicas=[_EXISTING_REPLICA], + observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, + cache=_cache("qwen"), + ), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-b-0": fnv1.Resource(resource=resource.dict_to_struct(_CACHED_REPLICA_B)), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-b", ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="InsufficientCapacity", - message="0 of 1 replicas scheduled (checked 1 clusters)", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoReplicasScheduled", - ), - ], - context=structpb.Struct(), - ) + ], + context=structpb.Struct(), ), + cache_name="qwen", ), - Case( - # Zero desired parks the deployment before resolve_inputs runs: - # no requirements are declared (the want carries none), nothing - # is composed, and both conditions read True with the - # NoReplicasDesired reason rather than a capacity failure. - name="scaled to zero composes nothing and reports NoReplicasDesired", - req=_req(xr_zero), - want=fnv1.RunFunctionResponse( + ), + Case( + # The only candidate has no cache storage, so the cache can't + # stage there and nothing is placed. ReplicasScheduled says why + # rather than blaming capacity. + name="no candidate with cache storage places nothing", + req=_req(xr_cached, clusters=[_CLUSTER_A], cache=_cache("qwen")), + want=_want( + fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( composite=fnv1.Resource( resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_FALSE, ), ), conditions=[ fnv1.Condition( - type="ReplicasScheduled", + type="ModelCacheResolved", status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoCacheStorage", + message="0 of 1 replicas scheduled: no candidate cluster has storage for ModelCache qwen", ), fnv1.Condition( type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReplicasScheduled", ), ], context=structpb.Struct(), ), + cache_name="qwen", ), - Case( - # Scaling an existing deployment to zero: the observed replica - # and endpoint are absent from desired (pruned), and the - # transition is announced while they still exist. - name="scale to zero prunes observed replicas and emits an event", - req=_req( - xr_zero, - observed={ - "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True), - "endpoint-cluster-a-0": { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, - }, - }, - ), - want=fnv1.RunFunctionResponse( + ), + Case( + # A referenced cache Crossplane hasn't fetched yet leaves the + # footprint unknown. With no replicas to retain, the function + # holds off placing any rather than risk landing them outside the + # footprint: fill is suppressed, so nothing is composed, and + # ModelCacheResolved=False (Unresolved) says why. The wait is + # transient and self-clearing, so it's a condition, not an event. + # The cluster and replica requirements are still declared so the + # cache can resolve alongside them. + name="unresolved cache suppresses new placement", + req=_req(xr_cached, clusters=[_CLUSTER_A]), + want=_want( + fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( composite=fnv1.Resource( resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_FALSE, ), ), conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelCacheUnresolved", + message="Waiting for ModelCache qwen", + ), fnv1.Condition( type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", + status=fnv1.STATUS_CONDITION_FALSE, + reason="InsufficientCapacity", + message="0 of 1 replicas scheduled (checked 1 clusters)", ), fnv1.Condition( type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scaled to zero: removing all replicas", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReplicasScheduled", ), ], context=structpb.Struct(), ), + cache_name="qwen", ), - Case( - name="ready replica is preserved and keeps its endpoint", - req=_req( - _XR, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={ - "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True), - "endpoint-cluster-a-0": { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, - }, - }, - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 1}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "endpoint-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "origin": "https://cluster.clusters.example.com", - "api": { - "schema": "OpenAI", - "prefix": "/ml-team/my-model-5ab63/v1", - }, - "model": "ml-team/my-model", - }, - } - ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="AllReplicasReady", - message="1 of 1 ready", - ), - ], - context=structpb.Struct(), - ) - ), + ), + Case( + # The cache a live deployment depends on is deleted (the cache + # requirement resolves but matches nothing - ABSENT). The cache + # only matters when loading weights, which already happened, so + # its disappearance must not tear the deployment down: the + # existing replica is retained (retain ignores fill) even as + # ModelCacheResolved goes False (NotFound) and new placement is + # suppressed. + name="deleted cache retains existing replicas", + req=_req( + xr_cached, + clusters=[_CLUSTER_A], + replicas=[_EXISTING_REPLICA], + observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, + cache_resolved_empty=True, ), - Case( - name="offline pinned cluster keeps replica but drops endpoint", - req=_req( - _XR, - clusters=[_cluster("cluster-a", ready=False, hostname=None)], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _EXISTING_REPLICA}, - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 1}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, - } - ), + }, + "spec": { + "clusterName": "cluster-a", + "modelCacheRef": {"name": "qwen"}, + "engines": _REPLICA_ENGINES, + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - context=structpb.Struct(), - ) - ), - ), - Case( - name="deleted pinned cluster triggers replica re-placement", - req=_req( - _XR, - clusters=[_cluster("cluster-b", hostname="cluster-b.clusters.example.com")], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-b-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-f0b76", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-b", - "modelplane.ai/replica-index": "0", - }, + ready=fnv1.READY_TRUE, + ), + "endpoint-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, - "spec": { - "clusterName": "cluster-b", - "engines": _REPLICA_ENGINES, + }, + "spec": { + "origin": "https://cluster.clusters.example.com", + "api": { + "schema": "OpenAI", + "prefix": "/ml-team/my-model-5ab63/v1", }, - } - ), + "model": "ml-team/my-model", + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-b", ), - ], - context=structpb.Struct(), - ) + }, + ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelCacheNotFound", + message="ModelCache qwen not found; holding replica placement", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="AllReplicasReady", + message="1 of 1 ready", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="ModelCache qwen not found; holding replica placement", + ), + ], + context=structpb.Struct(), ), + cache_name="qwen", ), - Case( - name="modelCacheRef is propagated onto the composed replica", - req=_req(xr_cached, clusters=[_CLUSTER_A_CACHE], cache=_cache("qwen")), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "modelCacheRef": {"name": "qwen"}, - "engines": _REPLICA_ENGINES, + ), + Case( + name="two replicas co-locate on one cluster as distinct resources", + req=_req(xr_two, clusters=[_CLUSTER_A]), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 2, "ready": 0}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, - } - ), + }, + "spec": { + "clusterName": "cluster-a", + "engines": _REPLICA_ENGINES_NO_ARGS, + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ), - cache_name="qwen", - ), - ), - Case( - # The cache stages only to a subset of clusters; the scheduler - # intersects the cache's footprint with the deployment's own - # clusterSelector so replicas never land where the cache isn't. - name="cache clusterSelector is intersected with the deployment's", - req=_req( - xr_cached_selector, - clusters=[_CLUSTER_A_CACHE], - cache=_cache("qwen", match_labels={"tier": "gpu"}), - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "modelCacheRef": {"name": "qwen"}, - "engines": _REPLICA_ENGINES, + "replica-cluster-a-1": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-609c5", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "1", }, - } - ), + }, + "spec": { + "clusterName": "cluster-a", + "engines": _REPLICA_ENGINES_NO_ARGS, + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", ), - ], - context=structpb.Struct(), + }, ), - cluster_labels={"region": "us-east", "tier": "gpu"}, - cache_name="qwen", - ), - ), - Case( - # compose-model-cache stages only onto clusters that report cache - # storage, so cluster-a, which reports none, can't host the - # replica's PVC. The replica lands on cluster-b, though cluster-a - # would win the tiebreak by name. - name="a cached replica lands only on a cluster with cache storage", - req=_req( - xr_cached, - clusters=[_CLUSTER_A, _cluster("cluster-b", cache_storage=True)], - cache=_cache("qwen"), - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-b-0": fnv1.Resource( - resource=resource.dict_to_struct(_CACHED_REPLICA_B) - ), - }, + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-b", - ), - ], - context=structpb.Struct(), - ), - cache_name="qwen", - ), - ), - Case( - # A replica is running on cluster-a, which has no cache storage, - # say because the bug this guards against put it there. Its PVC - # never appears, so it's dropped and re-placed on cluster-b, the - # way a replica is when the cache's selector stops matching. - name="a running cached replica on a cluster without cache storage is re-placed", - req=_req( - xr_cached, - clusters=[_CLUSTER_A, _cluster("cluster-b", cache_storage=True)], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - cache=_cache("qwen"), - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-b-0": fnv1.Resource( - resource=resource.dict_to_struct(_CACHED_REPLICA_B) - ), - }, + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 2 ready", ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-b", - ), - ], - context=structpb.Struct(), - ), - cache_name="qwen", - ), - ), - Case( - # The only candidate has no cache storage, so the cache can't - # stage there and nothing is placed. ReplicasScheduled says why - # rather than blaming capacity. - name="no candidate with cache storage places nothing", - req=_req(xr_cached, clusters=[_CLUSTER_A], cache=_cache("qwen")), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_FALSE, - ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 2 replicas across 1 clusters: cluster-a", ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoCacheStorage", - message="0 of 1 replicas scheduled: no candidate cluster has storage for ModelCache qwen", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoReplicasScheduled", - ), - ], - context=structpb.Struct(), - ), - cache_name="qwen", - ), + ], + context=structpb.Struct(), + ) ), - Case( - # A referenced cache Crossplane hasn't fetched yet leaves the - # footprint unknown. With no replicas to retain, the function - # holds off placing any rather than risk landing them outside the - # footprint: fill is suppressed, so nothing is composed, and - # ModelCacheResolved=False (Unresolved) says why. The wait is - # transient and self-clearing, so it's a condition, not an event. - # The cluster and replica requirements are still declared so the - # cache can resolve alongside them. - name="unresolved cache suppresses new placement", - req=_req(xr_cached, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_FALSE, - ), + ), + Case( + # PrefillDecode copies serving and each engine's phase onto the + # replica; the replica backend reads them to front the engines + # with an InferencePool + endpoint picker rather than a Service. + name="PrefillDecode copies serving and engine phases onto the replica", + req=_req(xr_pd, clusters=[_CLUSTER_A]), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelCacheUnresolved", - message="Waiting for ModelCache qwen", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="InsufficientCapacity", - message="0 of 1 replicas scheduled (checked 1 clusters)", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoReplicasScheduled", - ), - ], - context=structpb.Struct(), - ), - cache_name="qwen", - ), - ), - Case( - # The cache a live deployment depends on is deleted (the cache - # requirement resolves but matches nothing - ABSENT). The cache - # only matters when loading weights, which already happened, so - # its disappearance must not tear the deployment down: the - # existing replica is retained (retain ignores fill) even as - # ModelCacheResolved goes False (NotFound) and new placement is - # suppressed. - name="deleted cache retains existing replicas", - req=_req( - xr_cached, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - cache_resolved_empty=True, - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 1}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "modelCacheRef": {"name": "qwen"}, - "engines": _REPLICA_ENGINES, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "endpoint-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "origin": "https://cluster.clusters.example.com", - "api": { - "schema": "OpenAI", - "prefix": "/ml-team/my-model-5ab63/v1", - }, - "model": "ml-team/my-model", + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, - } - ), + }, + "spec": { + "clusterName": "cluster-a", + "serving": {"mode": "PrefillDecode"}, + "engines": _PD_REPLICA_ENGINES, + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelCacheNotFound", - message="ModelCache qwen not found; holding replica placement", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="AllReplicasReady", - message="1 of 1 ready", ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_WARNING, - message="ModelCache qwen not found; holding replica placement", - ), - ], - context=structpb.Struct(), + }, ), - cache_name="qwen", - ), - ), - Case( - name="two replicas co-locate on one cluster as distinct resources", - req=_req(xr_two, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 2, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES_NO_ARGS, - }, - } - ), - ), - "replica-cluster-a-1": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-609c5", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "1", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES_NO_ARGS, - }, - } - ), - ), - }, + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 2 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 2 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - ), - ), - Case( - # PrefillDecode copies serving and each engine's phase onto the - # replica; the replica backend reads them to front the engines - # with an InferencePool + endpoint picker rather than a Service. - name="PrefillDecode copies serving and engine phases onto the replica", - req=_req(xr_pd, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "serving": {"mode": "PrefillDecode"}, - "engines": _PD_REPLICA_ENGINES, - }, - } - ), - ), - }, + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + ) ), - ] + ), + ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """The function fans out ModelReplicas and, once they're Ready, ModelEndpoints.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) def _composed(resp: fnv1.RunFunctionResponse, kind: str) -> list[dict]: @@ -1545,183 +1534,191 @@ def _composed(resp: fnv1.RunFunctionResponse, kind: str) -> list[dict]: return result -class TestTemplateLabels(unittest.IsolatedAsyncioTestCase): - """spec.template.metadata.labels land on the composed ModelReplicas and - ModelEndpoints, alongside the labels Modelplane manages.""" +# spec.template.metadata.labels land on the composed ModelReplicas and +# ModelEndpoints, alongside the labels Modelplane manages. - async def test_stamped_on_replica_and_endpoint(self) -> None: - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - metadata=v1alpha1.Metadata(labels={"tier": "prod", "team": "search"}), - spec=v1alpha1.SpecModel(engines=[_ENGINE]), - ), + +def test_template_labels_stamped_on_replica_and_endpoint() -> None: + """Template labels land on the replica and endpoint beside the managed labels.""" + xr = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=1, + template=v1alpha1.TemplateModel( + metadata=v1alpha1.Metadata(labels={"tier": "prod", "team": "search"}), + spec=v1alpha1.SpecModel(engines=[_ENGINE]), ), - ).model_dump(exclude_none=True, mode="json") - # An observed, Ready replica lets the endpoint compose this reconcile. - req = _req( - xr, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - ) - got = await fn.FunctionRunner().RunFunction(req, None) - - composed = _composed(got, "ModelReplica") + _composed(got, "ModelEndpoint") - self.assertEqual(len(composed), 2, "expected one ModelReplica and one ModelEndpoint") - for obj in composed: - labels = obj["metadata"]["labels"] - self.assertEqual(labels.get("tier"), "prod") - self.assertEqual(labels.get("team"), "search") - self.assertEqual(labels.get("modelplane.ai/deployment"), "my-model") - self.assertEqual(labels.get("modelplane.ai/cluster"), "cluster-a") - self.assertEqual(labels.get("modelplane.ai/replica-index"), "0") - - async def test_managed_labels_win_a_collision(self) -> None: - """The XRD's CEL rejects a template label under the modelplane.ai/ prefix, - but the invariant lives in the function too: managed labels are stamped - last, so a colliding label can't override them even if that CEL rule is - relaxed or the function is reused elsewhere.""" - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - metadata=v1alpha1.Metadata(labels={"modelplane.ai/cluster": "wrong", "tier": "prod"}), - spec=v1alpha1.SpecModel(engines=[_ENGINE]), - ), + ), + ).model_dump(exclude_none=True, mode="json") + # An observed, Ready replica lets the endpoint compose this reconcile. + req = _req( + xr, + clusters=[_CLUSTER_A], + replicas=[_EXISTING_REPLICA], + observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + composed = _composed(got, "ModelReplica") + _composed(got, "ModelEndpoint") + assert len(composed) == 2, "expected one ModelReplica and one ModelEndpoint" + for obj in composed: + labels = obj["metadata"]["labels"] + assert labels.get("tier") == "prod" + assert labels.get("team") == "search" + assert labels.get("modelplane.ai/deployment") == "my-model" + assert labels.get("modelplane.ai/cluster") == "cluster-a" + assert labels.get("modelplane.ai/replica-index") == "0" + + +def test_template_labels_managed_labels_win_a_collision() -> None: + """A managed label beats a template label of the same key.""" + # The XRD's CEL rejects a template label under the modelplane.ai/ prefix, + # but the invariant lives in the function too: managed labels are stamped + # last, so a colliding label can't override them even if that CEL rule is + # relaxed or the function is reused elsewhere. + xr = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=1, + template=v1alpha1.TemplateModel( + metadata=v1alpha1.Metadata(labels={"modelplane.ai/cluster": "wrong", "tier": "prod"}), + spec=v1alpha1.SpecModel(engines=[_ENGINE]), ), - ).model_dump(exclude_none=True, mode="json") - got = await fn.FunctionRunner().RunFunction(_req(xr, clusters=[_CLUSTER_A]), None) + ), + ).model_dump(exclude_none=True, mode="json") + got = asyncio.run(fn.FunctionRunner().RunFunction(_req(xr, clusters=[_CLUSTER_A]), None)) - replica = _composed(got, "ModelReplica")[0] - self.assertEqual(replica["metadata"]["labels"]["modelplane.ai/cluster"], "cluster-a") - self.assertEqual(replica["metadata"]["labels"]["tier"], "prod") + replica = _composed(got, "ModelReplica")[0] + assert replica["metadata"]["labels"]["modelplane.ai/cluster"] == "cluster-a" + assert replica["metadata"]["labels"]["tier"] == "prod" -class TestResolveRequired(unittest.TestCase): - """Tests for fn.resolve_required - the three-state required-resource read.""" +def test_resolve_required() -> None: + """resolve_required tells a found, a missing, and an unfetched requirement apart.""" + cache = {"apiVersion": "modelplane.ai/v1alpha1", "kind": "ModelCache", "metadata": {"name": "qwen"}} - def test_resolve_required(self) -> None: - cache = {"apiVersion": "modelplane.ai/v1alpha1", "kind": "ModelCache", "metadata": {"name": "qwen"}} + # PRESENT: the requirement resolved and matched a resource. + req = fnv1.RunFunctionRequest() + req.required_resources["cache"].items.append(fnv1.Resource(resource=resource.dict_to_struct(cache))) + assert fn.resolve_required(req, "cache") == (fn.Resolution.PRESENT, cache) - # PRESENT: the requirement resolved and matched a resource. - req = fnv1.RunFunctionRequest() - req.required_resources["cache"].items.append(fnv1.Resource(resource=resource.dict_to_struct(cache))) - self.assertEqual((fn.Resolution.PRESENT, cache), fn.resolve_required(req, "cache")) + # ABSENT: the requirement resolved but matched nothing (key present, no items). + req = fnv1.RunFunctionRequest() + req.required_resources["cache"].SetInParent() + assert fn.resolve_required(req, "cache") == (fn.Resolution.ABSENT, None) - # ABSENT: the requirement resolved but matched nothing (key present, no items). - req = fnv1.RunFunctionRequest() - req.required_resources["cache"].SetInParent() - self.assertEqual((fn.Resolution.ABSENT, None), fn.resolve_required(req, "cache")) - - # UNRESOLVED: Crossplane has not fetched the requirement (key absent). - req = fnv1.RunFunctionRequest() - self.assertEqual((fn.Resolution.UNRESOLVED, None), fn.resolve_required(req, "cache")) - - -class TestServedModelName(unittest.TestCase): - """The name an engine is started under, and how it gets there.""" - - def test_it_goes_ahead_of_the_users_env(self) -> None: - """Env expansion is left to right, so an arg or a later entry - referencing $(MODELPLANE_SERVED_MODEL_NAME) only resolves if it's - first.""" - template = mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container( - name="engine", - image="vllm/vllm-openai:latest", - env=[mrv1alpha1.EnvItem(name="HF_TOKEN", value="x")], - ) - ] - ) - ) - fn._inject_served_model_name(template, "ml-team/kimi-k2") - assert template.spec is not None - self.assertEqual( - [(e.name, e.value) for e in template.spec.containers[0].env or []], - [("MODELPLANE_SERVED_MODEL_NAME", "ml-team/kimi-k2"), ("HF_TOKEN", "x")], - ) + # UNRESOLVED: Crossplane has not fetched the requirement (key absent). + req = fnv1.RunFunctionRequest() + assert fn.resolve_required(req, "cache") == (fn.Resolution.UNRESOLVED, None) - def test_a_user_override_is_dropped(self) -> None: - """Modelplane decides this value. Honouring an override would let the - engine answer to a name nothing routes to, which surfaces as a 404 from - the engine rather than anything visible in status.""" - template = mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container( - name="engine", - image="vllm/vllm-openai:latest", - env=[mrv1alpha1.EnvItem(name="MODELPLANE_SERVED_MODEL_NAME", value="mine")], - ) - ] - ) - ) - fn._inject_served_model_name(template, "ml-team/kimi-k2") - assert template.spec is not None - self.assertEqual( - [(e.name, e.value) for e in template.spec.containers[0].env or []], - [("MODELPLANE_SERVED_MODEL_NAME", "ml-team/kimi-k2")], - ) - def test_it_is_namespaced(self) -> None: - """So two deployments in different namespaces can't collide, and a - ModelService can rewrite one name for a whole deployment.""" - self.assertEqual(fn.served_model_name("ml-team", "kimi-k2"), "ml-team/kimi-k2") +# The name an engine is started under, and how it gets there. -class TestPlacementLabels(unittest.IsolatedAsyncioTestCase): - """A cluster's spec.placement.metadata.labels land on the ModelReplicas and - ModelEndpoints composed there. +def test_served_model_name_goes_ahead_of_the_users_env() -> None: + """The served model name env var comes before the container's own env.""" + # Env expansion is left to right, so an arg or a later entry referencing + # $(MODELPLANE_SERVED_MODEL_NAME) only resolves if it's first. + template = mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + env=[mrv1alpha1.EnvItem(name="HF_TOKEN", value="x")], + ) + ] + ) + ) + fn._inject_served_model_name(template, "ml-team/kimi-k2") + assert template.spec is not None + assert [(e.name, e.value) for e in template.spec.containers[0].env or []] == [ + ("MODELPLANE_SERVED_MODEL_NAME", "ml-team/kimi-k2"), + ("HF_TOKEN", "x"), + ] - This is the endpoint half of residency: a ModelService selects endpoints by - label, so without it a region-scoped service can't select its own replicas, - and nobody can label them by hand because Modelplane owns them. The gateway - half is an InferenceGateway's serviceSelector. - """ - async def test_stamped_on_replica_and_endpoint(self) -> None: - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE])), - ), - ).model_dump(exclude_none=True, mode="json") - req = _req( - xr, - clusters=[_cluster("cluster-a", placement_labels={"example.org/region": "eu"})], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, +def test_served_model_name_user_override_is_dropped() -> None: + """A user's own MODELPLANE_SERVED_MODEL_NAME is replaced, not kept.""" + # Modelplane decides this value. Honouring an override would let the engine + # answer to a name nothing routes to, which surfaces as a 404 from the + # engine rather than anything visible in status. + template = mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + env=[mrv1alpha1.EnvItem(name="MODELPLANE_SERVED_MODEL_NAME", value="mine")], + ) + ] ) - got = await fn.FunctionRunner().RunFunction(req, None) - - composed = _composed(got, "ModelReplica") + _composed(got, "ModelEndpoint") - self.assertEqual(len(composed), 2, "expected one ModelReplica and one ModelEndpoint") - for obj in composed: - self.assertEqual(obj["metadata"]["labels"].get("example.org/region"), "eu") - - async def test_a_cluster_label_beats_a_template_label(self) -> None: - """The cluster is the authority on where it is, so its placement labels - are stamped after the deployment's own template labels.""" - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - metadata=v1alpha1.Metadata(labels={"example.org/region": "wrong"}), - spec=v1alpha1.SpecModel(engines=[_ENGINE]), - ), + ) + fn._inject_served_model_name(template, "ml-team/kimi-k2") + assert template.spec is not None + assert [(e.name, e.value) for e in template.spec.containers[0].env or []] == [ + ("MODELPLANE_SERVED_MODEL_NAME", "ml-team/kimi-k2"), + ] + + +def test_served_model_name_is_namespaced() -> None: + """served_model_name prefixes the deployment's name with its namespace.""" + # So two deployments in different namespaces can't collide, and a + # ModelService can rewrite one name for a whole deployment. + assert fn.served_model_name("ml-team", "kimi-k2") == "ml-team/kimi-k2" + + +# A cluster's spec.placement.metadata.labels land on the ModelReplicas and +# ModelEndpoints composed there. +# +# This is the endpoint half of residency: a ModelService selects endpoints by +# label, so without it a region-scoped service can't select its own replicas, +# and nobody can label them by hand because Modelplane owns them. The gateway +# half is an InferenceGateway's serviceSelector. + + +def test_placement_labels_stamped_on_replica_and_endpoint() -> None: + """A cluster's placement labels land on the replica and endpoint composed there.""" + xr = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=1, + template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE])), + ), + ).model_dump(exclude_none=True, mode="json") + req = _req( + xr, + clusters=[_cluster("cluster-a", placement_labels={"example.org/region": "eu"})], + replicas=[_EXISTING_REPLICA], + observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + composed = _composed(got, "ModelReplica") + _composed(got, "ModelEndpoint") + assert len(composed) == 2, "expected one ModelReplica and one ModelEndpoint" + for obj in composed: + assert obj["metadata"]["labels"].get("example.org/region") == "eu" + + +def test_placement_labels_cluster_label_beats_a_template_label() -> None: + """A cluster's placement label beats a template label of the same key.""" + # The cluster is the authority on where it is, so its placement labels are + # stamped after the deployment's own template labels. + xr = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=1, + template=v1alpha1.TemplateModel( + metadata=v1alpha1.Metadata(labels={"example.org/region": "wrong"}), + spec=v1alpha1.SpecModel(engines=[_ENGINE]), ), - ).model_dump(exclude_none=True, mode="json") - got = await fn.FunctionRunner().RunFunction( + ), + ).model_dump(exclude_none=True, mode="json") + got = asyncio.run( + fn.FunctionRunner().RunFunction( _req(xr, clusters=[_cluster("cluster-a", placement_labels={"example.org/region": "eu"})]), None, ) - replica = _composed(got, "ModelReplica")[0] - self.assertEqual(replica["metadata"]["labels"]["example.org/region"], "eu") + ) + replica = _composed(got, "ModelReplica")[0] + assert replica["metadata"]["labels"]["example.org/region"] == "eu" diff --git a/functions/compose-model-deployment/tests/test_quantity.py b/functions/compose-model-deployment/tests/test_quantity.py index d196962ff..5c0429415 100644 --- a/functions/compose-model-deployment/tests/test_quantity.py +++ b/functions/compose-model-deployment/tests/test_quantity.py @@ -37,8 +37,8 @@ """ import dataclasses -import unittest +import pytest from function import cel, quantity @@ -60,169 +60,164 @@ class ParseErrCase: input: str -class TestQuantityCEL(unittest.TestCase): - """Mirrors quantity_test.go TestQuantity (and the doc-comment examples).""" - - def test_quantity(self) -> None: - cases = [ - # parse + isQuantity. - Case(name="parse", expr='quantity("12Mi").compareTo(quantity("12Mi")) == 0', want=True), - Case(name="isQuantity int string", expr='isQuantity("20")', want=True), - Case(name="isQuantity megabytes", expr='isQuantity("20M")', want=True), - Case(name="isQuantity mebibytes", expr='isQuantity("20Mi")', want=True), - Case(name="isQuantity invalid suffix", expr='isQuantity("20Mo")', want=False), - Case(name="isQuantity passing regex bad suffix", expr='isQuantity("10Mm")', want=False), - # resource.Quantity accepts decimal exponents and nano/micro suffixes. - Case(name="isQuantity exponent lowercase", expr='isQuantity("256e3")', want=True), - Case(name="isQuantity exponent uppercase", expr='isQuantity("1E3")', want=True), - Case(name="exponent value", expr='quantity("256e3").compareTo(quantity("256000")) == 0', want=True), - Case(name="isQuantity nano", expr='isQuantity("100n")', want=True), - Case(name="isQuantity micro", expr='isQuantity("100u")', want=True), - Case(name="isQuantity trailing dot", expr='isQuantity("5.")', want=True), - # The quantity() constructor does NOT trim whitespace. - Case(name="isQuantity leading whitespace false", expr='isQuantity(" 5Gi")', want=False), - Case(name="isQuantity trailing whitespace false", expr='isQuantity("5Gi ")', want=False), - # Values equal at nano resolution compare equal (resource.Quantity.Cmp - # rounds to nano). - Case( - name="nano rounding equality", - expr='quantity("0.0000000004").compareTo(quantity("0.000000001")) == 0', - want=True, - ), - # doc-comment isQuantity examples. - Case(name="isQuantity 1.3G", expr='isQuantity("1.3G")', want=True), - Case(name="isQuantity 1.3Gi", expr='isQuantity("1.3Gi")', want=True), - Case(name="isQuantity comma", expr='isQuantity("1,3G")', want=False), - Case(name="isQuantity 10000k", expr='isQuantity("10000k")', want=True), - Case(name="isQuantity capital K", expr='isQuantity("200K")', want=False), - Case(name="isQuantity Three", expr='isQuantity("Three")', want=False), - Case(name="isQuantity bare suffix", expr='isQuantity("Mi")', want=False), - # equality. - Case(name="equality reflexivity", expr='quantity("200M") == quantity("200M")', want=True), - Case( - name="equality symmetry", - expr='quantity("200M") == quantity("0.2G") && quantity("0.2G") == quantity("200M")', - want=True, - ), - Case( - name="equality transitivity", - expr=( - 'quantity("2M") == quantity("0.002G") && quantity("2000k") == quantity("2M") && ' - 'quantity("0.002G") == quantity("2000k")' - ), - want=True, - ), - Case(name="inequality", expr='quantity("200M") == quantity("0.3G")', want=False), - # isLessThan / isGreaterThan. - Case(name="less", expr='quantity("50M").isLessThan(quantity("50Mi"))', want=True), - Case(name="less obvious", expr='quantity("50M").isLessThan(quantity("100M"))', want=True), - Case(name="less false", expr='quantity("100M").isLessThan(quantity("50M"))', want=False), - Case(name="greater", expr='quantity("50Mi").isGreaterThan(quantity("50M"))', want=True), - Case(name="greater obvious", expr='quantity("150Mi").isGreaterThan(quantity("100Mi"))', want=True), - Case(name="greater false", expr='quantity("50M").isGreaterThan(quantity("100M"))', want=False), - # compareTo. - Case(name="compare equal", expr='quantity("200M").compareTo(quantity("0.2G")) == 0', want=True), - Case(name="compare less", expr='quantity("50M").compareTo(quantity("50Mi")) == -1', want=True), - Case(name="compare greater", expr='quantity("50Mi").compareTo(quantity("50M")) == 1', want=True), - # add / sub (quantity and int overloads). - Case(name="add quantity", expr='quantity("50k").add(quantity("20")) == quantity("50.02k")', want=True), - Case(name="add int not less", expr='quantity("50k").add(20).isLessThan(quantity("50020"))', want=False), - Case(name="sub quantity", expr='quantity("50k").sub(quantity("20")) == quantity("49.98k")', want=True), - Case(name="sub int", expr='quantity("50k").sub(20) == quantity("49980")', want=True), - Case( - name="arith chain 1", - expr='quantity("50k").add(20).sub(quantity("100k")).asInteger() == -49980', - want=True, - ), - Case( - name="arith chain 2", - expr='quantity("50k").add(20).sub(quantity("100k")).sub(-50000).asInteger() == 20', - want=True, - ), - # sign (doc comment). Upstream declares sign as a GLOBAL function - # (sign(q)), not a member (q.sign()); the global form is the parity - # surface. celpy can't tell the two call styles apart, so we accept - # both, but the test asserts the upstream-correct global form (see - # cel.py's documented divergences). - Case(name="sign positive", expr='sign(quantity("50k")) == 1', want=True), - Case(name="sign negative", expr='sign(quantity("-50k")) == -1', want=True), - Case(name="sign zero", expr='sign(quantity("0")) == 0', want=True), - # Binary-suffix overflow saturates to int64-max, keeping sign, so - # 8Ei/10Ei/100Ei all compare equal to int64-max (resource.Quantity - # stores BinarySI in an int64). Confirmed against resource.Quantity. - Case( - name="Ei saturates to int64 max", - expr='quantity("8Ei").compareTo(quantity("9223372036854775807")) == 0', - want=True, - ), - Case(name="8Ei equals 10Ei", expr='quantity("8Ei").compareTo(quantity("10Ei")) == 0', want=True), - Case(name="10Ei equals 100Ei", expr='quantity("10Ei").compareTo(quantity("100Ei")) == 0', want=True), - Case( - name="negative Ei saturates", - expr='quantity("-10Ei").compareTo(quantity("-9223372036854775807")) == 0', - want=True, - ), - # 7Ei is below int64-max, so it does NOT saturate and stays less. - Case(name="7Ei below saturation", expr='quantity("7Ei").isLessThan(quantity("8Ei"))', want=True), - # Large DECIMAL-path values do not saturate (only the binary path - # does) and must not raise on nano-rounding. isQuantity must be true - # and the value must round-trip. - Case(name="isQuantity 256E", expr='isQuantity("256E")', want=True), - Case(name="isQuantity 10E", expr='isQuantity("10E")', want=True), - Case( - name="256E value", expr='quantity("256E").compareTo(quantity("256000000000000000000")) == 0', want=True - ), - Case(name="256E greater than 1Ei", expr='quantity("256E").isGreaterThan(quantity("1Ei"))', want=True), - # asInteger / isInteger. - Case(name="as integer", expr='quantity("50k").asInteger() == 50000', want=True), - Case(name="is integer true small", expr='quantity("50").isInteger()', want=True), - Case(name="is integer true big magnitude", expr='quantity("50000000G").isInteger()', want=True), - Case( - name="is integer false overflow", - expr='quantity("9999999999999999999999999999999999999G").isInteger()', - want=False, - ), - # asInteger overflow is a runtime error upstream -> non-match here. - Case( - name="as integer overflow is non-match", - expr='quantity("9999999999999999999999999999999999999G").asInteger() > 0', - want=False, - ), - # asApproximateFloat. - Case(name="as approximate float", expr='quantity("50.703k").asApproximateFloat() == 50703.0', want=True), - # An invalid suffix is a runtime error upstream -> non-match here. - # (Uses a member method upstream accepts, isGreaterThan, so the - # non-match is the parse failure, not a rejected call form.) - Case( - name="invalid suffix is non-match", - expr='quantity("10Mo").isGreaterThan(quantity("1"))', - want=False, - ), - ] - for case in cases: - with self.subTest(case.name): - self.assertEqual(case.want, _eval(case.expr), f"{case.name}: -want, +got") - - -class TestParseRejects(unittest.TestCase): - """parse() rejects what resource.Quantity rejects (drives the non-matches above). - - The bare-suffix row ("Mi") is a DELIBERATE divergence, not parity: upstream - parses most bare suffixes as 0 but inconsistently errors on a few (see - parse()'s docstring). We reject every bare suffix; no device capacity is - ever a bare suffix. - """ - - def test_parse_rejects(self) -> None: - cases = [ - ParseErrCase(name="invalid suffix Mo", input="10Mo"), - ParseErrCase(name="passing regex bad suffix Mm", input="10Mm"), - ParseErrCase(name="capital K", input="200K"), - ParseErrCase(name="comma", input="1,3G"), - ParseErrCase(name="word", input="Three"), - ParseErrCase(name="bare suffix (deliberate divergence)", input="Mi"), - ParseErrCase(name="empty", input=""), - ] - for case in cases: - with self.subTest(case.name), self.assertRaises(ValueError): - quantity.parse(case.input) +QUANTITY_CASES = [ + # parse + isQuantity. + Case(name="parse", expr='quantity("12Mi").compareTo(quantity("12Mi")) == 0', want=True), + Case(name="isQuantity int string", expr='isQuantity("20")', want=True), + Case(name="isQuantity megabytes", expr='isQuantity("20M")', want=True), + Case(name="isQuantity mebibytes", expr='isQuantity("20Mi")', want=True), + Case(name="isQuantity invalid suffix", expr='isQuantity("20Mo")', want=False), + Case(name="isQuantity passing regex bad suffix", expr='isQuantity("10Mm")', want=False), + # resource.Quantity accepts decimal exponents and nano/micro suffixes. + Case(name="isQuantity exponent lowercase", expr='isQuantity("256e3")', want=True), + Case(name="isQuantity exponent uppercase", expr='isQuantity("1E3")', want=True), + Case(name="exponent value", expr='quantity("256e3").compareTo(quantity("256000")) == 0', want=True), + Case(name="isQuantity nano", expr='isQuantity("100n")', want=True), + Case(name="isQuantity micro", expr='isQuantity("100u")', want=True), + Case(name="isQuantity trailing dot", expr='isQuantity("5.")', want=True), + # The quantity() constructor does NOT trim whitespace. + Case(name="isQuantity leading whitespace false", expr='isQuantity(" 5Gi")', want=False), + Case(name="isQuantity trailing whitespace false", expr='isQuantity("5Gi ")', want=False), + # Values equal at nano resolution compare equal (resource.Quantity.Cmp + # rounds to nano). + Case( + name="nano rounding equality", + expr='quantity("0.0000000004").compareTo(quantity("0.000000001")) == 0', + want=True, + ), + # doc-comment isQuantity examples. + Case(name="isQuantity 1.3G", expr='isQuantity("1.3G")', want=True), + Case(name="isQuantity 1.3Gi", expr='isQuantity("1.3Gi")', want=True), + Case(name="isQuantity comma", expr='isQuantity("1,3G")', want=False), + Case(name="isQuantity 10000k", expr='isQuantity("10000k")', want=True), + Case(name="isQuantity capital K", expr='isQuantity("200K")', want=False), + Case(name="isQuantity Three", expr='isQuantity("Three")', want=False), + Case(name="isQuantity bare suffix", expr='isQuantity("Mi")', want=False), + # equality. + Case(name="equality reflexivity", expr='quantity("200M") == quantity("200M")', want=True), + Case( + name="equality symmetry", + expr='quantity("200M") == quantity("0.2G") && quantity("0.2G") == quantity("200M")', + want=True, + ), + Case( + name="equality transitivity", + expr=( + 'quantity("2M") == quantity("0.002G") && quantity("2000k") == quantity("2M") && ' + 'quantity("0.002G") == quantity("2000k")' + ), + want=True, + ), + Case(name="inequality", expr='quantity("200M") == quantity("0.3G")', want=False), + # isLessThan / isGreaterThan. + Case(name="less", expr='quantity("50M").isLessThan(quantity("50Mi"))', want=True), + Case(name="less obvious", expr='quantity("50M").isLessThan(quantity("100M"))', want=True), + Case(name="less false", expr='quantity("100M").isLessThan(quantity("50M"))', want=False), + Case(name="greater", expr='quantity("50Mi").isGreaterThan(quantity("50M"))', want=True), + Case(name="greater obvious", expr='quantity("150Mi").isGreaterThan(quantity("100Mi"))', want=True), + Case(name="greater false", expr='quantity("50M").isGreaterThan(quantity("100M"))', want=False), + # compareTo. + Case(name="compare equal", expr='quantity("200M").compareTo(quantity("0.2G")) == 0', want=True), + Case(name="compare less", expr='quantity("50M").compareTo(quantity("50Mi")) == -1', want=True), + Case(name="compare greater", expr='quantity("50Mi").compareTo(quantity("50M")) == 1', want=True), + # add / sub (quantity and int overloads). + Case(name="add quantity", expr='quantity("50k").add(quantity("20")) == quantity("50.02k")', want=True), + Case(name="add int not less", expr='quantity("50k").add(20).isLessThan(quantity("50020"))', want=False), + Case(name="sub quantity", expr='quantity("50k").sub(quantity("20")) == quantity("49.98k")', want=True), + Case(name="sub int", expr='quantity("50k").sub(20) == quantity("49980")', want=True), + Case( + name="arith chain 1", + expr='quantity("50k").add(20).sub(quantity("100k")).asInteger() == -49980', + want=True, + ), + Case( + name="arith chain 2", + expr='quantity("50k").add(20).sub(quantity("100k")).sub(-50000).asInteger() == 20', + want=True, + ), + # sign (doc comment). Upstream declares sign as a GLOBAL function + # (sign(q)), not a member (q.sign()); the global form is the parity + # surface. celpy can't tell the two call styles apart, so we accept + # both, but the test asserts the upstream-correct global form (see + # cel.py's documented divergences). + Case(name="sign positive", expr='sign(quantity("50k")) == 1', want=True), + Case(name="sign negative", expr='sign(quantity("-50k")) == -1', want=True), + Case(name="sign zero", expr='sign(quantity("0")) == 0', want=True), + # Binary-suffix overflow saturates to int64-max, keeping sign, so + # 8Ei/10Ei/100Ei all compare equal to int64-max (resource.Quantity + # stores BinarySI in an int64). Confirmed against resource.Quantity. + Case( + name="Ei saturates to int64 max", + expr='quantity("8Ei").compareTo(quantity("9223372036854775807")) == 0', + want=True, + ), + Case(name="8Ei equals 10Ei", expr='quantity("8Ei").compareTo(quantity("10Ei")) == 0', want=True), + Case(name="10Ei equals 100Ei", expr='quantity("10Ei").compareTo(quantity("100Ei")) == 0', want=True), + Case( + name="negative Ei saturates", + expr='quantity("-10Ei").compareTo(quantity("-9223372036854775807")) == 0', + want=True, + ), + # 7Ei is below int64-max, so it does NOT saturate and stays less. + Case(name="7Ei below saturation", expr='quantity("7Ei").isLessThan(quantity("8Ei"))', want=True), + # Large DECIMAL-path values do not saturate (only the binary path + # does) and must not raise on nano-rounding. isQuantity must be true + # and the value must round-trip. + Case(name="isQuantity 256E", expr='isQuantity("256E")', want=True), + Case(name="isQuantity 10E", expr='isQuantity("10E")', want=True), + Case(name="256E value", expr='quantity("256E").compareTo(quantity("256000000000000000000")) == 0', want=True), + Case(name="256E greater than 1Ei", expr='quantity("256E").isGreaterThan(quantity("1Ei"))', want=True), + # asInteger / isInteger. + Case(name="as integer", expr='quantity("50k").asInteger() == 50000', want=True), + Case(name="is integer true small", expr='quantity("50").isInteger()', want=True), + Case(name="is integer true big magnitude", expr='quantity("50000000G").isInteger()', want=True), + Case( + name="is integer false overflow", + expr='quantity("9999999999999999999999999999999999999G").isInteger()', + want=False, + ), + # asInteger overflow is a runtime error upstream -> non-match here. + Case( + name="as integer overflow is non-match", + expr='quantity("9999999999999999999999999999999999999G").asInteger() > 0', + want=False, + ), + # asApproximateFloat. + Case(name="as approximate float", expr='quantity("50.703k").asApproximateFloat() == 50703.0', want=True), + # An invalid suffix is a runtime error upstream -> non-match here. + # (Uses a member method upstream accepts, isGreaterThan, so the + # non-match is the parse failure, not a rejected call form.) + Case( + name="invalid suffix is non-match", + expr='quantity("10Mo").isGreaterThan(quantity("1"))', + want=False, + ), +] + + +@pytest.mark.parametrize("case", QUANTITY_CASES, ids=lambda case: case.name) +def test_quantity(case: Case) -> None: + """A quantity CEL expression evaluates as it does upstream.""" + assert _eval(case.expr) == case.want + + +# The bare-suffix row ("Mi") is a DELIBERATE divergence, not parity: upstream +# parses most bare suffixes as 0 but inconsistently errors on a few (see +# parse()'s docstring). We reject every bare suffix; no device capacity is +# ever a bare suffix. +PARSE_REJECTS_CASES = [ + ParseErrCase(name="invalid suffix Mo", input="10Mo"), + ParseErrCase(name="passing regex bad suffix Mm", input="10Mm"), + ParseErrCase(name="capital K", input="200K"), + ParseErrCase(name="comma", input="1,3G"), + ParseErrCase(name="word", input="Three"), + ParseErrCase(name="bare suffix (deliberate divergence)", input="Mi"), + ParseErrCase(name="empty", input=""), +] + + +@pytest.mark.parametrize("case", PARSE_REJECTS_CASES, ids=lambda case: case.name) +def test_parse_rejects(case: ParseErrCase) -> None: + """parse() rejects what resource.Quantity rejects, which drives the non-matches above.""" + with pytest.raises(ValueError, match="invalid quantity"): + quantity.parse(case.input) diff --git a/functions/compose-model-deployment/tests/test_scheduling.py b/functions/compose-model-deployment/tests/test_scheduling.py index 31449aa39..a584c9c9f 100644 --- a/functions/compose-model-deployment/tests/test_scheduling.py +++ b/functions/compose-model-deployment/tests/test_scheduling.py @@ -25,8 +25,8 @@ import dataclasses import datetime -import unittest +import pytest from function import cel, scheduling from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 from models.ai.modelplane.modeldeployment import v1alpha1 as mdv1alpha1 @@ -405,810 +405,803 @@ def _cand( ) -class TestSchedule(unittest.TestCase): - """Tests for scheduling.schedule placement: retain, spread, scale, capacity. - - Deployments use the default single-GPU nodeSelector request (any pool's GPU - device satisfies it), so these focus on placement rather than pool matching; - TestScheduleNodeSelector covers request-to-device matching. - """ - - def test_schedule(self) -> None: - """The scheduler retains existing pins and places new replicas.""" - - cases = [ - Case( - name="no clusters returns no candidates", - deployment=_deployment(), - clusters=[], - all_replicas=[], - want=[], - ), - Case( - name="single ready cluster is picked", - deployment=_deployment(), - clusters=[_cluster("cluster-a")], - all_replicas=[], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default")], - ), - Case( - name="not-ready cluster is not picked for a new replica", - deployment=_deployment(), - clusters=[_cluster("cluster-a", ready=False)], - all_replicas=[], - want=[], - ), - Case( - name="cluster without gateway address is not picked", - deployment=_deployment(), - clusters=[_cluster("cluster-a", gateway_hostname="")], - all_replicas=[], - want=[], - ), - Case( - name="multi-node deployment needs enough nodes", - deployment=_deployment(pipeline=4), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], - all_replicas=[], - want=[], - ), - Case( - name="existing replica is retained on its pinned cluster", - deployment=_deployment(), - clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - # cluster-a wins even though cluster-b is also viable. The pin - # still matches, so it's retained with its resolved pool/requests. - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="degraded pinned cluster is retained with empty gateway", - deployment=_deployment(), - clusters=[_cluster("cluster-a", ready=False, gateway_hostname="")], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="")], - ), - Case( - name="deleted pinned cluster triggers re-placement", - deployment=_deployment(), - clusters=[_cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com")], - all_replicas=[_replica("my-model", "cluster-a")], - want=[_cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default")], - ), - Case( - name="scale up places new replicas on additional clusters", - deployment=_deployment(replicas=2), - clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ], - all_replicas=[_replica("my-model", "cluster-a")], - want=[ - _cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com"), - _cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="scale up with no extra capacity returns only retained", - deployment=_deployment(replicas=2), - # Single-node pool, already filled by the retained replica, so no - # second replica can be placed - not even on the same cluster. - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default")], - ), - Case( - name="two replicas pack onto one cluster when it is the only option", - deployment=_deployment(replicas=2), - # One cluster, a 2-node pool, two 1-node replicas. With nowhere - # to spread, both pack onto cluster-a at indices 0 and 1. - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], - all_replicas=[], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - ], - ), - Case( - name="two replicas spread across two clusters before packing", - deployment=_deployment(replicas=2), - # Both clusters can hold two replicas, but we prefer one each. - clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=2)]), - _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=2)], - ), - ], - all_replicas=[], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="three replicas spread first then pack the remainder", - deployment=_deployment(replicas=3), - # Two clusters, plenty of room. Spread gives a, b one each, then - # the third lands back on cluster-a (lowest load, name tiebreak). - clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), - _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], - ), - ], - all_replicas=[], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="capacity forces packing past the spread preference", - deployment=_deployment(replicas=3), - # cluster-b holds one replica; cluster-a has room for the rest. - # Spread puts one on each, then the third can't fit on b (full), - # so it packs onto a. - clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), - _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=1)], - ), - ], - all_replicas=[], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="new replica spreads onto an empty cluster before doubling up", - deployment=_deployment(replicas=2), - # cluster-a already hosts a replica; cluster-b is empty. The new - # replica prefers empty cluster-b over packing onto a. - clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), - _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], - ), - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="new replica takes the lowest free index on a packed cluster", - deployment=_deployment(replicas=3), - # Only cluster-a exists, already hosting indices 0 and 2 (1 was - # deleted). The new replica fills the gap at index 1. - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], - all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=2), - ], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=2, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - ], - ), - Case( - name="scale down packs off by dropping the highest index first", - deployment=_deployment(replicas=2), - # cluster-a hosts indices 0 and 1; cluster-b hosts index 0. Three - # replicas, want two. Highest index (a/1) is dropped, keeping the - # spread across a/0 and b/0. - clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), - _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], - ), - ], - all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=1), - _replica_with_pool("my-model", "cluster-b", pool="default", index=0), - ], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="retained replica is charged at its own node cost, not the new shape", - # The deployment's workers grew to pipeline=4 (4 nodes/replica), - # but the existing replica was created at pipeline=2 and is - # retained (no nodeSelector change rolls it). It still consumes - # only its original 2 nodes. The pool has 6, so a second replica - # at the new 4-node cost must still fit (6 - 2 = 4). Regression: - # charging the retained replica at the new shape (4) would leave - # 2 free and wrongly refuse the placement. - deployment=_deployment(replicas=2, pipeline=4), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=6)])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default", pipeline=2)], - # The retained replica is re-stamped to the deployment's current - # pipeline=4 shape but still charged its observed 2 nodes in the - # ledger. - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pipeline=4), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pipeline=4), - ], - ), - Case( - name="scale down drops from the most-loaded cluster to preserve spread", - deployment=_deployment(replicas=2), - # cluster-a hosts two replicas, cluster-b one. Scaling 3->2 must - # drop a's extra (a/1), NOT b's sole replica - otherwise we'd - # leave a packed and b empty, the opposite of spread. b's index - # is 3 (higher than a/1) to prove we drop by cluster load, not by - # a global index comparison. - clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), - _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], - ), - ], - all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=1), - _replica_with_pool("my-model", "cluster-b", pool="default", index=3), - ], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=3, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="co-located replicas are both retained across a reconcile", - deployment=_deployment(replicas=2), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], - all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=1), - ], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - ], - ), - Case( - name="scale down across clusters drops higher cluster name at equal index", - deployment=_deployment(replicas=1), - clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ], - all_replicas=[ - _replica("my-model", "cluster-b"), - _replica("my-model", "cluster-a"), - ], - # Both at index 0, so the (index, name) tiebreak keeps cluster-a. - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="new placement is alphabetical for determinism", - deployment=_deployment(replicas=2), - clusters=[ - _cluster("cluster-c", gateway_hostname="cluster-c.clusters.example.com"), - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), +# Deployments use the default single-GPU nodeSelector request (any pool's GPU +# device satisfies it), so these focus on placement rather than pool matching; +# NODE_SELECTOR_CASES covers request-to-device matching. +SCHEDULE_CASES = [ + Case( + name="no clusters returns no candidates", + deployment=_deployment(), + clusters=[], + all_replicas=[], + want=[], + ), + Case( + name="single ready cluster is picked", + deployment=_deployment(), + clusters=[_cluster("cluster-a")], + all_replicas=[], + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default")], + ), + Case( + name="not-ready cluster is not picked for a new replica", + deployment=_deployment(), + clusters=[_cluster("cluster-a", ready=False)], + all_replicas=[], + want=[], + ), + Case( + name="cluster without gateway address is not picked", + deployment=_deployment(), + clusters=[_cluster("cluster-a", gateway_hostname="")], + all_replicas=[], + want=[], + ), + Case( + name="multi-node deployment needs enough nodes", + deployment=_deployment(pipeline=4), + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], + all_replicas=[], + want=[], + ), + Case( + name="existing replica is retained on its pinned cluster", + deployment=_deployment(), + clusters=[ + _cluster("cluster-a"), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + # cluster-a wins even though cluster-b is also viable. The pin + # still matches, so it's retained with its resolved pool/requests. + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], + ), + Case( + name="degraded pinned cluster is retained with empty gateway", + deployment=_deployment(), + clusters=[_cluster("cluster-a", ready=False, gateway_hostname="")], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + want=[_cand(name="cluster-a", gateway_hostname="")], + ), + Case( + name="deleted pinned cluster triggers re-placement", + deployment=_deployment(), + clusters=[_cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com")], + all_replicas=[_replica("my-model", "cluster-a")], + want=[_cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default")], + ), + Case( + name="scale up places new replicas on additional clusters", + deployment=_deployment(replicas=2), + clusters=[ + _cluster("cluster-a"), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ], + all_replicas=[_replica("my-model", "cluster-a")], + want=[ + _cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com"), + _cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="scale up with no extra capacity returns only retained", + deployment=_deployment(replicas=2), + # Single-node pool, already filled by the retained replica, so no + # second replica can be placed - not even on the same cluster. + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default")], + ), + Case( + name="two replicas pack onto one cluster when it is the only option", + deployment=_deployment(replicas=2), + # One cluster, a 2-node pool, two 1-node replicas. With nowhere + # to spread, both pack onto cluster-a at indices 0 and 1. + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], + all_replicas=[], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + ], + ), + Case( + name="two replicas spread across two clusters before packing", + deployment=_deployment(replicas=2), + # Both clusters can hold two replicas, but we prefer one each. + clusters=[ + _cluster("cluster-a", pools=[_pool("default", nodes=2)]), + _cluster( + "cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + pools=[_pool("default", nodes=2)], + ), + ], + all_replicas=[], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="three replicas spread first then pack the remainder", + deployment=_deployment(replicas=3), + # Two clusters, plenty of room. Spread gives a, b one each, then + # the third lands back on cluster-a (lowest load, name tiebreak). + clusters=[ + _cluster("cluster-a", pools=[_pool("default", nodes=4)]), + _cluster( + "cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + pools=[_pool("default", nodes=4)], + ), + ], + all_replicas=[], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="capacity forces packing past the spread preference", + deployment=_deployment(replicas=3), + # cluster-b holds one replica; cluster-a has room for the rest. + # Spread puts one on each, then the third can't fit on b (full), + # so it packs onto a. + clusters=[ + _cluster("cluster-a", pools=[_pool("default", nodes=4)]), + _cluster( + "cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + pools=[_pool("default", nodes=1)], + ), + ], + all_replicas=[], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="new replica spreads onto an empty cluster before doubling up", + deployment=_deployment(replicas=2), + # cluster-a already hosts a replica; cluster-b is empty. The new + # replica prefers empty cluster-b over packing onto a. + clusters=[ + _cluster("cluster-a", pools=[_pool("default", nodes=4)]), + _cluster( + "cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + pools=[_pool("default", nodes=4)], + ), + ], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="new replica takes the lowest free index on a packed cluster", + deployment=_deployment(replicas=3), + # Only cluster-a exists, already hosting indices 0 and 2 (1 was + # deleted). The new replica fills the gap at index 1. + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], + all_replicas=[ + _replica_with_pool("my-model", "cluster-a", pool="default", index=0), + _replica_with_pool("my-model", "cluster-a", pool="default", index=2), + ], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-a", index=2, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + ], + ), + Case( + name="scale down packs off by dropping the highest index first", + deployment=_deployment(replicas=2), + # cluster-a hosts indices 0 and 1; cluster-b hosts index 0. Three + # replicas, want two. Highest index (a/1) is dropped, keeping the + # spread across a/0 and b/0. + clusters=[ + _cluster("cluster-a", pools=[_pool("default", nodes=4)]), + _cluster( + "cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + pools=[_pool("default", nodes=4)], + ), + ], + all_replicas=[ + _replica_with_pool("my-model", "cluster-a", pool="default", index=0), + _replica_with_pool("my-model", "cluster-a", pool="default", index=1), + _replica_with_pool("my-model", "cluster-b", pool="default", index=0), + ], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="retained replica is charged at its own node cost, not the new shape", + # The deployment's workers grew to pipeline=4 (4 nodes/replica), + # but the existing replica was created at pipeline=2 and is + # retained (no nodeSelector change rolls it). It still consumes + # only its original 2 nodes. The pool has 6, so a second replica + # at the new 4-node cost must still fit (6 - 2 = 4). Regression: + # charging the retained replica at the new shape (4) would leave + # 2 free and wrongly refuse the placement. + deployment=_deployment(replicas=2, pipeline=4), + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=6)])], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default", pipeline=2)], + # The retained replica is re-stamped to the deployment's current + # pipeline=4 shape but still charged its observed 2 nodes in the + # ledger. + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pipeline=4), + _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pipeline=4), + ], + ), + Case( + name="scale down drops from the most-loaded cluster to preserve spread", + deployment=_deployment(replicas=2), + # cluster-a hosts two replicas, cluster-b one. Scaling 3->2 must + # drop a's extra (a/1), NOT b's sole replica - otherwise we'd + # leave a packed and b empty, the opposite of spread. b's index + # is 3 (higher than a/1) to prove we drop by cluster load, not by + # a global index comparison. + clusters=[ + _cluster("cluster-a", pools=[_pool("default", nodes=4)]), + _cluster( + "cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + pools=[_pool("default", nodes=4)], + ), + ], + all_replicas=[ + _replica_with_pool("my-model", "cluster-a", pool="default", index=0), + _replica_with_pool("my-model", "cluster-a", pool="default", index=1), + _replica_with_pool("my-model", "cluster-b", pool="default", index=3), + ], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-b", index=3, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="co-located replicas are both retained across a reconcile", + deployment=_deployment(replicas=2), + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], + all_replicas=[ + _replica_with_pool("my-model", "cluster-a", pool="default", index=0), + _replica_with_pool("my-model", "cluster-a", pool="default", index=1), + ], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + ], + ), + Case( + name="scale down across clusters drops higher cluster name at equal index", + deployment=_deployment(replicas=1), + clusters=[ + _cluster("cluster-a"), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ], + all_replicas=[ + _replica("my-model", "cluster-b"), + _replica("my-model", "cluster-a"), + ], + # Both at index 0, so the (index, name) tiebreak keeps cluster-a. + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], + ), + Case( + name="new placement is alphabetical for determinism", + deployment=_deployment(replicas=2), + clusters=[ + _cluster("cluster-c", gateway_hostname="cluster-c.clusters.example.com"), + _cluster("cluster-a"), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ], + all_replicas=[], + want=[ + _cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="other deployment's replicas consume node capacity", + deployment=_deployment(pipeline=1), + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], + # other-model occupies the single node on cluster-a. + all_replicas=[_replica("other-model", "cluster-a")], + want=[], + ), + Case( + name="our own observed replicas don't double-count against us", + deployment=_deployment(pipeline=1), + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + # Retained on its pin: the single node it already occupies isn't + # charged against itself, so it stays rather than being evicted. + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], + ), + Case( + name="replica labeled for our deployment but pinned to unknown cluster is ignored", + deployment=_deployment(), + clusters=[_cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com")], + all_replicas=[_replica("my-model", "cluster-a")], + want=[_cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default")], + ), + Case( + name="another deployment pinned to a deleted pool consumes no capacity", + # other-model is pinned to pool "gone", which the cluster no + # longer publishes. Its pods are pinned to a node label no node + # carries, so they're unschedulable and occupy nothing. The one + # published node on "frontier" is therefore free for our replica. + # Charging the unattributable replica would wrongly report the + # cluster full. + deployment=_deployment(requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], + all_replicas=[_replica_with_pool("other-model", "cluster-a", pool="gone")], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved()], + ) + ], + ), + Case( + name="colliding (cluster, index) retains deterministically by replica name", + # Two of our replicas collide on (cluster-a, index 0) with + # different pinned pools. Retain keeps the first by replica name + # (my-model-cluster-a-0 on "a" sorts before the "-dup" replica on + # "b"), independent of input order, so the schedule is a function + # of state not of delivery order. Both pools match, so either + # would be a valid placement - only determinism is under test. + deployment=_deployment(requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("a", devices=[_gpu_device()]), + _pool("b", devices=[_gpu_device()]), ], - all_replicas=[], - want=[ - _cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ) + ], + all_replicas=[ + _collision_replica("my-model-cluster-a-0-dup", "cluster-a", pool="b", index=0), + _replica_with_pool("my-model", "cluster-a", pool="a", index=0), + ], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="a", + device_requests=[_resolved()], + ) + ], + ), +] + + +@pytest.mark.parametrize("case", SCHEDULE_CASES, ids=lambda case: case.name) +def test_schedule(case: Case) -> None: + """The scheduler retains existing pins and places new replicas.""" + got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) + assert got == case.want + + +# A caller passes fill=False when it can't yet trust the candidate set (for the +# ModelDeployment, when a referenced ModelCache is unresolved). Retain runs +# unconditionally; only the placement of new replicas is held. +FILL_FALSE_CASES = [ + Case( + name="no replicas yet: nothing is placed", + deployment=_deployment(), + clusters=[_cluster("cluster-a")], + all_replicas=[], + want=[], + ), + Case( + name="existing replica is retained despite fill=False", + deployment=_deployment(), + clusters=[_cluster("cluster-a")], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], + ), + Case( + name="scale-up shortfall is not filled, only the retained replica remains", + deployment=_deployment(replicas=3), + clusters=[ + _cluster("cluster-a"), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], + ), +] + + +@pytest.mark.parametrize("case", FILL_FALSE_CASES, ids=lambda case: case.name) +def test_fill_false_is_retain_only(case: Case) -> None: + """fill=False retains existing replicas but places no new ones.""" + got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas, fill=False) + assert got == case.want + + +# nodeSelector device-request matching and pool pinning. +NODE_SELECTOR_CASES = [ + Case( + name="matching request picks the cluster and records the pool", + deployment=_deployment(requests=[_request(cel_exprs=[_MEM_141])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], + all_replicas=[], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved()], + ) + ], + ), + Case( + name="non-matching request filters the cluster out", + deployment=_deployment(requests=[_request(cel_exprs=[_MEM_200])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], + all_replicas=[], + want=[], + ), + Case( + name="device count not covered filters out", + # Request 8 GPUs, pool device has only 4. + deployment=_deployment(requests=[_request(count=8, cel_exprs=[_MEM_141])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=4)])])], + all_replicas=[], + want=[], + ), + Case( + name="published device count of zero satisfies no request", + # A pool device published with count 0 must read as "none + # available", not default to 1. Regression: `d.count or 1` + # treated 0 as 1 and placed a replica whose ResourceClaim no + # device could satisfy. The status schema permits 0 even though + # an InferenceClass device count is floored at 1. + deployment=_deployment(requests=[_request(count=1, cel_exprs=[_MEM_141])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=0)])])], + all_replicas=[], + want=[], + ), + Case( + name="published pool node count of zero hosts nothing", + # An autoscaled-to-zero pool has a matching GPU device but no + # nodes, so it can host no replica. + deployment=_deployment(requests=[_request(count=1, cel_exprs=[_MEM_141])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=0)])], + all_replicas=[], + want=[], + ), + Case( + name="synthetic NIC device matches but is not in resolved requests", + deployment=_deployment( + requests=[ + _request(name="gpu", cel_exprs=[_MEM_141]), + _request(name="nic", cel_exprs=[_IB]), + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[_pool("frontier", devices=[_gpu_device(), _nic_device()])], + ) + ], + all_replicas=[], + # Only the claim: DRA gpu request is resolved; the synthetic nic + # matched for scheduling but isn't claimed. + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved(name="gpu")], + ) + ], + ), + Case( + name="multi-device: missing NIC filters the pool out", + deployment=_deployment( + requests=[ + _request(name="gpu", cel_exprs=[_MEM_141]), + _request(name="nic", cel_exprs=[_IB]), + ] + ), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device()])])], + all_replicas=[], + want=[], + ), + Case( + name="two requests cannot both claim one single-count device", + # Two distinct requests, each matching the same single GPU + # device. DRA allocates distinct devices per request, so a + # count:1 device can satisfy only one. The pool must not match. + deployment=_deployment( + requests=[ + _request(name="gpu-a", cel_exprs=[_MEM_141]), + _request(name="gpu-b", cel_exprs=[_MEM_141]), + ] + ), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=1)])])], + all_replicas=[], + want=[], + ), + Case( + name="two requests against one device must fit within its count", + # Two count:5 requests need 10 GPUs total; the device has 8. + # Capacity is consumed across requests, so the pool must not + # match (regression: an earlier version checked each request + # against the full device count independently). + deployment=_deployment( + requests=[ + _request(name="gpu-a", count=5, cel_exprs=[_MEM_141]), + _request(name="gpu-b", count=5, cel_exprs=[_MEM_141]), + ] + ), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=8)])])], + all_replicas=[], + want=[], + ), + Case( + name="two requests sharing a device fit when count covers both", + # 8-GPU device, two count:4 requests = 8 total. Both resolve. + deployment=_deployment( + requests=[ + _request(name="gpu-a", count=4, cel_exprs=[_MEM_141]), + _request(name="gpu-b", count=4, cel_exprs=[_MEM_141]), + ] + ), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=8)])])], + all_replicas=[], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[ + _resolved(name="gpu-a", count=4), + _resolved(name="gpu-b", count=4), ], - ), - Case( - name="other deployment's replicas consume node capacity", - deployment=_deployment(pipeline=1), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], - # other-model occupies the single node on cluster-a. - all_replicas=[_replica("other-model", "cluster-a")], - want=[], - ), - Case( - name="our own observed replicas don't double-count against us", - deployment=_deployment(pipeline=1), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - # Retained on its pin: the single node it already occupies isn't - # charged against itself, so it stays rather than being evicted. - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="replica labeled for our deployment but pinned to unknown cluster is ignored", - deployment=_deployment(), - clusters=[_cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com")], - all_replicas=[_replica("my-model", "cluster-a")], - want=[_cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default")], - ), - Case( - name="another deployment pinned to a deleted pool consumes no capacity", - # other-model is pinned to pool "gone", which the cluster no - # longer publishes. Its pods are pinned to a node label no node - # carries, so they're unschedulable and occupy nothing. The one - # published node on "frontier" is therefore free for our replica. - # Charging the unattributable replica would wrongly report the - # cluster full. - deployment=_deployment(requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], - all_replicas=[_replica_with_pool("other-model", "cluster-a", pool="gone")], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], - ) + ) + ], + ), + Case( + name="first matching pool wins (deterministic)", + # Both pools carry a claimable GPU; the synthetic NIC's link type + # is the discriminator. Only the infiniband pool satisfies the + # nic selector, so it wins regardless of pool order. + deployment=_deployment( + requests=[ + _request(name="gpu", cel_exprs=[_MEM_141]), + _request(name="nic", cel_exprs=[_IB]), + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("dev", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")]), + _pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), ], - ), - Case( - name="colliding (cluster, index) retains deterministically by replica name", - # Two of our replicas collide on (cluster-a, index 0) with - # different pinned pools. Retain keeps the first by replica name - # (my-model-cluster-a-0 on "a" sorts before the "-dup" replica on - # "b"), independent of input order, so the schedule is a function - # of state not of delivery order. Both pools match, so either - # would be a valid placement - only determinism is under test. - deployment=_deployment(requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("a", devices=[_gpu_device()]), - _pool("b", devices=[_gpu_device()]), - ], - ) + ) + ], + all_replicas=[], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved(name="gpu")], + ) + ], + ), + Case( + name="synthetic-only selector leaves nothing to claim, pool ineligible", + # The sole request matches a synthetic NIC. The replica's serving + # workload would have no ResourceClaim to bind GPUs through, so + # the pool is not a viable host and nothing is scheduled. + deployment=_deployment(requests=[_request(name="nic", cel_exprs=[_IB])]), + clusters=[ + _cluster( + "cluster-a", + pools=[_pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")])], + ) + ], + all_replicas=[], + want=[], + ), + Case( + name="retained replica keeps its pinned pool", + deployment=_deployment(requests=[_request(cel_exprs=[_MEM_141])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="frontier")], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved()], + ) + ], + ), + Case( + name="selector drift re-places replica onto a now-matching pool", + # A claimable GPU keeps both pools viable hosts; the synthetic + # NIC's link type is the drifting discriminator. + deployment=_deployment( + requests=[ + _request(name="gpu", cel_exprs=[_MEM_141]), + _request(name="nic", cel_exprs=[_IB]), + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("a", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")]), + _pool("b", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), ], - all_replicas=[ - _collision_replica("my-model-cluster-a-0-dup", "cluster-a", pool="b", index=0), - _replica_with_pool("my-model", "cluster-a", pool="a", index=0), + ) + ], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="b", + device_requests=[_resolved(name="gpu")], + ) + ], + ), + Case( + name="pinned pool that still matches stays pinned (attribute drift is sticky)", + deployment=_deployment( + requests=[ + _request(name="gpu", cel_exprs=[_MEM_141]), + _request(name="nic", cel_exprs=[_IB]), + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("a", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), + _pool("b", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), ], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="a", - device_requests=[_resolved()], - ) + ) + ], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="a", + device_requests=[_resolved(name="gpu")], + ) + ], + ), + Case( + name="no matching pool anywhere drops the replica entirely", + deployment=_deployment( + requests=[ + _request(name="gpu", cel_exprs=[_MEM_141]), + _request(name="nic", cel_exprs=[_IB]), + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[_pool("a", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")])], + ) + ], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], + want=[], + ), + Case( + name="replica with no pool pin is re-placed when a selector now applies", + deployment=_deployment( + requests=[ + _request(name="gpu", cel_exprs=[_MEM_141]), + _request(name="nic", cel_exprs=[_IB]), + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[_pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")])], + ) + ], + all_replicas=[_replica("my-model", "cluster-a")], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved(name="gpu")], + ) + ], + ), + Case( + name="dropping a non-matching replica frees its node for the refill", + # a/0 is pinned to a pool that still matches (retained). a/1 is + # pinned to a pool no longer published, so it's dropped and will + # be re-placed. The pool has just 2 nodes; both are notionally in + # use by a/0 and a/1. The refill must see a/1's node freeing up + # (it's being deleted) and re-place onto frontier at index 1. + # Regression: the ledger must not charge dropped replicas. + deployment=_deployment(replicas=2, requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=2)])], + all_replicas=[ + _replica_with_pool("my-model", "cluster-a", pool="frontier", index=0), + _replica_with_pool("my-model", "cluster-a", pool="gone", index=1), + ], + want=[ + _cand( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved()], + ), + _cand( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved()], + ), + ], + ), + Case( + name="device count is checked against the pinned pool, not a cluster-wide sum", + # Request 8 GPUs. Pool 'a' has 4/node (doesn't fit); pool 'b' + # has 8 and does. The replica must pin to 'b'. + deployment=_deployment(requests=[_request(count=8, cel_exprs=[_MEM_141])]), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("a", devices=[_gpu_device(count=4)]), + _pool("b", devices=[_gpu_device(count=8)]), ], - ), - ] + ) + ], + all_replicas=[], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="b", + device_requests=[_resolved(count=8)], + ) + ], + ), +] - for case in cases: - with self.subTest(case.name): - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) - self.assertEqual(case.want, got, f"{case.name}: -want, +got") - - def test_fill_false_is_retain_only(self) -> None: - """fill=False retains existing replicas but places no new ones. - - A caller passes fill=False when it can't yet trust the candidate set - (for the ModelDeployment, when a referenced ModelCache is unresolved). - Retain runs unconditionally; only the placement of new replicas is held. - """ - cases = [ - Case( - name="no replicas yet: nothing is placed", - deployment=_deployment(), - clusters=[_cluster("cluster-a")], - all_replicas=[], - want=[], - ), - Case( - name="existing replica is retained despite fill=False", - deployment=_deployment(), - clusters=[_cluster("cluster-a")], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="scale-up shortfall is not filled, only the retained replica remains", - deployment=_deployment(replicas=3), - clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - ] - for case in cases: - with self.subTest(case.name): - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas, fill=False) - self.assertEqual(case.want, got, f"{case.name}: -want, +got") - - -class TestScheduleNodeSelector(unittest.TestCase): - """Tests for nodeSelector device-request matching and pool pinning.""" - - def test_node_selector(self) -> None: - cases = [ - Case( - name="matching request picks the cluster and records the pool", - deployment=_deployment(requests=[_request(cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], - all_replicas=[], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], - ) - ], - ), - Case( - name="non-matching request filters the cluster out", - deployment=_deployment(requests=[_request(cel_exprs=[_MEM_200])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], - all_replicas=[], - want=[], - ), - Case( - name="device count not covered filters out", - # Request 8 GPUs, pool device has only 4. - deployment=_deployment(requests=[_request(count=8, cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=4)])])], - all_replicas=[], - want=[], - ), - Case( - name="published device count of zero satisfies no request", - # A pool device published with count 0 must read as "none - # available", not default to 1. Regression: `d.count or 1` - # treated 0 as 1 and placed a replica whose ResourceClaim no - # device could satisfy. The status schema permits 0 even though - # an InferenceClass device count is floored at 1. - deployment=_deployment(requests=[_request(count=1, cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=0)])])], - all_replicas=[], - want=[], - ), - Case( - name="published pool node count of zero hosts nothing", - # An autoscaled-to-zero pool has a matching GPU device but no - # nodes, so it can host no replica. - deployment=_deployment(requests=[_request(count=1, cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=0)])], - all_replicas=[], - want=[], - ), - Case( - name="synthetic NIC device matches but is not in resolved requests", - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device()])], - ) - ], - all_replicas=[], - # Only the claim: DRA gpu request is resolved; the synthetic nic - # matched for scheduling but isn't claimed. - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved(name="gpu")], - ) - ], - ), - Case( - name="multi-device: missing NIC filters the pool out", - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device()])])], - all_replicas=[], - want=[], - ), - Case( - name="two requests cannot both claim one single-count device", - # Two distinct requests, each matching the same single GPU - # device. DRA allocates distinct devices per request, so a - # count:1 device can satisfy only one. The pool must not match. - deployment=_deployment( - requests=[ - _request(name="gpu-a", cel_exprs=[_MEM_141]), - _request(name="gpu-b", cel_exprs=[_MEM_141]), - ] - ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=1)])])], - all_replicas=[], - want=[], - ), - Case( - name="two requests against one device must fit within its count", - # Two count:5 requests need 10 GPUs total; the device has 8. - # Capacity is consumed across requests, so the pool must not - # match (regression: an earlier version checked each request - # against the full device count independently). - deployment=_deployment( - requests=[ - _request(name="gpu-a", count=5, cel_exprs=[_MEM_141]), - _request(name="gpu-b", count=5, cel_exprs=[_MEM_141]), - ] - ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=8)])])], - all_replicas=[], - want=[], - ), - Case( - name="two requests sharing a device fit when count covers both", - # 8-GPU device, two count:4 requests = 8 total. Both resolve. - deployment=_deployment( - requests=[ - _request(name="gpu-a", count=4, cel_exprs=[_MEM_141]), - _request(name="gpu-b", count=4, cel_exprs=[_MEM_141]), - ] - ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=8)])])], - all_replicas=[], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[ - _resolved(name="gpu-a", count=4), - _resolved(name="gpu-b", count=4), - ], - ) - ], - ), - Case( - name="first matching pool wins (deterministic)", - # Both pools carry a claimable GPU; the synthetic NIC's link type - # is the discriminator. Only the infiniband pool satisfies the - # nic selector, so it wins regardless of pool order. - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("dev", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")]), - _pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), - ], - ) - ], - all_replicas=[], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved(name="gpu")], - ) - ], - ), - Case( - name="synthetic-only selector leaves nothing to claim, pool ineligible", - # The sole request matches a synthetic NIC. The replica's serving - # workload would have no ResourceClaim to bind GPUs through, so - # the pool is not a viable host and nothing is scheduled. - deployment=_deployment(requests=[_request(name="nic", cel_exprs=[_IB])]), - clusters=[ - _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")])], - ) - ], - all_replicas=[], - want=[], - ), - Case( - name="retained replica keeps its pinned pool", - deployment=_deployment(requests=[_request(cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="frontier")], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], - ) - ], - ), - Case( - name="selector drift re-places replica onto a now-matching pool", - # A claimable GPU keeps both pools viable hosts; the synthetic - # NIC's link type is the drifting discriminator. - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("a", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")]), - _pool("b", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), - ], - ) - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="b", - device_requests=[_resolved(name="gpu")], - ) - ], - ), - Case( - name="pinned pool that still matches stays pinned (attribute drift is sticky)", - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("a", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), - _pool("b", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), - ], - ) - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="a", - device_requests=[_resolved(name="gpu")], - ) - ], - ), - Case( - name="no matching pool anywhere drops the replica entirely", - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[_pool("a", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")])], - ) - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], - want=[], - ), - Case( - name="replica with no pool pin is re-placed when a selector now applies", - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")])], - ) - ], - all_replicas=[_replica("my-model", "cluster-a")], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved(name="gpu")], - ) - ], - ), - Case( - name="dropping a non-matching replica frees its node for the refill", - # a/0 is pinned to a pool that still matches (retained). a/1 is - # pinned to a pool no longer published, so it's dropped and will - # be re-placed. The pool has just 2 nodes; both are notionally in - # use by a/0 and a/1. The refill must see a/1's node freeing up - # (it's being deleted) and re-place onto frontier at index 1. - # Regression: the ledger must not charge dropped replicas. - deployment=_deployment(replicas=2, requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=2)])], - all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="frontier", index=0), - _replica_with_pool("my-model", "cluster-a", pool="gone", index=1), - ], - want=[ - _cand( - name="cluster-a", - index=0, - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], - ), - _cand( - name="cluster-a", - index=1, - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], - ), - ], - ), - Case( - name="device count is checked against the pinned pool, not a cluster-wide sum", - # Request 8 GPUs. Pool 'a' has 4/node (doesn't fit); pool 'b' - # has 8 and does. The replica must pin to 'b'. - deployment=_deployment(requests=[_request(count=8, cel_exprs=[_MEM_141])]), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("a", devices=[_gpu_device(count=4)]), - _pool("b", devices=[_gpu_device(count=8)]), - ], - ) - ], - all_replicas=[], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="b", - device_requests=[_resolved(count=8)], - ) - ], - ), - ] +@pytest.mark.parametrize("case", NODE_SELECTOR_CASES, ids=lambda case: case.name) +def test_node_selector(case: Case) -> None: + """The scheduler places replicas only on pools whose devices satisfy the nodeSelector.""" + got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) + assert got == case.want - for case in cases: - with self.subTest(case.name): - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) - self.assertEqual(case.want, got, f"{case.name}: -want, +got") - def test_invalid_cel_raises(self) -> None: - """A malformed expression raises CELCompileError (caller handles it).""" - deployment = _deployment(requests=[_request(cel_exprs=["this is ) not valid ("])]) - with self.assertRaises(cel.CELCompileError): - scheduling.schedule(deployment, [_cluster("cluster-a", pools=[_pool("frontier")])], []) +def test_node_selector_invalid_cel_raises() -> None: + """A malformed expression raises CELCompileError, which the caller handles.""" + deployment = _deployment(requests=[_request(cel_exprs=["this is ) not valid ("])]) + with pytest.raises(cel.CELCompileError, match=r"this is \) not valid \("): + scheduling.schedule(deployment, [_cluster("cluster-a", pools=[_pool("frontier")])], []) def _gang( @@ -1231,561 +1224,558 @@ def _gang( return mdv1alpha1.Engine(name=_ENGINE, members=[leader, worker]) -class TestScheduleMembers(unittest.TestCase): - """Tests for per-member placement: single-pool engines, rejection when no - pool fits, and claimless ride-along members.""" - - def test_members(self) -> None: - cases = [ - Case( - name="a single pool satisfying every member hosts the whole engine", - # The leader's request matches both pools; the worker's only - # matches big. The whole-engine pass must put both members on - # big - the one pool that satisfies them all. - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_200])], - ) - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("small", devices=[_gpu_device(memory="141Gi")]), - _pool("big", devices=[_gpu_device(memory="200Gi")]), - ], - ) +# Per-member placement: single-pool engines, rejection when no pool fits, and +# claimless ride-along members. +MEMBERS_CASES = [ + Case( + name="a single pool satisfying every member hosts the whole engine", + # The leader's request matches both pools; the worker's only + # matches big. The whole-engine pass must put both members on + # big - the one pool that satisfies them all. + deployment=_deployment( + engines=[ + _gang( + [_request(cel_exprs=[_MEM_141])], + [_request(cel_exprs=[_MEM_200])], + ) + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("small", devices=[_gpu_device(memory="141Gi")]), + _pool("big", devices=[_gpu_device(memory="200Gi")]), ], - all_replicas=[], - want=[ - scheduling.Candidate( - name="cluster-a", - index=0, - gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement( - role="Leader", pool="big", device_requests=[_resolved()] - ), - scheduling.MemberPlacement( - role="Worker", - pool="big", - device_requests=[_resolved(cel_exprs=[_MEM_200])], - ), - ], - ) + ) + ], + all_replicas=[], + want=[ + scheduling.Candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + engines=[ + scheduling.EnginePlacement( + name=_ENGINE, + members=[ + scheduling.MemberPlacement(role="Leader", pool="big", device_requests=[_resolved()]), + scheduling.MemberPlacement( + role="Worker", + pool="big", + device_requests=[_resolved(cel_exprs=[_MEM_200])], + ), ], ) ], - ), - Case( - name="members no single pool satisfies are not scheduled", - # The leader only fits big (>= 200Gi); the worker only fits - # small (< 200Gi). No single pool satisfies both. The scheduler - # never splits an engine across pools - it can't tell whether - # big and small share a fabric - so the engine is rejected and - # the replica goes unplaced (#149). - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_200])], - [_request(cel_exprs=[_MEM_LT_200])], - ) - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("small", devices=[_gpu_device(memory="141Gi")]), - _pool("big", devices=[_gpu_device(memory="200Gi")]), - ], - ) + ) + ], + ), + Case( + name="members no single pool satisfies are not scheduled", + # The leader only fits big (>= 200Gi); the worker only fits + # small (< 200Gi). No single pool satisfies both. The scheduler + # never splits an engine across pools - it can't tell whether + # big and small share a fabric - so the engine is rejected and + # the replica goes unplaced (#149). + deployment=_deployment( + engines=[ + _gang( + [_request(cel_exprs=[_MEM_200])], + [_request(cel_exprs=[_MEM_LT_200])], + ) + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("small", devices=[_gpu_device(memory="141Gi")]), + _pool("big", devices=[_gpu_device(memory="200Gi")]), ], - all_replicas=[], - want=[], - ), - Case( - name="a gang too big for its only matching pool is rejected", - # Both members match only big (>= 141Gi); small (40Gi) matches - # neither. big has one free node but the gang needs two. The - # engine doesn't fit any single pool, so it's rejected; with big - # the only cluster the replica goes unplaced. - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_141])], - ) - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("small", nodes=8, devices=[_gpu_device(memory="40Gi")]), - _pool("big", nodes=1, devices=[_gpu_device(memory="141Gi")]), + ) + ], + all_replicas=[], + want=[], + ), + Case( + name="a gang too big for its only matching pool is rejected", + # Both members match only big (>= 141Gi); small (40Gi) matches + # neither. big has one free node but the gang needs two. The + # engine doesn't fit any single pool, so it's rejected; with big + # the only cluster the replica goes unplaced. + deployment=_deployment( + engines=[ + _gang( + [_request(cel_exprs=[_MEM_141])], + [_request(cel_exprs=[_MEM_141])], + ) + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("small", nodes=8, devices=[_gpu_device(memory="40Gi")]), + _pool("big", nodes=1, devices=[_gpu_device(memory="141Gi")]), + ], + ) + ], + all_replicas=[], + want=[], + ), + Case( + name="a gang too big for one cluster's pool lands whole on another", + # Same gang. cluster-a's matching pool has only one free node + # (too few for the two-member gang), so the scheduler rejects + # cluster-a and places the whole gang on cluster-b, whose pool + # has room for both members. + deployment=_deployment( + engines=[ + _gang( + [_request(cel_exprs=[_MEM_141])], + [_request(cel_exprs=[_MEM_141])], + ) + ] + ), + clusters=[ + _cluster( + "cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pools=[ + _pool("small", nodes=8, devices=[_gpu_device(memory="40Gi")]), + _pool("big", nodes=1, devices=[_gpu_device(memory="141Gi")]), + ], + ), + _cluster( + "cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + pools=[_pool("big", nodes=2, devices=[_gpu_device(memory="141Gi")])], + ), + ], + all_replicas=[], + want=[ + scheduling.Candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + engines=[ + scheduling.EnginePlacement( + name=_ENGINE, + members=[ + scheduling.MemberPlacement(role="Leader", pool="big", device_requests=[_resolved()]), + scheduling.MemberPlacement(role="Worker", pool="big", device_requests=[_resolved()]), ], ) ], - all_replicas=[], - want=[], - ), - Case( - name="a gang too big for one cluster's pool lands whole on another", - # Same gang. cluster-a's matching pool has only one free node - # (too few for the two-member gang), so the scheduler rejects - # cluster-a and places the whole gang on cluster-b, whose pool - # has room for both members. - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_141])], - ) - ] - ), - clusters=[ - _cluster( - "cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pools=[ - _pool("small", nodes=8, devices=[_gpu_device(memory="40Gi")]), - _pool("big", nodes=1, devices=[_gpu_device(memory="141Gi")]), + ) + ], + ), + Case( + name="a member claimable elsewhere is not stranded on a synthetic match", + # On pool-a the leader's request matches only a Synthetic + # device (nothing to claim) while the worker claims, so the + # whole engine *could* land there - but pool-b satisfies the + # leader claimably. The engine must go to pool-b; placing on + # pool-a would run the leader without the GPU it asked for. + deployment=_deployment( + engines=[ + _gang( + [_request(cel_exprs=[_MEM_200])], + [_request(cel_exprs=[_MEM_141])], + ) + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool( + "a", + devices=[ + _gpu_device(memory="141Gi"), + _gpu_device(name="syn", claim="Synthetic", memory="200Gi"), ], ), - _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("big", nodes=2, devices=[_gpu_device(memory="141Gi")])], - ), - ], - all_replicas=[], - want=[ - scheduling.Candidate( - name="cluster-b", - index=0, - gateway_hostname="cluster-b.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement( - role="Leader", pool="big", device_requests=[_resolved()] - ), - scheduling.MemberPlacement( - role="Worker", pool="big", device_requests=[_resolved()] - ), - ], - ) - ], - ) + _pool("b", devices=[_gpu_device(memory="200Gi")]), ], - ), - Case( - name="a member claimable elsewhere is not stranded on a synthetic match", - # On pool-a the leader's request matches only a Synthetic - # device (nothing to claim) while the worker claims, so the - # whole engine *could* land there - but pool-b satisfies the - # leader claimably. The engine must go to pool-b; placing on - # pool-a would run the leader without the GPU it asked for. - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_200])], - [_request(cel_exprs=[_MEM_141])], - ) - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool( - "a", - devices=[ - _gpu_device(memory="141Gi"), - _gpu_device(name="syn", claim="Synthetic", memory="200Gi"), - ], + ) + ], + all_replicas=[], + want=[ + scheduling.Candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + engines=[ + scheduling.EnginePlacement( + name=_ENGINE, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="b", + device_requests=[_resolved(cel_exprs=[_MEM_200])], + ), + scheduling.MemberPlacement( + role="Worker", + pool="b", + device_requests=[_resolved(cel_exprs=[_MEM_141])], ), - _pool("b", devices=[_gpu_device(memory="200Gi")]), - ], - ) - ], - all_replicas=[], - want=[ - scheduling.Candidate( - name="cluster-a", - index=0, - gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement( - role="Leader", - pool="b", - device_requests=[_resolved(cel_exprs=[_MEM_200])], - ), - scheduling.MemberPlacement( - role="Worker", - pool="b", - device_requests=[_resolved(cel_exprs=[_MEM_141])], - ), - ], - ) ], ) ], - ), - Case( - name="a member synthetic-only everywhere places claimless with its gang", - # The leader's request matches only the pool's synthetic NIC on - # every pool - deliberate (a selector that pins without - # claiming). It places claimless alongside the claiming worker. - deployment=_deployment( - engines=[ - _gang( - [_request(name="nic", cel_exprs=[_IB])], - [_request(cel_exprs=[_MEM_141])], - ) - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device()])], - ) - ], - all_replicas=[], - want=[ - scheduling.Candidate( - name="cluster-a", - index=0, - gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), - scheduling.MemberPlacement( - role="Worker", pool="frontier", device_requests=[_resolved()] - ), - ], - ) + ) + ], + ), + Case( + name="a member synthetic-only everywhere places claimless with its gang", + # The leader's request matches only the pool's synthetic NIC on + # every pool - deliberate (a selector that pins without + # claiming). It places claimless alongside the claiming worker. + deployment=_deployment( + engines=[ + _gang( + [_request(name="nic", cel_exprs=[_IB])], + [_request(cel_exprs=[_MEM_141])], + ) + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[_pool("frontier", devices=[_gpu_device(), _nic_device()])], + ) + ], + all_replicas=[], + want=[ + scheduling.Candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + engines=[ + scheduling.EnginePlacement( + name=_ENGINE, + members=[ + scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), + scheduling.MemberPlacement(role="Worker", pool="frontier", device_requests=[_resolved()]), ], ) ], - ), - Case( - name="a member that matches nowhere fails the whole replica", - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_200])], - ) - ] - ), - clusters=[_cluster("cluster-a", pools=[_pool("default", devices=[_gpu_device(memory="141Gi")])])], - all_replicas=[], - want=[], - ), - Case( - name="a claimless leader rides along on its gang's pool at zero cost", - # The leader carries no nodeSelector: it claims nothing, follows - # the worker's pool, and costs no nodes - the 1-node pool fits - # the whole gang because only the worker occupies a node. - deployment=_deployment(engines=[_gang(None, [_request(cel_exprs=[_MEM_141])])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], - all_replicas=[], - want=[ - scheduling.Candidate( - name="cluster-a", - index=0, - gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), - scheduling.MemberPlacement( - role="Worker", pool="frontier", device_requests=[_resolved()] - ), - ], - ) + ) + ], + ), + Case( + name="a member that matches nowhere fails the whole replica", + deployment=_deployment( + engines=[ + _gang( + [_request(cel_exprs=[_MEM_141])], + [_request(cel_exprs=[_MEM_200])], + ) + ] + ), + clusters=[_cluster("cluster-a", pools=[_pool("default", devices=[_gpu_device(memory="141Gi")])])], + all_replicas=[], + want=[], + ), + Case( + name="a claimless leader rides along on its gang's pool at zero cost", + # The leader carries no nodeSelector: it claims nothing, follows + # the worker's pool, and costs no nodes - the 1-node pool fits + # the whole gang because only the worker occupies a node. + deployment=_deployment(engines=[_gang(None, [_request(cel_exprs=[_MEM_141])])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], + all_replicas=[], + want=[ + scheduling.Candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + engines=[ + scheduling.EnginePlacement( + name=_ENGINE, + members=[ + scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), + scheduling.MemberPlacement(role="Worker", pool="frontier", device_requests=[_resolved()]), ], ) ], - ), - Case( - name="a retained replica's claimless member keeps its pin", - deployment=_deployment(engines=[_gang(None, [_request(cel_exprs=[_MEM_141])])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], - all_replicas=[ - _replica( - "my-model", - "cluster-a", - engines=[ - mrv1alpha1.Engine( - name=_ENGINE, - members=[ - mrv1alpha1.Member( - role="Leader", - nodePoolName="frontier", - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - mrv1alpha1.Member( - role="Worker", - worker=mrv1alpha1.Worker(nodes=1), - nodePoolName="frontier", - deviceRequests=_replica_device_requests(), - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - ], - ) + ) + ], + ), + Case( + name="a retained replica's claimless member keeps its pin", + deployment=_deployment(engines=[_gang(None, [_request(cel_exprs=[_MEM_141])])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], + all_replicas=[ + _replica( + "my-model", + "cluster-a", + engines=[ + mrv1alpha1.Engine( + name=_ENGINE, + members=[ + mrv1alpha1.Member( + role="Leader", + nodePoolName="frontier", + template=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") + ] + ) + ), + ), + mrv1alpha1.Member( + role="Worker", + worker=mrv1alpha1.Worker(nodes=1), + nodePoolName="frontier", + deviceRequests=_replica_device_requests(), + template=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") + ] + ) + ), + ), ], ) ], - want=[ - scheduling.Candidate( - name="cluster-a", - index=0, - gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), - scheduling.MemberPlacement( - role="Worker", pool="frontier", device_requests=[_resolved()] - ), - ], - ) + ) + ], + want=[ + scheduling.Candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + engines=[ + scheduling.EnginePlacement( + name=_ENGINE, + members=[ + scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), + scheduling.MemberPlacement(role="Worker", pool="frontier", device_requests=[_resolved()]), ], ) ], - ), - Case( - name="another deployment's claimless member consumes no capacity", - # other-model's gang occupies only its worker's node: its - # claimless leader shares that node. The 2-node pool has 1 node - # free, so our 1-node deployment fits. Charging the claimless - # leader a node would wrongly report insufficient capacity. - deployment=_deployment(), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], - all_replicas=[ - _replica( - "other-model", - "cluster-a", - engines=[ - mrv1alpha1.Engine( - name="main", - members=[ - mrv1alpha1.Member( - role="Leader", - nodePoolName="default", - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - mrv1alpha1.Member( - role="Worker", - worker=mrv1alpha1.Worker(nodes=1), - nodePoolName="default", - deviceRequests=_replica_device_requests(), - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - ], - ) + ) + ], + ), + Case( + name="another deployment's claimless member consumes no capacity", + # other-model's gang occupies only its worker's node: its + # claimless leader shares that node. The 2-node pool has 1 node + # free, so our 1-node deployment fits. Charging the claimless + # leader a node would wrongly report insufficient capacity. + deployment=_deployment(), + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], + all_replicas=[ + _replica( + "other-model", + "cluster-a", + engines=[ + mrv1alpha1.Engine( + name="main", + members=[ + mrv1alpha1.Member( + role="Leader", + nodePoolName="default", + template=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") + ] + ) + ), + ), + mrv1alpha1.Member( + role="Worker", + worker=mrv1alpha1.Worker(nodes=1), + nodePoolName="default", + deviceRequests=_replica_device_requests(), + template=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") + ] + ) + ), + ), ], ) ], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="a member shape change re-places the replica", - # The deployment grew a Worker (Standalone -> Leader+Worker). - # The observed single-member replica no longer lines up, so it - # is re-placed with the new shape. - deployment=_deployment(engines=[_gang([_request()], [_request()])]), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], - all_replicas=[_replica("my-model", "cluster-a")], - want=[ - scheduling.Candidate( - name="cluster-a", - index=0, - gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement( - role="Leader", pool="default", device_requests=[_resolved()] - ), - scheduling.MemberPlacement( - role="Worker", pool="default", device_requests=[_resolved()] - ), - ], - ) + ) + ], + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], + ), + Case( + name="a member shape change re-places the replica", + # The deployment grew a Worker (Standalone -> Leader+Worker). + # The observed single-member replica no longer lines up, so it + # is re-placed with the new shape. + deployment=_deployment(engines=[_gang([_request()], [_request()])]), + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], + all_replicas=[_replica("my-model", "cluster-a")], + want=[ + scheduling.Candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + engines=[ + scheduling.EnginePlacement( + name=_ENGINE, + members=[ + scheduling.MemberPlacement(role="Leader", pool="default", device_requests=[_resolved()]), + scheduling.MemberPlacement(role="Worker", pool="default", device_requests=[_resolved()]), ], ) ], - ), - ] + ) + ], + ), +] - for case in cases: - with self.subTest(case.name): - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) - self.assertEqual(case.want, got, f"{case.name}: -want, +got") +@pytest.mark.parametrize("case", MEMBERS_CASES, ids=lambda case: case.name) +def test_members(case: Case) -> None: + """The scheduler places every member of an engine on one pool that fits them all.""" + got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) + assert got == case.want -class TestScheduleTaints(unittest.TestCase): - """Taints on InferenceClusters gate placement; a matching toleration on the - ModelDeployment overrides them. NoSchedule keeps new replicas off a cluster - but leaves existing ones; NoExecute additionally drains the existing ones, - which fill reschedules onto a tolerated cluster.""" - _MAINT = icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule") - _DECOMM = icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute") +# Taints on InferenceClusters gate placement; a matching toleration on the +# ModelDeployment overrides them. NoSchedule keeps new replicas off a cluster but +# leaves existing ones; NoExecute additionally drains the existing ones, which +# fill reschedules onto a tolerated cluster. +_MAINT = icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule") +_DECOMM = icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute") - def _names(self, got: list[scheduling.Candidate]) -> list[tuple[str, int]]: - return [(c.name, c.index) for c in got] - def test_noschedule_keeps_new_replicas_off(self) -> None: - clusters = [ - _cluster("cluster-a", taints=[self._MAINT]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1), clusters, []) - self.assertEqual(self._names(got), [("cluster-b", 0)]) - - def test_noschedule_leaves_existing_replica_in_place(self) -> None: - existing = _replica("my-model", "cluster-a") - got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a", taints=[self._MAINT])], [existing]) - self.assertEqual(self._names(got), [("cluster-a", 0)]) - - def test_toleration_allows_placement_on_tainted(self) -> None: - tol = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists") - got = scheduling.schedule( - _deployment(replicas=1, tolerations=[tol]), [_cluster("cluster-a", taints=[self._MAINT])], [] - ) - self.assertEqual(self._names(got), [("cluster-a", 0)]) +def _names(got: list[scheduling.Candidate]) -> list[tuple[str, int]]: + return [(c.name, c.index) for c in got] - def test_noexecute_drains_and_reschedules(self) -> None: - existing = _replica("my-model", "cluster-a") - clusters = [ - _cluster("cluster-a", taints=[self._DECOMM]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1), clusters, [existing]) - self.assertEqual(self._names(got), [("cluster-b", 0)]) - - def test_noexecute_toleration_retains_in_place(self) -> None: - tol = mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists") - existing = _replica("my-model", "cluster-a") - got = scheduling.schedule( - _deployment(replicas=1, tolerations=[tol]), [_cluster("cluster-a", taints=[self._DECOMM])], [existing] - ) - self.assertEqual(self._names(got), [("cluster-a", 0)]) - - def test_noexecute_drain_leaves_count_unmet_when_nowhere_to_go(self) -> None: - """Draining with no tolerated cluster to reschedule onto yields fewer - than spec.replicas; the deploy function surfaces the shortfall.""" - existing = _replica("my-model", "cluster-a") - got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a", taints=[self._DECOMM])], [existing]) - self.assertEqual(got, []) - - def test_noschedule_toleration_does_not_cover_a_noexecute_taint(self) -> None: - """Matching the key but not the effect doesn't tolerate: an operator who - tolerates only NoSchedule is still drained by a NoExecute taint.""" - tol = mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists", effect="NoSchedule") - existing = _replica("my-model", "cluster-a") - clusters = [ - _cluster("cluster-a", taints=[self._DECOMM]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, [existing]) - self.assertEqual(self._names(got), [("cluster-b", 0)]) - - def test_untolerated_second_taint_still_repels(self) -> None: - """Tolerating one of a cluster's taints isn't enough; any untolerated - taint keeps new replicas off.""" - other = icv1alpha1.Taint(key="modelplane.ai/reserved", effect="NoSchedule") - tol = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists") - clusters = [ - _cluster("cluster-a", taints=[self._MAINT, other]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, []) - self.assertEqual(self._names(got), [("cluster-b", 0)]) - - def test_equal_toleration_matches_on_value(self) -> None: - """Equal tolerates only when key and value both match.""" - match = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="on") - placed = scheduling.schedule( - _deployment(replicas=1, tolerations=[match]), [_cluster("cluster-a", taints=[self._MAINT])], [] - ) - self.assertEqual(self._names(placed), [("cluster-a", 0)]) - mismatch = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="off") - repelled = scheduling.schedule( - _deployment(replicas=1, tolerations=[mismatch]), [_cluster("cluster-a", taints=[self._MAINT])], [] - ) - self.assertEqual(repelled, []) +def test_taints_noschedule_keeps_new_replicas_off() -> None: + """A NoSchedule taint keeps a new replica off the cluster.""" + clusters = [ + _cluster("cluster-a", taints=[_MAINT]), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ] + got = scheduling.schedule(_deployment(replicas=1), clusters, []) + assert _names(got) == [("cluster-b", 0)] - def test_keyless_exists_tolerates_every_taint(self) -> None: - """An Exists toleration with no key tolerates any taint on the cluster.""" - tol = mdv1alpha1.Toleration(operator="Exists") - clusters = [_cluster("cluster-a", taints=[self._MAINT, self._DECOMM])] - got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, []) - self.assertEqual(self._names(got), [("cluster-a", 0)]) +def test_taints_noschedule_leaves_existing_replica_in_place() -> None: + """A NoSchedule taint leaves an existing replica where it is.""" + existing = _replica("my-model", "cluster-a") + got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a", taints=[_MAINT])], [existing]) + assert _names(got) == [("cluster-a", 0)] -class TestPlacementLabels(unittest.TestCase): - """A cluster's placement labels reach the Candidate, and so the ModelReplica - and ModelEndpoint composed from it. - This is how a self-hosted endpoint gets its region: a ModelService selects - endpoints by label, so without it a region-scoped service can't select its - own replicas, and it can't label them by hand because Modelplane owns them. - """ +def test_taints_toleration_allows_placement_on_tainted() -> None: + """A matching toleration lets a new replica onto a tainted cluster.""" + tol = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists") + got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), [_cluster("cluster-a", taints=[_MAINT])], []) + assert _names(got) == [("cluster-a", 0)] + + +def test_taints_noexecute_drains_and_reschedules() -> None: + """A NoExecute taint drains an existing replica onto another cluster.""" + existing = _replica("my-model", "cluster-a") + clusters = [ + _cluster("cluster-a", taints=[_DECOMM]), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ] + got = scheduling.schedule(_deployment(replicas=1), clusters, [existing]) + assert _names(got) == [("cluster-b", 0)] + + +def test_taints_noexecute_toleration_retains_in_place() -> None: + """A NoExecute toleration keeps an existing replica on a draining cluster.""" + tol = mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists") + existing = _replica("my-model", "cluster-a") + got = scheduling.schedule( + _deployment(replicas=1, tolerations=[tol]), [_cluster("cluster-a", taints=[_DECOMM])], [existing] + ) + assert _names(got) == [("cluster-a", 0)] + + +def test_taints_noexecute_drain_leaves_count_unmet_when_nowhere_to_go() -> None: + """Draining with no tolerated cluster to go to yields fewer than spec.replicas.""" + # The deploy function surfaces the shortfall. + existing = _replica("my-model", "cluster-a") + got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a", taints=[_DECOMM])], [existing]) + assert got == [] + + +def test_taints_noschedule_toleration_does_not_cover_a_noexecute_taint() -> None: + """A toleration that matches the key but not the effect doesn't tolerate.""" + # An operator who tolerates only NoSchedule is still drained by a NoExecute + # taint. + tol = mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists", effect="NoSchedule") + existing = _replica("my-model", "cluster-a") + clusters = [ + _cluster("cluster-a", taints=[_DECOMM]), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ] + got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, [existing]) + assert _names(got) == [("cluster-b", 0)] + + +def test_taints_untolerated_second_taint_still_repels() -> None: + """Any untolerated taint keeps new replicas off, even if another is tolerated.""" + other = icv1alpha1.Taint(key="modelplane.ai/reserved", effect="NoSchedule") + tol = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists") + clusters = [ + _cluster("cluster-a", taints=[_MAINT, other]), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ] + got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, []) + assert _names(got) == [("cluster-b", 0)] + + +def test_taints_equal_toleration_matches_on_value() -> None: + """An Equal toleration tolerates only when key and value both match.""" + match = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="on") + placed = scheduling.schedule( + _deployment(replicas=1, tolerations=[match]), [_cluster("cluster-a", taints=[_MAINT])], [] + ) + assert _names(placed) == [("cluster-a", 0)] + + mismatch = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="off") + repelled = scheduling.schedule( + _deployment(replicas=1, tolerations=[mismatch]), [_cluster("cluster-a", taints=[_MAINT])], [] + ) + assert repelled == [] + + +def test_taints_keyless_exists_tolerates_every_taint() -> None: + """An Exists toleration with no key tolerates any taint on the cluster.""" + tol = mdv1alpha1.Toleration(operator="Exists") + clusters = [_cluster("cluster-a", taints=[_MAINT, _DECOMM])] + got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, []) + assert _names(got) == [("cluster-a", 0)] + + +# A cluster's placement labels reach the Candidate, and so the ModelReplica and +# ModelEndpoint composed from it. +# +# This is how a self-hosted endpoint gets its region: a ModelService selects +# endpoints by label, so without it a region-scoped service can't select its own +# replicas, and it can't label them by hand because Modelplane owns them. + + +def test_placement_labels_reach_the_candidate() -> None: + """A cluster's placement labels reach the Candidate.""" + got = scheduling.schedule( + _deployment(replicas=1), + [_cluster("cluster-a", placement_labels={"example.org/region": "eu"})], + [], + ) + assert [c.placement_labels for c in got] == [{"example.org/region": "eu"}] - def test_labels_reach_the_candidate(self) -> None: - got = scheduling.schedule( - _deployment(replicas=1), - [_cluster("cluster-a", placement_labels={"example.org/region": "eu"})], - [], - ) - self.assertEqual([c.placement_labels for c in got], [{"example.org/region": "eu"}]) - def test_a_cluster_declaring_none_yields_none(self) -> None: - got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a")], []) - self.assertEqual([c.placement_labels for c in got], [{}]) +def test_placement_labels_a_cluster_declaring_none_yields_none() -> None: + """A cluster that declares no placement labels yields a Candidate with none.""" + got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a")], []) + assert [c.placement_labels for c in got] == [{}] diff --git a/functions/compose-model-deployment/tests/test_semver.py b/functions/compose-model-deployment/tests/test_semver.py index e7b58ae85..433a3f058 100644 --- a/functions/compose-model-deployment/tests/test_semver.py +++ b/functions/compose-model-deployment/tests/test_semver.py @@ -25,12 +25,12 @@ Upstream cases that don't apply: the compile-time overload error (isSemver([1,2,3])) - celpy doesn't type-check overloads; and the runtime parse error for semver("v1.0") - upstream raises, we treat a bad version as a -non-match (driven through the parse layer in TestParseRejects). +non-match (driven through the parse layer in test_parse_rejects). """ import dataclasses -import unittest +import pytest from function import cel, semver @@ -52,79 +52,80 @@ class ParseErrCase: input: str -class TestSemverCEL(unittest.TestCase): - """Mirrors semver_test.go TestSemver (and the doc-comment examples).""" - - def test_semver(self) -> None: - cases = [ - # parse + doc-comment examples. - Case(name="parse", expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', want=True), - Case(name="parse with prerelease", expr='semver("0.1.0-alpha.1").major() == 0', want=True), - # isSemver strict. - Case(name="isSemver full", expr='isSemver("1.2.3-beta.1+build.1")', want=True), - Case(name="isSemver simple", expr='isSemver("1.0.0")', want=True), - Case(name="isSemver hello", expr='isSemver("hello")', want=False), - Case(name="isSemver empty false", expr='isSemver("")', want=False), - Case(name="isSemver v prefix false", expr='isSemver("v1.0.0")', want=False), - Case(name="isSemver v1.0 false", expr='isSemver("v1.0")', want=False), - Case(name="isSemver leading whitespace false", expr='isSemver(" 1.0.0")', want=False), - Case(name="isSemver inner whitespace false", expr='isSemver("1. 0.0")', want=False), - Case(name="isSemver trailing whitespace false", expr='isSemver("1.0.0 ")', want=False), - Case(name="isSemver leading zeros false", expr='isSemver("01.01.01")', want=False), - Case(name="isSemver major only false", expr='isSemver("1")', want=False), - Case(name="isSemver major minor only false", expr='isSemver("1.1")', want=False), - Case(name="isSemver 200K", expr='isSemver("200K")', want=False), - Case(name="isSemver Mi", expr='isSemver("Mi")', want=False), - # isSemver normalize overload. Normalization does NOT trim whitespace. - Case(name="isSemver empty normalize false", expr='isSemver("", true)', want=False), - Case(name="isSemver leading whitespace normalize false", expr='isSemver(" 1.0.0", true)', want=False), - Case(name="isSemver inner whitespace normalize false", expr='isSemver("1. 0.0", true)', want=False), - Case(name="isSemver trailing whitespace normalize false", expr='isSemver("1.0.0 ", true)', want=False), - Case(name="isSemver v prefix normalize true", expr='isSemver("v1.0.0", true)', want=True), - Case(name="isSemver leading zeros normalize true", expr='isSemver("01.01.01", true)', want=True), - Case(name="isSemver major only normalize true", expr='isSemver("1", true)', want=True), - Case(name="isSemver major minor only normalize true", expr='isSemver("1.1", true)', want=True), - # normalize equality and semver(...) examples. - Case(name="equality normalize", expr='semver("v01.01", true) == semver("1.1.0")', want=True), - Case(name="semver v prefix normalize major", expr='semver("v1.0.0", true).major() == 1', want=True), - Case(name="semver short normalize patch", expr='semver("1.0", true).patch() == 0', want=True), - Case(name="semver leading zeros normalize", expr='semver("01.01.01", true).minor() == 1', want=True), - # equality / comparison. - Case(name="equality reflexivity", expr='semver("1.2.3") == semver("1.2.3")', want=True), - Case(name="inequality", expr='semver("1.2.3") == semver("1.0.0")', want=False), - Case(name="less", expr='semver("1.0.0").isLessThan(semver("1.2.3"))', want=True), - Case(name="less false", expr='semver("1.0.0").isLessThan(semver("1.0.0"))', want=False), - Case(name="greater", expr='semver("1.2.3").isGreaterThan(semver("1.0.0"))', want=True), - Case(name="greater false", expr='semver("1.0.0").isGreaterThan(semver("1.0.0"))', want=False), - Case(name="compare equal", expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', want=True), - Case(name="compare less", expr='semver("1.2.3").compareTo(semver("2.0.0")) == -1', want=True), - Case(name="compare greater", expr='semver("1.2.3").compareTo(semver("0.1.2")) == 1', want=True), - # major / minor / patch. - Case(name="major", expr='semver("1.2.3").major() == 1', want=True), - Case(name="minor", expr='semver("1.2.3").minor() == 2', want=True), - Case(name="patch", expr='semver("1.2.3").patch() == 3', want=True), - # A bad version is a runtime error upstream -> non-match here. - Case(name="bad version is non-match", expr='semver("v1.0").major() == 1', want=False), - ] - for case in cases: - with self.subTest(case.name): - self.assertEqual(case.want, _eval(case.expr), f"{case.name}: -want, +got") - - -class TestParseRejects(unittest.TestCase): +SEMVER_CASES = [ + # parse + doc-comment examples. + Case(name="parse", expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', want=True), + Case(name="parse with prerelease", expr='semver("0.1.0-alpha.1").major() == 0', want=True), + # isSemver strict. + Case(name="isSemver full", expr='isSemver("1.2.3-beta.1+build.1")', want=True), + Case(name="isSemver simple", expr='isSemver("1.0.0")', want=True), + Case(name="isSemver hello", expr='isSemver("hello")', want=False), + Case(name="isSemver empty false", expr='isSemver("")', want=False), + Case(name="isSemver v prefix false", expr='isSemver("v1.0.0")', want=False), + Case(name="isSemver v1.0 false", expr='isSemver("v1.0")', want=False), + Case(name="isSemver leading whitespace false", expr='isSemver(" 1.0.0")', want=False), + Case(name="isSemver inner whitespace false", expr='isSemver("1. 0.0")', want=False), + Case(name="isSemver trailing whitespace false", expr='isSemver("1.0.0 ")', want=False), + Case(name="isSemver leading zeros false", expr='isSemver("01.01.01")', want=False), + Case(name="isSemver major only false", expr='isSemver("1")', want=False), + Case(name="isSemver major minor only false", expr='isSemver("1.1")', want=False), + Case(name="isSemver 200K", expr='isSemver("200K")', want=False), + Case(name="isSemver Mi", expr='isSemver("Mi")', want=False), + # isSemver normalize overload. Normalization does NOT trim whitespace. + Case(name="isSemver empty normalize false", expr='isSemver("", true)', want=False), + Case(name="isSemver leading whitespace normalize false", expr='isSemver(" 1.0.0", true)', want=False), + Case(name="isSemver inner whitespace normalize false", expr='isSemver("1. 0.0", true)', want=False), + Case(name="isSemver trailing whitespace normalize false", expr='isSemver("1.0.0 ", true)', want=False), + Case(name="isSemver v prefix normalize true", expr='isSemver("v1.0.0", true)', want=True), + Case(name="isSemver leading zeros normalize true", expr='isSemver("01.01.01", true)', want=True), + Case(name="isSemver major only normalize true", expr='isSemver("1", true)', want=True), + Case(name="isSemver major minor only normalize true", expr='isSemver("1.1", true)', want=True), + # normalize equality and semver(...) examples. + Case(name="equality normalize", expr='semver("v01.01", true) == semver("1.1.0")', want=True), + Case(name="semver v prefix normalize major", expr='semver("v1.0.0", true).major() == 1', want=True), + Case(name="semver short normalize patch", expr='semver("1.0", true).patch() == 0', want=True), + Case(name="semver leading zeros normalize", expr='semver("01.01.01", true).minor() == 1', want=True), + # equality / comparison. + Case(name="equality reflexivity", expr='semver("1.2.3") == semver("1.2.3")', want=True), + Case(name="inequality", expr='semver("1.2.3") == semver("1.0.0")', want=False), + Case(name="less", expr='semver("1.0.0").isLessThan(semver("1.2.3"))', want=True), + Case(name="less false", expr='semver("1.0.0").isLessThan(semver("1.0.0"))', want=False), + Case(name="greater", expr='semver("1.2.3").isGreaterThan(semver("1.0.0"))', want=True), + Case(name="greater false", expr='semver("1.0.0").isGreaterThan(semver("1.0.0"))', want=False), + Case(name="compare equal", expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', want=True), + Case(name="compare less", expr='semver("1.2.3").compareTo(semver("2.0.0")) == -1', want=True), + Case(name="compare greater", expr='semver("1.2.3").compareTo(semver("0.1.2")) == 1', want=True), + # major / minor / patch. + Case(name="major", expr='semver("1.2.3").major() == 1', want=True), + Case(name="minor", expr='semver("1.2.3").minor() == 2', want=True), + Case(name="patch", expr='semver("1.2.3").patch() == 3', want=True), + # A bad version is a runtime error upstream -> non-match here. + Case(name="bad version is non-match", expr='semver("v1.0").major() == 1', want=False), +] + + +@pytest.mark.parametrize("case", SEMVER_CASES, ids=lambda case: case.name) +def test_semver(case: Case) -> None: + """A semver CEL expression evaluates as it does upstream.""" + assert _eval(case.expr) == case.want + + +PARSE_REJECTS_CASES = [ + ParseErrCase(name="v prefix", input="v1.0"), + ParseErrCase(name="major only", input="1"), + ParseErrCase(name="major minor only", input="1.1"), + ParseErrCase(name="leading zeros", input="01.01.01"), + ParseErrCase(name="leading whitespace", input=" 1.0.0"), + ParseErrCase(name="trailing whitespace", input="1.0.0 "), + ParseErrCase(name="empty", input=""), + ParseErrCase(name="word", input="hello"), +] + + +@pytest.mark.parametrize("case", PARSE_REJECTS_CASES, ids=lambda case: case.name) +def test_parse_rejects(case: ParseErrCase) -> None: """parse() (strict) rejects what blang/semver Parse rejects.""" - - def test_parse_rejects(self) -> None: - cases = [ - ParseErrCase(name="v prefix", input="v1.0"), - ParseErrCase(name="major only", input="1"), - ParseErrCase(name="major minor only", input="1.1"), - ParseErrCase(name="leading zeros", input="01.01.01"), - ParseErrCase(name="leading whitespace", input=" 1.0.0"), - ParseErrCase(name="trailing whitespace", input="1.0.0 "), - ParseErrCase(name="empty", input=""), - ParseErrCase(name="word", input="hello"), - ] - for case in cases: - with self.subTest(case.name), self.assertRaises(ValueError): - semver.parse(case.input) + with pytest.raises( + ValueError, match=r"version string empty|no Major\.Minor\.Patch|invalid character|leading zeroes" + ): + semver.parse(case.input) diff --git a/functions/compose-model-endpoint/tests/__init__.py b/functions/compose-model-endpoint/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-model-endpoint/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-model-endpoint/tests/test_fn.py b/functions/compose-model-endpoint/tests/test_fn.py index a52a88996..a149608a9 100644 --- a/functions/compose-model-endpoint/tests/test_fn.py +++ b/functions/compose-model-endpoint/tests/test_fn.py @@ -14,15 +14,17 @@ """Tests for the compose-model-endpoint function.""" +import asyncio import base64 import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.modelendpoint import v1alpha1 @@ -100,147 +102,131 @@ def _response( return rsp -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - cases = [ - Case( - name="no credential: usable as soon as it exists", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - ), - want=_response( - reason=fn.CONDITION_REASON_ENDPOINT_USABLE, - status=fnv1.STATUS_CONDITION_TRUE, - ), +COMPOSE_CASES = [ + Case( + name="no credential: usable as soon as it exists", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + ), + want=_response( + reason=fn.CONDITION_REASON_ENDPOINT_USABLE, + status=fnv1.STATUS_CONDITION_TRUE, + ), + ), + Case( + name="a credential that resolves", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) ), - Case( - name="a credential that resolves", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key"))) + required_resources={ + "credential": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct(_secret("together-api-key", {"apiKey": "sk-abc"})) ) - ), - required_resources={ - "credential": fnv1.Resources( - items=[ - fnv1.Resource( - resource=resource.dict_to_struct(_secret("together-api-key", {"apiKey": "sk-abc"})) - ) - ] - ) - }, - ), - want=_response( - reason=fn.CONDITION_REASON_ENDPOINT_USABLE, - status=fnv1.STATUS_CONDITION_TRUE, - requirements=_credential_requirement("together-api-key"), - ), + ] + ) + }, + ), + want=_response( + reason=fn.CONDITION_REASON_ENDPOINT_USABLE, + status=fnv1.STATUS_CONDITION_TRUE, + requirements=_credential_requirement("together-api-key"), + ), + ), + Case( + name="a credential Secret that does not exist", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) ), - Case( - name="a credential Secret that does not exist", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key"))) - ) - ), - required_resources={"credential": fnv1.Resources(items=[])}, - ), - want=_response( - reason=fn.CONDITION_REASON_CREDENTIAL_MISSING, - status=fnv1.STATUS_CONDITION_FALSE, - message="Secret together-api-key does not exist", - requirements=_credential_requirement("together-api-key"), - ), + required_resources={"credential": fnv1.Resources(items=[])}, + ), + want=_response( + reason=fn.CONDITION_REASON_CREDENTIAL_MISSING, + status=fnv1.STATUS_CONDITION_FALSE, + message="Secret together-api-key does not exist", + requirements=_credential_requirement("together-api-key"), + ), + ), + Case( + # A Secret that exists but lacks the key is the likelier mistake, + # and would otherwise surface as a 401 from the provider. + name="a credential Secret missing the key", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) ), - Case( - # A Secret that exists but lacks the key is the likelier mistake, - # and would otherwise surface as a 401 from the provider. - name="a credential Secret missing the key", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key"))) - ) - ), - required_resources={ - "credential": fnv1.Resources( - items=[ - fnv1.Resource( - resource=resource.dict_to_struct(_secret("together-api-key", {"token": "sk-abc"})) - ) - ] + required_resources={ + "credential": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct(_secret("together-api-key", {"token": "sk-abc"})) ) - }, - ), - want=_response( - reason=fn.CONDITION_REASON_CREDENTIAL_MISSING, - status=fnv1.STATUS_CONDITION_FALSE, - message="Secret together-api-key has no key apiKey", - requirements=_credential_requirement("together-api-key"), - ), + ] + ) + }, + ), + want=_response( + reason=fn.CONDITION_REASON_CREDENTIAL_MISSING, + status=fnv1.STATUS_CONDITION_FALSE, + message="Secret together-api-key has no key apiKey", + requirements=_credential_requirement("together-api-key"), + ), + ), + Case( + name="a credential under a non-default key", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr(credential=_api_key("together-api-key", key="TOGETHER_API_KEY")) + ) + ) ), - Case( - name="a credential under a non-default key", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( + required_resources={ + "credential": fnv1.Resources( + items=[ + fnv1.Resource( resource=resource.dict_to_struct( - _xr(credential=_api_key("together-api-key", key="TOGETHER_API_KEY")) + _secret("together-api-key", {"TOGETHER_API_KEY": "sk-abc"}) ) ) - ), - required_resources={ - "credential": fnv1.Resources( - items=[ - fnv1.Resource( - resource=resource.dict_to_struct( - _secret("together-api-key", {"TOGETHER_API_KEY": "sk-abc"}) - ) - ) - ] - ) - }, - ), - want=_response( - reason=fn.CONDITION_REASON_ENDPOINT_USABLE, - status=fnv1.STATUS_CONDITION_TRUE, - requirements=_credential_requirement("together-api-key"), - ), - ), - Case( - name="an unresolved credential requirement", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key"))) - ) - ), - ), - want=_response( - reason=fn.CONDITION_REASON_WAITING_FOR_CREDENTIAL, - status=fnv1.STATUS_CONDITION_FALSE, - message="Waiting for Secret together-api-key to resolve", - requirements=_credential_requirement("together-api-key"), - ), - ), - ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", + ] ) + }, + ), + want=_response( + reason=fn.CONDITION_REASON_ENDPOINT_USABLE, + status=fnv1.STATUS_CONDITION_TRUE, + requirements=_credential_requirement("together-api-key"), + ), + ), + Case( + name="an unresolved credential requirement", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) + ), + ), + want=_response( + reason=fn.CONDITION_REASON_WAITING_FOR_CREDENTIAL, + status=fnv1.STATUS_CONDITION_FALSE, + message="Waiting for Secret together-api-key to resolve", + requirements=_credential_requirement("together-api-key"), + ), + ), +] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction reports whether the endpoint's credential makes it usable.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-model-replica/tests/__init__.py b/functions/compose-model-replica/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-model-replica/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-model-replica/tests/test_backends.py b/functions/compose-model-replica/tests/test_backends.py index 79728fa0b..c53c4a1a2 100644 --- a/functions/compose-model-replica/tests/test_backends.py +++ b/functions/compose-model-replica/tests/test_backends.py @@ -23,9 +23,9 @@ """ import dataclasses -import unittest -from typing import Any, ClassVar +from typing import Any +import pytest from crossplane.function import resource from function import routing from function.backends import base, grove, llmd, native @@ -385,7 +385,7 @@ class Case: stack: str = "Standard" -_CASES = [ +MANIFESTS_CASES = [ Case( name="native Standalone engine composes a Deployment", backend=native.NativeBackend(), @@ -402,1033 +402,1087 @@ class Case: ] -class TestBackendManifests(unittest.TestCase): - def test_manifests(self) -> None: - for case in _CASES: - with self.subTest(case.name): - replica = _replica(engines=[case.engine]) - out = case.backend.build(replica, case.engine, _PC, base.serving_label(replica), case.stack) - got = {key: obj.spec.forProvider.manifest for key, obj in out.items()} - self.assertEqual(case.want, got, "-want, +got") - - def test_leader_address_env_injected_but_not_rank(self) -> None: - # The Grove backend injects MODELPLANE_LEADER_ADDRESS (aliasing Grove's - # own GROVE_PCSG_* vars) but not MODELPLANE_RANK: Grove exposes no - # group-wide pod index yet (grove#755, open), so a gang engine's - # command computes its own rank from GROVE_PCLQ_POD_INDEX directly - # (see grove.py and the multinode example). - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - # Spelled out rather than compared against grove_leader_address_env(), - # which would pass whatever that function returned. The PCSG vars are - # what make the address vary per gang; the PCS-scoped ones are - # identical across gangs and would silently point every copy at gang - # 0's leader. - want = { - "name": "MODELPLANE_LEADER_ADDRESS", - "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", - } - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - self.assertEqual(container["env"], [want]) - - def test_user_env_passed_through(self) -> None: - # A member's own env passes through verbatim, after the leader-address - # alias (see test_leader_address_env_injected_but_not_rank). - engine = _gang_engine( - leader_command=_LEADER_CMD, - worker_command=_WORKER_CMD, - ) - spec = engine.members[0].template.spec - assert spec is not None - spec.containers[0].env = [v1alpha1.EnvItem(name="HF_TOKEN", value="x")] - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader")["spec"]["podSpec"] - env = leader["containers"][0]["env"] - self.assertEqual(env, [base.grove_leader_address_env(), {"name": "HF_TOKEN", "value": "x"}]) - - def test_fieldref_env_passes_through(self) -> None: - # A pod-field env (e.g. VLLM_HOST_IP from status.podIP, which multi-NIC - # RDMA nodes need so the engine binds the right interface — #141) survives - # model_dump into the composed manifest. - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - spec = engine.members[0].template.spec - assert spec is not None - spec.containers[0].env = [ - v1alpha1.EnvItem( - name="VLLM_HOST_IP", - valueFrom=v1alpha1.ValueFrom(fieldRef=v1alpha1.FieldRef(fieldPath="status.podIP")), - ) - ] - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader")["spec"]["podSpec"] - env = leader["containers"][0]["env"] - self.assertEqual( - env, - [ - base.grove_leader_address_env(), - {"name": "VLLM_HOST_IP", "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}}, - ], +@pytest.mark.parametrize("case", MANIFESTS_CASES, ids=lambda case: case.name) +def test_manifests(case: Case) -> None: + """A backend composes an engine's manifests.""" + replica = _replica(engines=[case.engine]) + out = case.backend.build(replica, case.engine, _PC, base.serving_label(replica), case.stack) + got = {key: obj.spec.forProvider.manifest for key, obj in out.items()} + assert got == case.want + + +def test_leader_address_env_injected_but_not_rank() -> None: + """The Grove backend injects a leader address alias, but no rank.""" + # The Grove backend injects MODELPLANE_LEADER_ADDRESS (aliasing Grove's + # own GROVE_PCSG_* vars) but not MODELPLANE_RANK: Grove exposes no + # group-wide pod index yet (grove#755, open), so a gang engine's + # command computes its own rank from GROVE_PCLQ_POD_INDEX directly + # (see grove.py and the multinode example). + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + replica = _replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + # Spelled out rather than compared against grove_leader_address_env(), + # which would pass whatever that function returned. The PCSG vars are + # what make the address vary per gang; the PCS-scoped ones are + # identical across gangs and would silently point every copy at gang + # 0's leader. + want = { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + for clique_name in ("leader", "worker"): + container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] + assert container["env"] == [want] + + +def test_user_env_passed_through() -> None: + """A Grove member's own env follows the leader address alias.""" + # A member's own env passes through verbatim, after the leader-address + # alias (see test_leader_address_env_injected_but_not_rank). + engine = _gang_engine( + leader_command=_LEADER_CMD, + worker_command=_WORKER_CMD, + ) + spec = engine.members[0].template.spec + assert spec is not None + spec.containers[0].env = [v1alpha1.EnvItem(name="HF_TOKEN", value="x")] + replica = _replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + leader = _clique(manifest, "leader")["spec"]["podSpec"] + env = leader["containers"][0]["env"] + assert env == [base.grove_leader_address_env(), {"name": "HF_TOKEN", "value": "x"}] + + +def test_fieldref_env_passes_through() -> None: + """A Grove member's pod-field env survives into the composed manifest.""" + # A pod-field env (e.g. VLLM_HOST_IP from status.podIP, which multi-NIC + # RDMA nodes need so the engine binds the right interface — #141) survives + # model_dump into the composed manifest. + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + spec = engine.members[0].template.spec + assert spec is not None + spec.containers[0].env = [ + v1alpha1.EnvItem( + name="VLLM_HOST_IP", + valueFrom=v1alpha1.ValueFrom(fieldRef=v1alpha1.FieldRef(fieldPath="status.podIP")), ) + ] + replica = _replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + leader = _clique(manifest, "leader")["spec"]["podSpec"] + env = leader["containers"][0]["env"] + assert env == [ + base.grove_leader_address_env(), + {"name": "VLLM_HOST_IP", "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}}, + ] - def test_member_metadata_propagates_to_native_pod_template(self) -> None: - # A Standalone member's template.metadata labels and annotations land - # on the Deployment's pod template, merged with the managed labels - # (#378). - engine = _standalone_engine() - engine.members[0].template.metadata = v1alpha1.Metadata( - labels={"example.com/role": "standalone"}, - annotations={"example.com/config": "standalone"}, - ) - replica = _replica(engines=[engine]) - out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - meta = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["metadata"] - self.assertEqual( - meta["labels"], - { - "example.com/role": "standalone", - _ENGINE: "main", - _ROLE: "Standalone", - _SERVING: "r", - _WORKLOAD: _WORKLOAD_NAME, - }, - ) - self.assertEqual(meta["annotations"], {"example.com/config": "standalone"}) - - def test_member_metadata_propagates_to_cliques_independently(self) -> None: - # Leader metadata lands on the leader clique and worker metadata on the - # worker clique; neither leaks into the other. Grove propagates a - # clique's labels and annotations to its pods (#378). - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - engine.members[0].template.metadata = v1alpha1.Metadata( - labels={"example.com/role": "leader"}, annotations={"example.com/config": "leader"} - ) - engine.members[1].template.metadata = v1alpha1.Metadata( - labels={"example.com/role": "worker"}, annotations={"example.com/config": "worker"} - ) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader") - self.assertEqual( - leader["labels"], - { - "example.com/role": "leader", - _ENGINE: "main", - _ROLE: "Leader", - _SERVING: "r", - _QUEUE_LABEL: _QUEUE, - _CLIQUE_ROLE: "leader", - }, - ) - self.assertEqual(leader["annotations"], {"example.com/config": "leader"}) - worker = _clique(manifest, "worker") - self.assertEqual( - worker["labels"], {"example.com/role": "worker", _ENGINE: "main", _ROLE: "Worker", _QUEUE_LABEL: _QUEUE} - ) - self.assertEqual(worker["annotations"], {"example.com/config": "worker"}) - - def test_worker_without_metadata_composes_only_managed_labels(self) -> None: - # A worker member with no template.metadata composes a worker clique - # carrying only the managed queue label and no annotations key. - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - worker = _clique(manifest, "worker") - self.assertEqual(worker["labels"], {_ENGINE: "main", _ROLE: "Worker", _QUEUE_LABEL: _QUEUE}) - self.assertNotIn("annotations", worker) - - @staticmethod - def _names(out: dict[str, k8sobjv1alpha1.Object]) -> set[str]: - return {o.spec.forProvider.manifest["metadata"]["name"] for o in out.values()} - - def test_co_located_replicas_get_distinct_names(self) -> None: - # Two replicas of one deployment on the same cluster must produce - # distinct resource names on the remote cluster. - a = _replica("dep-clusterA") - b = _replica("dep-clusterB") - out_a = native.NativeBackend().build(a, a.spec.engines[0], _PC, base.serving_label(a), "Standard") - out_b = native.NativeBackend().build(b, b.spec.engines[0], _PC, base.serving_label(b), "Standard") - self.assertEqual(self._names(out_a) & self._names(out_b), set()) - - def test_multi_engine_qualifies_workload_names(self) -> None: - # A replica with two engines names each engine's workload distinctly so - # they don't collide on the remote cluster. - engines = [_standalone_engine("prefill"), _standalone_engine("decode")] - replica = _replica(engines=engines) - names = set() - for g in engines: - out = native.NativeBackend().build(replica, g, _PC, base.serving_label(replica), "Standard") - names |= self._names(out) - self.assertEqual(len(names), 4) # 2 deployments + 2 claim templates - - def test_workload_readiness_policies(self) -> None: - # A Deployment reports readiness from its Available condition; a - # PodCliqueSet publishes no such condition, so it's derived from its - # replica counters instead (base.GROVE_AVAILABLE_CEL). Either way the - # claim templates are ready on create. - for name, backend, engine, stack, want_cel in ( - ("native", native.NativeBackend(), _standalone_engine(), "Standard", base.AVAILABLE_CEL), - ( - "grove", - grove.GroveBackend(), - _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD), - "Dynamo", - base.GROVE_AVAILABLE_CEL, - ), - ): - with self.subTest(name): - replica = _replica(engines=[engine]) - out = backend.build(replica, engine, _PC, base.serving_label(replica), stack) - serving = out["model-serving-main"].spec.readiness - assert serving is not None - self.assertEqual(serving.policy, "DeriveFromCelQuery") - self.assertEqual(serving.celQuery, want_cel) - for key, obj in out.items(): - if key.startswith("resource-claim"): - readiness = obj.spec.readiness - assert readiness is not None - self.assertEqual(readiness.policy, "SuccessfulCreate") - - def test_multiple_device_requests_single_container_claim(self) -> None: - # resources.claims is a list-map keyed on name alone, so N device - # requests must NOT produce N container claims all named "devices". The - # container references the whole pod claim once; the template carries all - # requests. - engine = _standalone_engine( - device_requests=[ - v1alpha1.DeviceRequest(name="gpu", deviceClassName="gpu.nvidia.com", count=8), - v1alpha1.DeviceRequest(name="nic", deviceClassName="nic.nvidia.com", count=8), - ], - ) - replica = _replica(engines=[engine]) - out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - pod = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"] - claims = pod["containers"][0]["resources"]["claims"] - self.assertEqual(claims, [{"name": "devices"}]) - self.assertEqual(pod["resourceClaims"][0]["name"], "devices") - template = out["resource-claim-main-standalone"].spec.forProvider.manifest - template_requests = template["spec"]["spec"]["devices"]["requests"] - self.assertEqual([r["name"] for r in template_requests], ["gpu", "nic"]) - claim_readiness = out["resource-claim-main-standalone"].spec.readiness - assert claim_readiness is not None - self.assertEqual(claim_readiness.policy, "SuccessfulCreate") - - def test_claimless_leader_gets_no_claim(self) -> None: - # A coordinator-only leader (e.g. a vLLM DP head running - # --data-parallel-size-local=0) carries no deviceRequests. Its pod must - # get no resourceClaims, its container no resources.claims, and no - # leader ResourceClaimTemplate must be composed - only the worker's. - # It still pins to its pool and tolerates the GPU taint. - engine = _gang_engine( - leader_command=_LEADER_CMD, - worker_command=_WORKER_CMD, - leader_device_requests=[], - ) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - - self.assertNotIn("resource-claim-main-leader", out) - self.assertIn("resource-claim-main-worker", out) - - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader")["spec"]["podSpec"] - self.assertNotIn("resourceClaims", leader) - self.assertNotIn("resources", leader["containers"][0]) - self.assertEqual(leader["nodeSelector"], {"modelplane.ai/pool": "frontier"}) - self.assertEqual( - leader["tolerations"], [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}] - ) - worker = _clique(manifest, "worker")["spec"]["podSpec"] - self.assertEqual(worker["resourceClaims"], _claims("worker")) - self.assertEqual(worker["containers"][0]["resources"], {"claims": [{"name": "devices"}]}) - - def test_members_pin_to_their_own_pools(self) -> None: - # The scheduler may split a gang across pools when no single pool - # satisfies every member. Each member's pods must pin to that member's - # pool, not a shared engine-wide one. - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD, leader_pool="head") - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - self.assertEqual(_clique(manifest, "leader")["spec"]["podSpec"]["nodeSelector"], {"modelplane.ai/pool": "head"}) - self.assertEqual( - _clique(manifest, "worker")["spec"]["podSpec"]["nodeSelector"], {"modelplane.ai/pool": "frontier"} - ) +def test_member_metadata_propagates_to_native_pod_template() -> None: + """A Standalone member's template metadata lands on the Deployment's pod template.""" + # A Standalone member's template.metadata labels and annotations land + # on the Deployment's pod template, merged with the managed labels + # (#378). + engine = _standalone_engine() + engine.members[0].template.metadata = v1alpha1.Metadata( + labels={"example.com/role": "standalone"}, + annotations={"example.com/config": "standalone"}, + ) + replica = _replica(engines=[engine]) + out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") + meta = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["metadata"] + assert meta["labels"] == { + "example.com/role": "standalone", + _ENGINE: "main", + _ROLE: "Standalone", + _SERVING: "r", + _WORKLOAD: _WORKLOAD_NAME, + } + assert meta["annotations"] == {"example.com/config": "standalone"} -class TestLLMDBackend(unittest.TestCase): - """The LeaderWorkerSet backend for a Leader/Worker gang engine.""" - - _LWS_ROLE = "modelplane.ai/lws-role" - - @staticmethod - def _lws(engine: v1alpha1.Engine, replica: v1alpha1.ModelReplica) -> dict: - out = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - return out["model-serving-main"].spec.forProvider.manifest - - def test_leader_worker_set_shape(self) -> None: - engine = _gang_engine(nodes=3, copies=2) - replica = _replica(engines=[engine]) - manifest = self._lws(engine, replica) - self.assertEqual(manifest["apiVersion"], "leaderworkerset.x-k8s.io/v1") - self.assertEqual(manifest["kind"], "LeaderWorkerSet") - self.assertEqual(manifest["metadata"], {"name": _WORKLOAD_NAME, "namespace": "mp-ml-team-51733"}) - self.assertEqual(manifest["spec"]["replicas"], 2) - # Gang size is the leader plus the worker's node count. - self.assertEqual(manifest["spec"]["leaderWorkerTemplate"]["size"], 4) - - def test_only_leader_carries_serving_label(self) -> None: - engine = _gang_engine() - replica = _replica(engines=[engine]) - lwt = self._lws(engine, replica)["spec"]["leaderWorkerTemplate"] - leader_labels = lwt["leaderTemplate"]["metadata"]["labels"] - self.assertEqual(leader_labels[_SERVING], "r") - self.assertEqual(leader_labels[self._LWS_ROLE], "leader") - # The worker followers never serve, so they carry no serving label and - # the replica's Service can't route to them. They do carry the - # telemetry identity: a worker holds GPUs, and its metrics are the - # deployment's. - worker_labels = lwt["workerTemplate"]["metadata"]["labels"] - self.assertNotIn(_SERVING, worker_labels) - self.assertEqual(worker_labels, {_ENGINE: "main", _ROLE: "Worker"}) - - def test_leader_address_and_rank_env_injected(self) -> None: - # Every gang container leads with the backend-neutral coordination vars - # aliasing LWS_LEADER_ADDRESS / LWS_WORKER_INDEX. - engine = _gang_engine() - replica = _replica(engines=[engine]) - lwt = self._lws(engine, replica)["spec"]["leaderWorkerTemplate"] - for tmpl in (lwt["leaderTemplate"], lwt["workerTemplate"]): - env = tmpl["spec"]["containers"][0]["env"] - self.assertEqual(env[0], {"name": "MODELPLANE_LEADER_ADDRESS", "value": "$(LWS_LEADER_ADDRESS)"}) - self.assertEqual(env[1], {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}) - - def test_no_modelexpress_env_even_with_a_cache(self) -> None: - # The llm-d (Standard) backend never injects ModelExpress env: that P2P - # wiring is the Grove (Dynamo) backend's, gated on the cluster stack. - engine = _gang_engine() - replica = v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="c"), - engines=[engine], - ), - ) - lwt = self._lws(engine, replica)["spec"]["leaderWorkerTemplate"] - for tmpl in (lwt["leaderTemplate"], lwt["workerTemplate"]): - container = tmpl["spec"]["containers"][0] - env_names = [e["name"] for e in container["env"]] - # HF_HUB_CACHE is the cache's own env (every stack); the MX bundle - # is not. - self.assertEqual(env_names, ["MODELPLANE_LEADER_ADDRESS", "MODELPLANE_RANK", "HF_HUB_CACHE"]) - self.assertNotIn("MX_SERVER_ADDRESS", env_names) - self.assertNotIn("securityContext", container) - - def test_workload_readiness_uses_available_cel(self) -> None: - engine = _gang_engine() - replica = _replica(engines=[engine]) - out = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - readiness = out["model-serving-main"].spec.readiness - assert readiness is not None - self.assertEqual(readiness.policy, "DeriveFromCelQuery") - self.assertEqual(readiness.celQuery, base.AVAILABLE_CEL) - - -class TestBackendSelection(unittest.TestCase): - def test_standalone_engine_is_native(self) -> None: - # A Standalone engine is native regardless of the cluster's stack. - self.assertEqual(base.select_backend(_standalone_engine(), "Standard"), base.NATIVE) - self.assertEqual(base.select_backend(_standalone_engine(), "Dynamo"), base.NATIVE) - - def test_leader_worker_engine_is_llmd(self) -> None: - self.assertEqual(base.select_backend(_gang_engine(), "Standard"), base.LLMD) - - def test_leader_worker_engine_is_grove(self) -> None: - self.assertEqual(base.select_backend(_gang_engine(), "Dynamo"), base.GROVE) - - -class TestCacheMounts(unittest.TestCase): - def _replica( - self, *, cache: str | None = None, args: list[str] | None = None, command: list[str] | None = None - ) -> v1alpha1.ModelReplica: - engine = _standalone_engine(args=args or [], command=command) - modelcache = v1alpha1.ModelCacheRef(name=cache) if cache else None - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(namespace="ml-team"), - spec=v1alpha1.SpecModel(clusterName="c", modelCacheRef=modelcache, engines=[engine]), - ) +def test_member_metadata_propagates_to_cliques_independently() -> None: + """Each Grove member's template metadata lands on its own clique only.""" + # Leader metadata lands on the leader clique and worker metadata on the + # worker clique; neither leaks into the other. Grove propagates a + # clique's labels and annotations to its pods (#378). + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + engine.members[0].template.metadata = v1alpha1.Metadata( + labels={"example.com/role": "leader"}, annotations={"example.com/config": "leader"} + ) + engine.members[1].template.metadata = v1alpha1.Metadata( + labels={"example.com/role": "worker"}, annotations={"example.com/config": "worker"} + ) + replica = _replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + leader = _clique(manifest, "leader") + assert leader["labels"] == { + "example.com/role": "leader", + _ENGINE: "main", + _ROLE: "Leader", + _SERVING: "r", + _QUEUE_LABEL: _QUEUE, + _CLIQUE_ROLE: "leader", + } + assert leader["annotations"] == {"example.com/config": "leader"} + worker = _clique(manifest, "worker") + assert worker["labels"] == {"example.com/role": "worker", _ENGINE: "main", _ROLE: "Worker", _QUEUE_LABEL: _QUEUE} + assert worker["annotations"] == {"example.com/config": "worker"} + + +def test_worker_without_metadata_composes_only_managed_labels() -> None: + """A Grove worker with no template metadata carries only the managed labels.""" + # A worker member with no template.metadata composes a worker clique + # carrying only the managed queue label and no annotations key. + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + replica = _replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + worker = _clique(manifest, "worker") + assert worker["labels"] == {_ENGINE: "main", _ROLE: "Worker", _QUEUE_LABEL: _QUEUE} + assert "annotations" not in worker + + +def _names(out: dict[str, k8sobjv1alpha1.Object]) -> set[str]: + """The names of the manifests a backend composed.""" + return {o.spec.forProvider.manifest["metadata"]["name"] for o in out.values()} + + +def test_co_located_replicas_get_distinct_names() -> None: + """Two replicas on one cluster compose distinct resource names.""" + # Two replicas of one deployment on the same cluster must produce + # distinct resource names on the remote cluster. + a = _replica("dep-clusterA") + b = _replica("dep-clusterB") + out_a = native.NativeBackend().build(a, a.spec.engines[0], _PC, base.serving_label(a), "Standard") + out_b = native.NativeBackend().build(b, b.spec.engines[0], _PC, base.serving_label(b), "Standard") + assert _names(out_a) & _names(out_b) == set() + + +def test_multi_engine_qualifies_workload_names() -> None: + """Each engine of a multi-engine replica composes distinctly named resources.""" + # A replica with two engines names each engine's workload distinctly so + # they don't collide on the remote cluster. + engines = [_standalone_engine("prefill"), _standalone_engine("decode")] + replica = _replica(engines=engines) + names = set() + for g in engines: + out = native.NativeBackend().build(replica, g, _PC, base.serving_label(replica), "Standard") + names |= _names(out) + assert len(names) == 4 # 2 deployments + 2 claim templates + + +@pytest.mark.parametrize( + ("backend", "engine", "stack", "want_cel"), + [ + pytest.param(native.NativeBackend(), _standalone_engine(), "Standard", base.AVAILABLE_CEL, id="native"), + pytest.param( + grove.GroveBackend(), + _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD), + "Dynamo", + base.GROVE_AVAILABLE_CEL, + id="grove", + ), + ], +) +def test_workload_readiness_policies(backend: base.Backend, engine: v1alpha1.Engine, stack: str, want_cel: str) -> None: + """A workload's readiness derives from its status, and a claim template's from its creation.""" + # A Deployment reports readiness from its Available condition; a + # PodCliqueSet publishes no such condition, so it's derived from its + # replica counters instead (base.GROVE_AVAILABLE_CEL). Either way the + # claim templates are ready on create. + replica = _replica(engines=[engine]) + out = backend.build(replica, engine, _PC, base.serving_label(replica), stack) + serving = out["model-serving-main"].spec.readiness + assert serving is not None + assert serving.policy == "DeriveFromCelQuery" + assert serving.celQuery == want_cel + for key, obj in out.items(): + if key.startswith("resource-claim"): + readiness = obj.spec.readiness + assert readiness is not None + assert readiness.policy == "SuccessfulCreate" + + +def test_multiple_device_requests_single_container_claim() -> None: + """Several device requests compose one container claim and one template carrying them all.""" + # resources.claims is a list-map keyed on name alone, so N device + # requests must NOT produce N container claims all named "devices". The + # container references the whole pod claim once; the template carries all + # requests. + engine = _standalone_engine( + device_requests=[ + v1alpha1.DeviceRequest(name="gpu", deviceClassName="gpu.nvidia.com", count=8), + v1alpha1.DeviceRequest(name="nic", deviceClassName="nic.nvidia.com", count=8), + ], + ) + replica = _replica(engines=[engine]) + out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") + pod = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"] + claims = pod["containers"][0]["resources"]["claims"] + assert claims == [{"name": "devices"}] + assert pod["resourceClaims"][0]["name"] == "devices" + template = out["resource-claim-main-standalone"].spec.forProvider.manifest + template_requests = template["spec"]["spec"]["devices"]["requests"] + assert [r["name"] for r in template_requests] == ["gpu", "nic"] + claim_readiness = out["resource-claim-main-standalone"].spec.readiness + assert claim_readiness is not None + assert claim_readiness.policy == "SuccessfulCreate" + + +def test_claimless_leader_gets_no_claim() -> None: + """A Grove leader with no device requests composes no claim, but still pins and tolerates.""" + # A coordinator-only leader (e.g. a vLLM DP head running + # --data-parallel-size-local=0) carries no deviceRequests. Its pod must + # get no resourceClaims, its container no resources.claims, and no + # leader ResourceClaimTemplate must be composed - only the worker's. + # It still pins to its pool and tolerates the GPU taint. + engine = _gang_engine( + leader_command=_LEADER_CMD, + worker_command=_WORKER_CMD, + leader_device_requests=[], + ) + replica = _replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + + assert "resource-claim-main-leader" not in out + assert "resource-claim-main-worker" in out + + manifest = out["model-serving-main"].spec.forProvider.manifest + leader = _clique(manifest, "leader")["spec"]["podSpec"] + assert "resourceClaims" not in leader + assert "resources" not in leader["containers"][0] + assert leader["nodeSelector"] == {"modelplane.ai/pool": "frontier"} + assert leader["tolerations"] == [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}] + + worker = _clique(manifest, "worker")["spec"]["podSpec"] + assert worker["resourceClaims"] == _claims("worker") + assert worker["containers"][0]["resources"] == {"claims": [{"name": "devices"}]} + + +def test_members_pin_to_their_own_pools() -> None: + """Each Grove member's pods pin to that member's own pool.""" + # The scheduler may split a gang across pools when no single pool + # satisfies every member. Each member's pods must pin to that member's + # pool, not a shared engine-wide one. + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD, leader_pool="head") + replica = _replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + assert _clique(manifest, "leader")["spec"]["podSpec"]["nodeSelector"] == {"modelplane.ai/pool": "head"} + assert _clique(manifest, "worker")["spec"]["podSpec"]["nodeSelector"] == {"modelplane.ai/pool": "frontier"} + + +# The LeaderWorkerSet backend for a Leader/Worker gang engine. + +_LWS_ROLE = "modelplane.ai/lws-role" + + +def _llmd_lws(engine: v1alpha1.Engine, replica: v1alpha1.ModelReplica) -> dict: + """The LeaderWorkerSet manifest the llm-d backend composes for engine.""" + out = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") + return out["model-serving-main"].spec.forProvider.manifest + + +def test_llmd_leader_worker_set_shape() -> None: + """The llm-d backend composes a LeaderWorkerSet of copies gangs, each the leader plus its workers.""" + engine = _gang_engine(nodes=3, copies=2) + replica = _replica(engines=[engine]) + manifest = _llmd_lws(engine, replica) + assert manifest["apiVersion"] == "leaderworkerset.x-k8s.io/v1" + assert manifest["kind"] == "LeaderWorkerSet" + assert manifest["metadata"] == {"name": _WORKLOAD_NAME, "namespace": "mp-ml-team-51733"} + assert manifest["spec"]["replicas"] == 2 + # Gang size is the leader plus the worker's node count. + assert manifest["spec"]["leaderWorkerTemplate"]["size"] == 4 + + +def test_llmd_only_leader_carries_serving_label() -> None: + """Only the LeaderWorkerSet's leader carries the serving label.""" + engine = _gang_engine() + replica = _replica(engines=[engine]) + lwt = _llmd_lws(engine, replica)["spec"]["leaderWorkerTemplate"] + leader_labels = lwt["leaderTemplate"]["metadata"]["labels"] + assert leader_labels[_SERVING] == "r" + assert leader_labels[_LWS_ROLE] == "leader" + # The worker followers never serve, so they carry no serving label and + # the replica's Service can't route to them. They do carry the + # telemetry identity: a worker holds GPUs, and its metrics are the + # deployment's. + worker_labels = lwt["workerTemplate"]["metadata"]["labels"] + assert _SERVING not in worker_labels + assert worker_labels == {_ENGINE: "main", _ROLE: "Worker"} + + +def test_llmd_leader_address_and_rank_env_injected() -> None: + """Every LeaderWorkerSet container leads with the leader address and rank aliases.""" + # Every gang container leads with the backend-neutral coordination vars + # aliasing LWS_LEADER_ADDRESS / LWS_WORKER_INDEX. + engine = _gang_engine() + replica = _replica(engines=[engine]) + lwt = _llmd_lws(engine, replica)["spec"]["leaderWorkerTemplate"] + for tmpl in (lwt["leaderTemplate"], lwt["workerTemplate"]): + env = tmpl["spec"]["containers"][0]["env"] + assert env[0] == {"name": "MODELPLANE_LEADER_ADDRESS", "value": "$(LWS_LEADER_ADDRESS)"} + assert env[1] == {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"} + + +def test_llmd_no_modelexpress_env_even_with_a_cache() -> None: + """The llm-d backend injects no ModelExpress env, even for a replica with a cache.""" + # The llm-d (Standard) backend never injects ModelExpress env: that P2P + # wiring is the Grove (Dynamo) backend's, gated on the cluster stack. + engine = _gang_engine() + replica = v1alpha1.ModelReplica( + metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), + spec=v1alpha1.SpecModel( + clusterName="cluster-a", + modelCacheRef=v1alpha1.ModelCacheRef(name="c"), + engines=[engine], + ), + ) + lwt = _llmd_lws(engine, replica)["spec"]["leaderWorkerTemplate"] + for tmpl in (lwt["leaderTemplate"], lwt["workerTemplate"]): + container = tmpl["spec"]["containers"][0] + env_names = [e["name"] for e in container["env"]] + # HF_HUB_CACHE is the cache's own env (every stack); the MX bundle + # is not. + assert env_names == ["MODELPLANE_LEADER_ADDRESS", "MODELPLANE_RANK", "HF_HUB_CACHE"] + assert "MX_SERVER_ADDRESS" not in env_names + assert "securityContext" not in container - @staticmethod - def _engine(replica: v1alpha1.ModelReplica) -> v1alpha1.Container: - spec = replica.spec.engines[0].members[0].template.spec - assert spec is not None - return spec.containers[0] - - def test_no_cache_no_mounts(self) -> None: - volumes, mounts = base.cache_mounts(self._replica()) - self.assertEqual((volumes, mounts), ([], [])) - - def test_cache_adds_volume_and_mount(self) -> None: - volumes, mounts = base.cache_mounts(self._replica(cache="qwen")) - self.assertEqual( - volumes, - [{"name": "model-cache", "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}}], - ) - self.assertEqual(mounts, [{"name": "model-cache", "mountPath": "/mnt/models"}]) - - def test_cache_env_points_huggingface_at_the_mount(self) -> None: - # The cache is staged in HuggingFace's cache layout, so pointing - # HF_HUB_CACHE at the mount is what lets an engine's own --model= - # resolve against it instead of pulling from HuggingFace (#407). - self.assertEqual( - base.cache_env(self._replica(cache="qwen")), - [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}], - ) - def test_cache_env_empty_without_cache(self) -> None: - self.assertEqual(base.cache_env(self._replica()), []) - - def test_cache_env_sets_no_offline_flag(self) -> None: - # HF_HUB_OFFLINE would break an engine that fetches a *different* repo - # at startup (kimi-k2's separately-gated tokenizer), and resolution - # doesn't need it. - names = {e["name"] for e in base.cache_env(self._replica(cache="qwen"))} - self.assertNotIn("HF_HUB_OFFLINE", names) - - -class TestNativeBackendCache(unittest.TestCase): - def _replica(self) -> v1alpha1.ModelReplica: - engine = _standalone_engine(args=[]) - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), - engines=[engine], - ), - ) +def test_llmd_workload_readiness_uses_available_cel() -> None: + """The LeaderWorkerSet's readiness derives from its Available condition.""" + engine = _gang_engine() + replica = _replica(engines=[engine]) + out = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") + readiness = out["model-serving-main"].spec.readiness + assert readiness is not None + assert readiness.policy == "DeriveFromCelQuery" + assert readiness.celQuery == base.AVAILABLE_CEL - def test_mounts_pvc_and_sets_cache_env(self) -> None: - # A cache contributes a volume, a mount, and the HF_HUB_CACHE that makes - # the engine's own --model= resolve against it. Modelplane injects - # no --model of its own: naming the model is the command's job. - replica = self._replica() - out = native.NativeBackend().build( - replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Standard" - ) - dep = out["model-serving-main"].spec.forProvider.manifest - pod = dep["spec"]["template"]["spec"] - vol_names = {v["name"] for v in pod["volumes"]} - self.assertIn("model-cache", vol_names) - container = pod["containers"][0] - self.assertIn({"name": "model-cache", "mountPath": "/mnt/models"}, container["volumeMounts"]) - self.assertIn({"name": "HF_HUB_CACHE", "value": "/mnt/models"}, container["env"]) - self.assertEqual(container["args"], []) - - def test_user_env_comes_after_cache_env(self) -> None: - # Kubernetes expands $(VAR) left to right, so Modelplane's own entries - # must precede the user's for a user entry to reference them. - replica = self._replica() - engine = replica.spec.engines[0] - spec = engine.members[0].template.spec - assert spec is not None - spec.containers[0].env = [v1alpha1.EnvItem(name="HF_TOKEN", value="x")] - out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] - self.assertEqual( - container["env"], - [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}, {"name": "HF_TOKEN", "value": "x"}], - ) +def test_select_backend_standalone_engine_is_native() -> None: + """A Standalone engine selects the native backend.""" + # A Standalone engine is native regardless of the cluster's stack. + assert base.select_backend(_standalone_engine(), "Standard") == base.NATIVE + assert base.select_backend(_standalone_engine(), "Dynamo") == base.NATIVE -class TestGroveBackendCache(unittest.TestCase): - def _replica( - self, - *, - leader_command: list[str] | None = None, - worker_command: list[str] | None = None, - leader_args: list[str] | None = None, - worker_args: list[str] | None = None, - ) -> v1alpha1.ModelReplica: - engine = _gang_engine( - leader_command=leader_command, - worker_command=worker_command, - leader_args=leader_args, - worker_args=worker_args, - ) - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="kimi"), - engines=[engine], - ), - ) - def test_both_grove_cliques_mount_cache(self) -> None: - replica = self._replica(leader_args=[], worker_command=["/bin/sh", "-c", "join"]) - manifest = ( - grove.GroveBackend() - .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] - .spec.forProvider.manifest - ) - for clique_name in ("leader", "worker"): - pod = _clique(manifest, clique_name)["spec"]["podSpec"] - self.assertIn("model-cache", {v["name"] for v in pod["volumes"]}) - self.assertIn( - {"name": "model-cache", "mountPath": "/mnt/models"}, - pod["containers"][0]["volumeMounts"], - ) - - def test_sets_cache_env_on_every_clique_and_injects_no_model(self) -> None: - # A cache gives both cliques HF_HUB_CACHE so their own --model= - # resolves against the mount; Modelplane adds no --model itself. - replica = self._replica(leader_args=[], worker_command=["/bin/sh", "-c", "join"]) - manifest = ( - grove.GroveBackend() - .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] - .spec.forProvider.manifest - ) - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - self.assertIn({"name": "HF_HUB_CACHE", "value": "/mnt/models"}, container["env"]) - self.assertNotIn("--model=/mnt/models", container.get("args", [])) - - def test_command_engine_mounts_cache_without_injecting_model(self) -> None: - # A member with its own command keeps it verbatim and gets no injected - # --model (it points at the cache with its own flag). - leader_cmd = [ - "/bin/sh", - "-c", - "python3 -m sglang.launch_server --model-path /mnt/models --tp 16", - ] - replica = self._replica(leader_command=leader_cmd, worker_command=["/bin/sh", "-c", "join"]) - manifest = ( - grove.GroveBackend() - .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] - .spec.forProvider.manifest - ) - leader = _clique(manifest, "leader")["spec"]["podSpec"]["containers"][0] - self.assertIn( - {"name": "model-cache", "mountPath": "/mnt/models"}, - leader["volumeMounts"], - ) - self.assertEqual(leader["command"], leader_cmd) - - -class TestDisaggregated(unittest.TestCase): - """serving.mode: PrefillDecode routing layers an InferencePool + endpoint - picker over two engines, role-labels them, and sidecars decode — no unified - Service. Mirrors how fn.py composes engines then calls routing.apply.""" - - def _apply(self) -> dict[str, k8sobjv1alpha1.Object]: - prefill = _standalone_engine(name="prefill") - prefill.phase = "Prefill" - decode = _standalone_engine(name="decode") - decode.phase = "Decode" - replica = _replica(engines=[prefill, decode]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = {} - for engine in replica.spec.engines: - composed.update(native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard")) - return routing.apply(composed, replica, _PC) - - def _serving_pod(self, out: dict[str, k8sobjv1alpha1.Object], engine_name: str) -> dict: - return out[f"model-serving-{engine_name}"].spec.forProvider.manifest["spec"]["template"] - - def test_replaces_unified_service_with_pool_and_epp(self) -> None: - out = self._apply() - self.assertIn("inference-pool", out) - self.assertIn("epp", out) - self.assertIn("epp-config", out) - pool = out["inference-pool"].spec.forProvider.manifest - self.assertEqual(pool["kind"], "InferencePool") - self.assertEqual(pool["spec"]["endpointPickerRef"]["name"], "r-epp") - - def test_the_picker_is_scrapeable_and_attributed(self) -> None: - """A built-in MetricMapping renames the picker's scheduling latency. - - Nothing can match it unless something scrapes the picker, and the - collector's engine job keeps a pod on two things: the deployment - label, and a container port named `http`. Without both, the mapping - is config that matches nothing for the life of the fleet. - - Its metrics endpoint authenticates callers by TokenReview by default, - which needs a ClusterRole the picker's namespaced ServiceAccount - cannot hold, so every scrape would be rejected. --secure-serving is - left alone: that one is the ext-proc gRPC server Envoy calls. - """ - replica = _replica() - replica.metadata = metav1.ObjectMeta( - name="r", - namespace="ml-team", - labels={base.LABEL_DEPLOYMENT: "qwen3-8b", "modelplane.ai/replica-index": "2"}, - ) - replica.spec.serving = v1alpha1.Serving(mode="Unified") - composed = {} - for engine in replica.spec.engines: - composed.update(native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard")) - template = routing.apply(composed, replica, _PC)["epp"].spec.forProvider.manifest["spec"]["template"] - - labels = template["metadata"]["labels"] - self.assertEqual(labels[base.LABEL_DEPLOYMENT], "qwen3-8b") - self.assertEqual(labels[base.LABEL_REPLICA], "2") - self.assertEqual(labels[base.LABEL_ROLE], "picker") - # Still the Deployment's own selector label, which must not move. - self.assertEqual(labels["app"], "r-epp") - - container = next(c for c in template["spec"]["containers"] if c["name"] == "epp") - self.assertIn({"name": base.ENGINE_PORT_NAME, "containerPort": 9090}, container["ports"]) - self.assertIn("--metrics-port=9090", container["args"]) - self.assertIn("--metrics-endpoint-auth=false", container["args"]) - self.assertNotIn("--secure-serving=false", container["args"]) - - def test_injects_nixl_plumbing(self) -> None: - """Both disagg engines get the NIXL plumbing the schema can't express: - a Memory /dev/shm and VLLM_NIXL_SIDE_CHANNEL_HOST = pod IP.""" - out = self._apply() - for role in ("prefill", "decode"): - pod = self._serving_pod(out, role)["spec"] - self.assertTrue( - any(v.get("emptyDir", {}).get("medium") == "Memory" for v in pod["volumes"]), - f"{role} missing Memory /dev/shm volume", - ) - engine = next(c for c in pod["containers"] if c["name"] == "engine") - self.assertIn("/dev/shm", [m["mountPath"] for m in engine["volumeMounts"]]) - host = next((e for e in engine["env"] if e["name"] == "VLLM_NIXL_SIDE_CHANNEL_HOST"), None) - assert host is not None, f"{role} missing VLLM_NIXL_SIDE_CHANNEL_HOST" - self.assertEqual(host["valueFrom"]["fieldRef"]["fieldPath"], "status.podIP") - self.assertIn("VLLM_NIXL_SIDE_CHANNEL_PORT", [e["name"] for e in engine["env"]]) - - def test_epp_config_arms_the_pd_decider(self) -> None: - """PrefillDecode silently serves decode-only unless the PD decider is armed. - - Selective prefix-based-pd-decider needs all of: nonCachedTokens > 0 (0 = - disabled), the approx-prefix-cache-producer plugin that populates the - attribute it reads, and that producer pinned to autoTune: false (the - true default never populates). And it must NOT carry the prepareDataPlugins - feature gate, which the v0.8.0 EPP image rejects and crashloops on. - """ - cfg = self._apply()["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - self.assertIn("prefix-based-pd-decider", cfg) - self.assertIn("nonCachedTokens: 16", cfg) - self.assertIn("approx-prefix-cache-producer", cfg) - self.assertIn("autoTune: false", cfg) - self.assertNotIn("nonCachedTokens: 0", cfg) - self.assertNotIn("prepareDataPlugins", cfg) - - def test_epp_and_sidecar_images_and_config_group_are_pinned(self) -> None: - """Lock the picker + sidecar images and the EndpointPickerConfig API group. - - Nothing else asserts these, so a wrong tag/registry path or a stale config - group passes CI and only surfaces as an EPP/sidecar crashloop at deploy. - These are deliberate literals, not routing._* constants: comparing to the - constant would be tautological (it can't catch a typo in the constant), and - a literal forces a bump to show up here and be reviewed. - """ - out = self._apply() - epp = out["epp"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"] - self.assertEqual( - next(c["image"] for c in epp if c["name"] == "epp"), - "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0", - ) - sidecar = next(c for c in self._serving_pod(out, "decode")["spec"]["containers"] if c["name"] == "pd-sidecar") - self.assertEqual(sidecar["image"], "ghcr.io/llm-d/llm-d-router-disagg-sidecar:v0.9.0") - cfg = out["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - self.assertIn("apiVersion: llm-d.ai/v1alpha1", cfg) - - def test_epp_role_watches_inferenceobjectives(self) -> None: - """The picker watches InferenceObjectives (GIE x-k8s.io group); the Role must allow it.""" - rules = self._apply()["epp-role"].spec.forProvider.manifest["rules"] - self.assertTrue( - any( - "inference.networking.x-k8s.io" in r["apiGroups"] and "inferenceobjectives" in r["resources"] - for r in rules - ), - f"EPP Role missing inferenceobjectives watch: {rules}", - ) +def test_select_backend_leader_worker_engine_is_llmd() -> None: + """A Leader/Worker engine on a Standard cluster selects the llm-d backend.""" + assert base.select_backend(_gang_engine(), "Standard") == base.LLMD - def test_decode_port_follows_user_arg(self) -> None: - """The sidecar and the decode container port track the user's --port, not a hardcoded one.""" - prefill = _standalone_engine(name="prefill") - prefill.phase = "Prefill" - decode = _standalone_engine(name="decode", args=["--model=m", "--port=9000"]) - decode.phase = "Decode" - replica = _replica(engines=[prefill, decode]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = {} - for e in replica.spec.engines: - composed.update(native.NativeBackend().build(replica, e, _PC, base.serving_label(replica), "Standard")) - out = routing.apply(composed, replica, _PC) - containers = self._serving_pod(out, "decode")["spec"]["containers"] - engine = next(c for c in containers if c["name"] == "engine") - sidecar = next(c for c in containers if c["name"] == "pd-sidecar") - self.assertEqual(engine["ports"][0]["containerPort"], 9000) - self.assertIn("--vllm-port=9000", sidecar["args"]) - self.assertEqual(sidecar["ports"][0]["containerPort"], 8000) - - def test_engines_role_labeled(self) -> None: - out = self._apply() - self.assertEqual(self._serving_pod(out, "prefill")["metadata"]["labels"]["llm-d.ai/role"], "prefill") - decode_labels = self._serving_pod(out, "decode")["metadata"]["labels"] - self.assertEqual(decode_labels["llm-d.ai/role"], "decode") - self.assertEqual(decode_labels["app"], "r") - - def test_decode_gets_sidecar_and_moves_engine_port(self) -> None: - out = self._apply() - containers = self._serving_pod(out, "decode")["spec"]["containers"] - names = [c["name"] for c in containers] - self.assertEqual(names, ["engine", "pd-sidecar"]) - engine = next(c for c in containers if c["name"] == "engine") - self.assertEqual(engine["ports"][0]["containerPort"], 8001) - self.assertEqual(engine["readinessProbe"]["timeoutSeconds"], 5) - sidecar = next(c for c in containers if c["name"] == "pd-sidecar") - self.assertEqual(sidecar["ports"][0]["containerPort"], 8000) - self.assertEqual(sidecar["readinessProbe"]["timeoutSeconds"], 5) - self.assertIn("--secure-proxy=false", sidecar["args"]) - - def test_a_decode_engine_keeps_a_scrapeable_port(self) -> None: - """The collector's engine job keeps a pod on the port named `http`. - - Moving the decode engine off 8000 for the sidecar drops the name with - it, and the sidecar takes the port unnamed because it serves inference - rather than /metrics. A decode pod with no named port anywhere is one - nothing scrapes, so a disaggregated deployment reports half its - engines and the shortfall looks like idle capacity. - """ - containers = self._serving_pod(self._apply(), "decode")["spec"]["containers"] - engine = next(c for c in containers if c["name"] == "engine") - self.assertEqual(engine["ports"], [{"name": base.ENGINE_PORT_NAME, "containerPort": 8001}]) - - def test_prefill_has_no_sidecar(self) -> None: - containers = self._serving_pod(self._apply(), "prefill")["spec"]["containers"] - self.assertEqual([c["name"] for c in containers], ["engine"]) - - def test_route_targets_inference_pool(self) -> None: - route = self._apply()[base.ROUTE_KEY].spec.forProvider.manifest - rule = route["spec"]["rules"][0] - ref = rule["backendRefs"][0] - self.assertEqual(ref["kind"], "InferencePool") - self.assertEqual(ref["name"], "r-pool") - # Disable the request timeout so long token streams aren't severed. - self.assertEqual(rule["timeouts"]["request"], "0s") - - def test_selects_engines_by_phase_not_name(self) -> None: - """Roles come from each engine's phase, not its name.""" - decode = _standalone_engine(name="alpha") - decode.phase = "Decode" - prefill = _standalone_engine(name="beta") - prefill.phase = "Prefill" - replica = _replica(engines=[decode, prefill]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = {} - for e in replica.spec.engines: - composed.update(native.NativeBackend().build(replica, e, _PC, base.serving_label(replica), "Standard")) - out = routing.apply(composed, replica, _PC) - # alpha is Decode -> sidecar; beta is Prefill -> none, despite their names. - self.assertEqual( - [c["name"] for c in self._serving_pod(out, "alpha")["spec"]["containers"]], ["engine", "pd-sidecar"] - ) - self.assertEqual([c["name"] for c in self._serving_pod(out, "beta")["spec"]["containers"]], ["engine"]) - self.assertEqual(self._serving_pod(out, "alpha")["metadata"]["labels"]["llm-d.ai/role"], "decode") - self.assertEqual(self._serving_pod(out, "beta")["metadata"]["labels"]["llm-d.ai/role"], "prefill") - - def test_decode_can_be_a_grove_gang(self) -> None: - """A PrefillDecode engine can itself be a Leader/Worker gang, so routing - must decorate a Grove PodCliqueSet's leader clique - role label, serving - label, pd-sidecar, NIXL plumbing - exactly like a Deployment's pod - template. Exercises the _serving_pod_templates normalization that lets - one routing layer decorate both workload shapes.""" - prefill = _standalone_engine(name="prefill") - prefill.phase = "Prefill" - decode = _gang_engine(name="decode", leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - decode.phase = "Decode" - replica = _replica(engines=[prefill, decode]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = { - **native.NativeBackend().build(replica, prefill, _PC, base.serving_label(replica), "Standard"), - **grove.GroveBackend().build(replica, decode, _PC, base.serving_label(replica), "Dynamo"), - } - out = routing.apply(composed, replica, _PC) - - manifest = out["model-serving-decode"].spec.forProvider.manifest - leader_clique = _clique(manifest, "leader") - self.assertEqual(leader_clique["labels"]["llm-d.ai/role"], "decode") - self.assertEqual(leader_clique["labels"]["app"], "r") - leader = leader_clique["spec"]["podSpec"] - self.assertEqual([c["name"] for c in leader["containers"]], ["engine", "pd-sidecar"]) - self.assertTrue( - any(v.get("emptyDir", {}).get("medium") == "Memory" for v in leader["volumes"]), - "leader clique missing Memory /dev/shm volume for NIXL", - ) - engine = next(c for c in leader["containers"] if c["name"] == "engine") - self.assertIn("VLLM_NIXL_SIDE_CHANNEL_HOST", [e["name"] for e in engine["env"]]) - - # The worker clique never serves; routing must not touch it at all - - # its labels stay exactly what the Grove backend composed (just the - # queue label), with no role or serving label added. - worker_clique = _clique(manifest, "worker") - worker = worker_clique["spec"]["podSpec"] - self.assertEqual([c["name"] for c in worker["containers"]], ["engine"]) - self.assertEqual(worker_clique["labels"], {_ENGINE: "decode", _ROLE: "Worker", _QUEUE_LABEL: _QUEUE}) - - -class TestUnifiedRouting(unittest.TestCase): - """Unified serving (or no serving block) fronts the pods with an - InferencePool + endpoint picker in place of a plain Service, so requests - route by prefix cache and load rather than round-robin - one pod or many. - Mirrors how fn.py composes engines then calls routing.apply.""" - - def _apply(self, copies: int = 1) -> dict[str, k8sobjv1alpha1.Object]: - engine = _standalone_engine(copies=copies) - replica = _replica(engines=[engine]) - composed = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - return routing.apply(composed, replica, _PC) - - def test_fronts_with_pool_and_epp(self) -> None: - out = self._apply() - self.assertIn("inference-pool", out) - self.assertIn("epp", out) - self.assertIn("epp-config", out) - pool = out["inference-pool"].spec.forProvider.manifest - self.assertEqual(pool["kind"], "InferencePool") - self.assertEqual(pool["spec"]["endpointPickerRef"]["name"], "r-epp") - - def test_single_pod_also_pools(self) -> None: - """A single serving pod has nothing to pick between, but still gets the - pool. Always fronting with one avoids swapping a Service for a pool when a - second pod appears - a swap that would drop in-flight requests.""" - for copies in (1, 2): - with self.subTest(copies=copies): - out = self._apply(copies=copies) - self.assertIn("inference-pool", out) - self.assertIn("epp", out) - - def test_fronts_a_leader_worker_set(self) -> None: - """A Standard multi-node engine composes a LeaderWorkerSet, and unified - routing must handle that shape too: it reads the engine args for the KV - block size through _serving_pod_templates, which has to normalize a - LeaderWorkerSet's leaderTemplate alongside a Deployment's pod template - and a Grove PodCliqueSet's leader clique. Regression for a shape - normalization that only knew Deployment and PodCliqueSet and raised - KeyError on a LeaderWorkerSet.""" - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _replica(engines=[engine]) - composed = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - out = routing.apply(composed, replica, _PC) - self.assertIn("inference-pool", out) - self.assertEqual(out["model-serving-main"].spec.forProvider.manifest["kind"], "LeaderWorkerSet") - - def test_pool_selects_pods_by_the_serving_label(self) -> None: - """The pool selects the pods by the serving label they already carry, so - no relabeling is needed.""" - pool = self._apply()["inference-pool"].spec.forProvider.manifest - self.assertEqual(pool["spec"]["selector"]["matchLabels"], {base.LABEL_SERVING: "r"}) - - def test_route_targets_inference_pool(self) -> None: - route = self._apply()[base.ROUTE_KEY].spec.forProvider.manifest - ref = route["spec"]["rules"][0]["backendRefs"][0] - self.assertEqual(ref["kind"], "InferencePool") - self.assertEqual(ref["name"], "r-pool") - - def test_epp_config_is_unified_not_disaggregated(self) -> None: - """The unified picker scores by prefix cache and queue depth in a single - profile, with no prefill/decode split, and still needs the - approx-prefix-cache-producer that feeds the prefix-cache scorer.""" - cfg = self._apply()["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - self.assertIn("prefix-cache-scorer", cfg) - self.assertIn("queue-scorer", cfg) - self.assertIn("approx-prefix-cache-producer", cfg) - self.assertNotIn("prefill", cfg) - self.assertNotIn("decider", cfg) - - def test_epp_image_and_config_group_are_pinned(self) -> None: - """Lock the picker image and the EndpointPickerConfig API group for the - unified path too. A deliberate literal (not routing._EPP_IMAGE) so a wrong - tag/registry or a stale config group is caught in review, not as a - deploy-time crashloop. Unified has no sidecar, so only the EPP is checked. - """ - out = self._apply() - epp = out["epp"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"] - self.assertEqual( - next(c["image"] for c in epp if c["name"] == "epp"), - "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0", + +def test_select_backend_leader_worker_engine_is_grove() -> None: + """A Leader/Worker engine on a Dynamo cluster selects the Grove backend.""" + assert base.select_backend(_gang_engine(), "Dynamo") == base.GROVE + + +def _cache_replica( + *, cache: str | None = None, args: list[str] | None = None, command: list[str] | None = None +) -> v1alpha1.ModelReplica: + """A replica with one Standalone engine, referencing cache if one's given.""" + engine = _standalone_engine(args=args or [], command=command) + modelcache = v1alpha1.ModelCacheRef(name=cache) if cache else None + return v1alpha1.ModelReplica( + metadata=metav1.ObjectMeta(namespace="ml-team"), + spec=v1alpha1.SpecModel(clusterName="c", modelCacheRef=modelcache, engines=[engine]), + ) + + +def test_no_cache_no_mounts() -> None: + """A replica with no cache mounts nothing.""" + volumes, mounts = base.cache_mounts(_cache_replica()) + assert (volumes, mounts) == ([], []) + + +def test_cache_adds_volume_and_mount() -> None: + """A replica with a cache mounts the cache's PVC.""" + volumes, mounts = base.cache_mounts(_cache_replica(cache="qwen")) + assert volumes == [{"name": "model-cache", "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}}] + assert mounts == [{"name": "model-cache", "mountPath": "/mnt/models"}] + + +def test_cache_env_points_huggingface_at_the_mount() -> None: + """A replica with a cache points HF_HUB_CACHE at the mount.""" + # The cache is staged in HuggingFace's cache layout, so pointing + # HF_HUB_CACHE at the mount is what lets an engine's own --model= + # resolve against it instead of pulling from HuggingFace (#407). + assert base.cache_env(_cache_replica(cache="qwen")) == [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}] + + +def test_cache_env_empty_without_cache() -> None: + """A replica with no cache gets no cache env.""" + assert base.cache_env(_cache_replica()) == [] + + +def test_cache_env_sets_no_offline_flag() -> None: + """A replica with a cache doesn't set HF_HUB_OFFLINE.""" + # HF_HUB_OFFLINE would break an engine that fetches a *different* repo + # at startup (kimi-k2's separately-gated tokenizer), and resolution + # doesn't need it. + names = {e["name"] for e in base.cache_env(_cache_replica(cache="qwen"))} + assert "HF_HUB_OFFLINE" not in names + + +def _native_cache_replica() -> v1alpha1.ModelReplica: + """A replica with one Standalone engine, referencing the qwen cache.""" + engine = _standalone_engine(args=[]) + return v1alpha1.ModelReplica( + metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), + spec=v1alpha1.SpecModel( + clusterName="cluster-a", + modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), + engines=[engine], + ), + ) + + +def test_native_cache_mounts_pvc_and_sets_cache_env() -> None: + """The native backend mounts a cache and points HF_HUB_CACHE at it, injecting no --model.""" + # A cache contributes a volume, a mount, and the HF_HUB_CACHE that makes + # the engine's own --model= resolve against it. Modelplane injects + # no --model of its own: naming the model is the command's job. + replica = _native_cache_replica() + out = native.NativeBackend().build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Standard") + dep = out["model-serving-main"].spec.forProvider.manifest + pod = dep["spec"]["template"]["spec"] + vol_names = {v["name"] for v in pod["volumes"]} + assert "model-cache" in vol_names + container = pod["containers"][0] + assert {"name": "model-cache", "mountPath": "/mnt/models"} in container["volumeMounts"] + assert {"name": "HF_HUB_CACHE", "value": "/mnt/models"} in container["env"] + assert container["args"] == [] + + +def test_native_cache_user_env_comes_after_cache_env() -> None: + """A Standalone member's own env follows the cache env.""" + # Kubernetes expands $(VAR) left to right, so Modelplane's own entries + # must precede the user's for a user entry to reference them. + replica = _native_cache_replica() + engine = replica.spec.engines[0] + spec = engine.members[0].template.spec + assert spec is not None + spec.containers[0].env = [v1alpha1.EnvItem(name="HF_TOKEN", value="x")] + out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") + container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] + assert container["env"] == [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}, {"name": "HF_TOKEN", "value": "x"}] + + +def _grove_cache_replica( + *, + leader_command: list[str] | None = None, + worker_command: list[str] | None = None, + leader_args: list[str] | None = None, + worker_args: list[str] | None = None, +) -> v1alpha1.ModelReplica: + """A replica with one Leader/Worker engine, referencing the kimi cache.""" + engine = _gang_engine( + leader_command=leader_command, + worker_command=worker_command, + leader_args=leader_args, + worker_args=worker_args, + ) + return v1alpha1.ModelReplica( + metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), + spec=v1alpha1.SpecModel( + clusterName="cluster-a", + modelCacheRef=v1alpha1.ModelCacheRef(name="kimi"), + engines=[engine], + ), + ) + + +def test_grove_cache_both_cliques_mount_cache() -> None: + """The Grove backend mounts a cache on both cliques.""" + replica = _grove_cache_replica(leader_args=[], worker_command=["/bin/sh", "-c", "join"]) + manifest = ( + grove.GroveBackend() + .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] + .spec.forProvider.manifest + ) + for clique_name in ("leader", "worker"): + pod = _clique(manifest, clique_name)["spec"]["podSpec"] + assert "model-cache" in {v["name"] for v in pod["volumes"]} + assert {"name": "model-cache", "mountPath": "/mnt/models"} in pod["containers"][0]["volumeMounts"] + + +def test_grove_cache_sets_cache_env_on_every_clique_and_injects_no_model() -> None: + """The Grove backend points both cliques' HF_HUB_CACHE at a cache, injecting no --model.""" + # A cache gives both cliques HF_HUB_CACHE so their own --model= + # resolves against the mount; Modelplane adds no --model itself. + replica = _grove_cache_replica(leader_args=[], worker_command=["/bin/sh", "-c", "join"]) + manifest = ( + grove.GroveBackend() + .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] + .spec.forProvider.manifest + ) + for clique_name in ("leader", "worker"): + container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] + assert {"name": "HF_HUB_CACHE", "value": "/mnt/models"} in container["env"] + assert "--model=/mnt/models" not in container.get("args", []) + + +def test_grove_cache_command_engine_mounts_cache_without_injecting_model() -> None: + """A Grove member with its own command mounts a cache and keeps its command verbatim.""" + # A member with its own command keeps it verbatim and gets no injected + # --model (it points at the cache with its own flag). + leader_cmd = [ + "/bin/sh", + "-c", + "python3 -m sglang.launch_server --model-path /mnt/models --tp 16", + ] + replica = _grove_cache_replica(leader_command=leader_cmd, worker_command=["/bin/sh", "-c", "join"]) + manifest = ( + grove.GroveBackend() + .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] + .spec.forProvider.manifest + ) + leader = _clique(manifest, "leader")["spec"]["podSpec"]["containers"][0] + assert {"name": "model-cache", "mountPath": "/mnt/models"} in leader["volumeMounts"] + assert leader["command"] == leader_cmd + + +# serving.mode: PrefillDecode routing layers an InferencePool + endpoint +# picker over two engines, role-labels them, and sidecars decode — no unified +# Service. Mirrors how fn.py composes engines then calls routing.apply. + + +def _disaggregated_apply() -> dict[str, k8sobjv1alpha1.Object]: + """Routing for a PrefillDecode replica of two native engines.""" + prefill = _standalone_engine(name="prefill") + prefill.phase = "Prefill" + decode = _standalone_engine(name="decode") + decode.phase = "Decode" + replica = _replica(engines=[prefill, decode]) + replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") + composed = {} + for engine in replica.spec.engines: + composed.update(native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard")) + return routing.apply(composed, replica, _PC) + + +def _serving_pod(out: dict[str, k8sobjv1alpha1.Object], engine_name: str) -> dict: + """The pod template of an engine's Deployment.""" + return out[f"model-serving-{engine_name}"].spec.forProvider.manifest["spec"]["template"] + + +def test_disaggregated_replaces_unified_service_with_pool_and_epp() -> None: + """PrefillDecode routing fronts the engines with an InferencePool and endpoint picker.""" + out = _disaggregated_apply() + assert "inference-pool" in out + assert "epp" in out + assert "epp-config" in out + pool = out["inference-pool"].spec.forProvider.manifest + assert pool["kind"] == "InferencePool" + assert pool["spec"]["endpointPickerRef"]["name"] == "r-epp" + + +def test_disaggregated_the_picker_is_scrapeable_and_attributed() -> None: + """A built-in MetricMapping renames the picker's scheduling latency. + + Nothing can match it unless something scrapes the picker, and the + collector's engine job keeps a pod on two things: the deployment + label, and a container port named `http`. Without both, the mapping + is config that matches nothing for the life of the fleet. + + Its metrics endpoint authenticates callers by TokenReview by default, + which needs a ClusterRole the picker's namespaced ServiceAccount + cannot hold, so every scrape would be rejected. --secure-serving is + left alone: that one is the ext-proc gRPC server Envoy calls. + """ + replica = _replica() + replica.metadata = metav1.ObjectMeta( + name="r", + namespace="ml-team", + labels={base.LABEL_DEPLOYMENT: "qwen3-8b", "modelplane.ai/replica-index": "2"}, + ) + replica.spec.serving = v1alpha1.Serving(mode="Unified") + composed = {} + for engine in replica.spec.engines: + composed.update(native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard")) + template = routing.apply(composed, replica, _PC)["epp"].spec.forProvider.manifest["spec"]["template"] + + labels = template["metadata"]["labels"] + assert labels[base.LABEL_DEPLOYMENT] == "qwen3-8b" + assert labels[base.LABEL_REPLICA] == "2" + assert labels[base.LABEL_ROLE] == "picker" + # Still the Deployment's own selector label, which must not move. + assert labels["app"] == "r-epp" + + container = next(c for c in template["spec"]["containers"] if c["name"] == "epp") + assert {"name": base.ENGINE_PORT_NAME, "containerPort": 9090} in container["ports"] + assert "--metrics-port=9090" in container["args"] + assert "--metrics-endpoint-auth=false" in container["args"] + assert "--secure-serving=false" not in container["args"] + + +def test_disaggregated_injects_nixl_plumbing() -> None: + """Both PrefillDecode engines get the NIXL plumbing the schema can't express.""" + # The plumbing is a Memory /dev/shm and VLLM_NIXL_SIDE_CHANNEL_HOST = pod IP. + out = _disaggregated_apply() + for role in ("prefill", "decode"): + pod = _serving_pod(out, role)["spec"] + assert any(v.get("emptyDir", {}).get("medium") == "Memory" for v in pod["volumes"]), ( + f"{role} missing Memory /dev/shm volume" ) - cfg = out["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - self.assertIn("apiVersion: llm-d.ai/v1alpha1", cfg) - - def test_epp_pod_carries_config_checksum(self) -> None: - """The EPP reads its config once at startup, so a config change must roll - the pod. The pod template carries a sha256 of the rendered config to drive - that rollout.""" - template = self._apply()["epp"].spec.forProvider.manifest["spec"]["template"] - checksum = template["metadata"]["annotations"]["modelplane.ai/epp-config-checksum"] - self.assertEqual(len(checksum), 64) - - -class TestModelExpressEnv(unittest.TestCase): - """On a Dynamo cluster the native (Standalone) and Grove (Leader/Worker) - backends inject the ModelExpress P2P env (MX_SERVER_ADDRESS/MODEL_EXPRESS_URL/ - MX_MODEL_REVISION/MX_P2P_METADATA/POD_*) and the IPC_LOCK security context - into every engine container of a replica that references a cache. The env is - inert unless the engine command opts in with --load-format modelexpress. It's - gated on the cluster's Dynamo stack: on Standard neither backend injects it - (the portable engine command falls back), and the llm-d backend never does. - - HF_HUB_CACHE is deliberately NOT in this set: it's the cache's own env, on - every stack (see base.cache_env), and ModelExpress reads it only as a - fallback for its cache root. Keeping it out here is what makes these - assertions fail if it ever leaks back into modelexpress_env as a duplicate.""" - - _MODELEXPRESS_ENV_NAMES: ClassVar[set[str]] = { - "MX_SERVER_ADDRESS", - "MODEL_EXPRESS_URL", - "MX_MODEL_REVISION", - "MX_P2P_METADATA", - "POD_NAME", - "POD_UID", - "POD_NAMESPACE", + engine = next(c for c in pod["containers"] if c["name"] == "engine") + assert "/dev/shm" in [m["mountPath"] for m in engine["volumeMounts"]] + host = next((e for e in engine["env"] if e["name"] == "VLLM_NIXL_SIDE_CHANNEL_HOST"), None) + assert host is not None, f"{role} missing VLLM_NIXL_SIDE_CHANNEL_HOST" + assert host["valueFrom"]["fieldRef"]["fieldPath"] == "status.podIP" + assert "VLLM_NIXL_SIDE_CHANNEL_PORT" in [e["name"] for e in engine["env"]] + + +def test_disaggregated_epp_config_arms_the_pd_decider() -> None: + """PrefillDecode silently serves decode-only unless the PD decider is armed.""" + # Selective prefix-based-pd-decider needs all of: nonCachedTokens > 0 (0 = + # disabled), the approx-prefix-cache-producer plugin that populates the + # attribute it reads, and that producer pinned to autoTune: false (the + # true default never populates). And it must NOT carry the prepareDataPlugins + # feature gate, which the v0.8.0 EPP image rejects and crashloops on. + cfg = _disaggregated_apply()["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] + assert "prefix-based-pd-decider" in cfg + assert "nonCachedTokens: 16" in cfg + assert "approx-prefix-cache-producer" in cfg + assert "autoTune: false" in cfg + assert "nonCachedTokens: 0" not in cfg + assert "prepareDataPlugins" not in cfg + + +def test_disaggregated_epp_and_sidecar_images_and_config_group_are_pinned() -> None: + """Lock the picker and sidecar images and the EndpointPickerConfig API group.""" + # Nothing else asserts these, so a wrong tag/registry path or a stale config + # group passes CI and only surfaces as an EPP/sidecar crashloop at deploy. + # These are deliberate literals, not routing._* constants: comparing to the + # constant would be tautological (it can't catch a typo in the constant), and + # a literal forces a bump to show up here and be reviewed. + out = _disaggregated_apply() + epp = out["epp"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"] + assert next(c["image"] for c in epp if c["name"] == "epp") == "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0" + sidecar = next(c for c in _serving_pod(out, "decode")["spec"]["containers"] if c["name"] == "pd-sidecar") + assert sidecar["image"] == "ghcr.io/llm-d/llm-d-router-disagg-sidecar:v0.9.0" + cfg = out["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] + assert "apiVersion: llm-d.ai/v1alpha1" in cfg + + +def test_disaggregated_epp_role_watches_inferenceobjectives() -> None: + """The picker watches InferenceObjectives (GIE x-k8s.io group); the Role must allow it.""" + rules = _disaggregated_apply()["epp-role"].spec.forProvider.manifest["rules"] + assert any( + "inference.networking.x-k8s.io" in r["apiGroups"] and "inferenceobjectives" in r["resources"] for r in rules + ), f"EPP Role missing inferenceobjectives watch: {rules}" + + +def test_disaggregated_decode_port_follows_user_arg() -> None: + """The sidecar and the decode container port track the user's --port, not a hardcoded one.""" + prefill = _standalone_engine(name="prefill") + prefill.phase = "Prefill" + decode = _standalone_engine(name="decode", args=["--model=m", "--port=9000"]) + decode.phase = "Decode" + replica = _replica(engines=[prefill, decode]) + replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") + composed = {} + for e in replica.spec.engines: + composed.update(native.NativeBackend().build(replica, e, _PC, base.serving_label(replica), "Standard")) + out = routing.apply(composed, replica, _PC) + containers = _serving_pod(out, "decode")["spec"]["containers"] + engine = next(c for c in containers if c["name"] == "engine") + sidecar = next(c for c in containers if c["name"] == "pd-sidecar") + assert engine["ports"][0]["containerPort"] == 9000 + assert "--vllm-port=9000" in sidecar["args"] + assert sidecar["ports"][0]["containerPort"] == 8000 + + +def test_disaggregated_engines_role_labeled() -> None: + """PrefillDecode routing labels each engine's pods with its role.""" + out = _disaggregated_apply() + assert _serving_pod(out, "prefill")["metadata"]["labels"]["llm-d.ai/role"] == "prefill" + decode_labels = _serving_pod(out, "decode")["metadata"]["labels"] + assert decode_labels["llm-d.ai/role"] == "decode" + assert decode_labels["app"] == "r" + + +def test_disaggregated_decode_gets_sidecar_and_moves_engine_port() -> None: + """The decode engine gets the pd-sidecar on the serving port, and moves to another.""" + out = _disaggregated_apply() + containers = _serving_pod(out, "decode")["spec"]["containers"] + names = [c["name"] for c in containers] + assert names == ["engine", "pd-sidecar"] + engine = next(c for c in containers if c["name"] == "engine") + assert engine["ports"][0]["containerPort"] == 8001 + assert engine["readinessProbe"]["timeoutSeconds"] == 5 + sidecar = next(c for c in containers if c["name"] == "pd-sidecar") + assert sidecar["ports"][0]["containerPort"] == 8000 + assert sidecar["readinessProbe"]["timeoutSeconds"] == 5 + assert "--secure-proxy=false" in sidecar["args"] + + +def test_disaggregated_a_decode_engine_keeps_a_scrapeable_port() -> None: + """The collector's engine job keeps a pod on the port named `http`. + + Moving the decode engine off 8000 for the sidecar drops the name with + it, and the sidecar takes the port unnamed because it serves inference + rather than /metrics. A decode pod with no named port anywhere is one + nothing scrapes, so a disaggregated deployment reports half its + engines and the shortfall looks like idle capacity. + """ + containers = _serving_pod(_disaggregated_apply(), "decode")["spec"]["containers"] + engine = next(c for c in containers if c["name"] == "engine") + assert engine["ports"] == [{"name": base.ENGINE_PORT_NAME, "containerPort": 8001}] + + +def test_disaggregated_prefill_has_no_sidecar() -> None: + """The prefill engine gets no sidecar.""" + containers = _serving_pod(_disaggregated_apply(), "prefill")["spec"]["containers"] + assert [c["name"] for c in containers] == ["engine"] + + +def test_disaggregated_route_targets_inference_pool() -> None: + """PrefillDecode routing points the HTTPRoute at the InferencePool, with no request timeout.""" + route = _disaggregated_apply()[base.ROUTE_KEY].spec.forProvider.manifest + rule = route["spec"]["rules"][0] + ref = rule["backendRefs"][0] + assert ref["kind"] == "InferencePool" + assert ref["name"] == "r-pool" + # Disable the request timeout so long token streams aren't severed. + assert rule["timeouts"]["request"] == "0s" + + +def test_disaggregated_selects_engines_by_phase_not_name() -> None: + """Roles come from each engine's phase, not its name.""" + decode = _standalone_engine(name="alpha") + decode.phase = "Decode" + prefill = _standalone_engine(name="beta") + prefill.phase = "Prefill" + replica = _replica(engines=[decode, prefill]) + replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") + composed = {} + for e in replica.spec.engines: + composed.update(native.NativeBackend().build(replica, e, _PC, base.serving_label(replica), "Standard")) + out = routing.apply(composed, replica, _PC) + # alpha is Decode -> sidecar; beta is Prefill -> none, despite their names. + assert [c["name"] for c in _serving_pod(out, "alpha")["spec"]["containers"]] == ["engine", "pd-sidecar"] + assert [c["name"] for c in _serving_pod(out, "beta")["spec"]["containers"]] == ["engine"] + assert _serving_pod(out, "alpha")["metadata"]["labels"]["llm-d.ai/role"] == "decode" + assert _serving_pod(out, "beta")["metadata"]["labels"]["llm-d.ai/role"] == "prefill" + + +def test_disaggregated_decode_can_be_a_grove_gang() -> None: + """PrefillDecode routing decorates a Grove decode gang's leader clique, and leaves its worker alone.""" + # A PrefillDecode engine can itself be a Leader/Worker gang, so routing + # must decorate a Grove PodCliqueSet's leader clique - role label, serving + # label, pd-sidecar, NIXL plumbing - exactly like a Deployment's pod + # template. Exercises the _serving_pod_templates normalization that lets + # one routing layer decorate both workload shapes. + prefill = _standalone_engine(name="prefill") + prefill.phase = "Prefill" + decode = _gang_engine(name="decode", leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + decode.phase = "Decode" + replica = _replica(engines=[prefill, decode]) + replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") + composed = { + **native.NativeBackend().build(replica, prefill, _PC, base.serving_label(replica), "Standard"), + **grove.GroveBackend().build(replica, decode, _PC, base.serving_label(replica), "Dynamo"), } - # What a cache-referencing engine carries on Dynamo: the cache's env plus - # the MX bundle, and nothing else. - _CACHE_ENV_NAME: ClassVar[str] = "HF_HUB_CACHE" - - def _replica(self, *, cache: bool = True, engines: list[v1alpha1.Engine] | None = None) -> v1alpha1.ModelReplica: - engines = engines if engines is not None else [_standalone_engine(args=[])] - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen") if cache else None, - engines=engines, - ), - ) + out = routing.apply(composed, replica, _PC) + + manifest = out["model-serving-decode"].spec.forProvider.manifest + leader_clique = _clique(manifest, "leader") + assert leader_clique["labels"]["llm-d.ai/role"] == "decode" + assert leader_clique["labels"]["app"] == "r" + leader = leader_clique["spec"]["podSpec"] + assert [c["name"] for c in leader["containers"]] == ["engine", "pd-sidecar"] + assert any(v.get("emptyDir", {}).get("medium") == "Memory" for v in leader["volumes"]), ( + "leader clique missing Memory /dev/shm volume for NIXL" + ) + engine = next(c for c in leader["containers"] if c["name"] == "engine") + assert "VLLM_NIXL_SIDE_CHANNEL_HOST" in [e["name"] for e in engine["env"]] + + # The worker clique never serves; routing must not touch it at all - + # its labels stay exactly what the Grove backend composed (just the + # queue label), with no role or serving label added. + worker_clique = _clique(manifest, "worker") + worker = worker_clique["spec"]["podSpec"] + assert [c["name"] for c in worker["containers"]] == ["engine"] + assert worker_clique["labels"] == {_ENGINE: "decode", _ROLE: "Worker", _QUEUE_LABEL: _QUEUE} + + +# Unified serving (or no serving block) fronts the pods with an +# InferencePool + endpoint picker in place of a plain Service, so requests +# route by prefix cache and load rather than round-robin - one pod or many. +# Mirrors how fn.py composes engines then calls routing.apply. + + +def _unified_apply(copies: int = 1) -> dict[str, k8sobjv1alpha1.Object]: + """Routing for a Unified replica of one native engine.""" + engine = _standalone_engine(copies=copies) + replica = _replica(engines=[engine]) + composed = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") + return routing.apply(composed, replica, _PC) + + +def test_unified_fronts_with_pool_and_epp() -> None: + """Unified routing fronts the engine with an InferencePool and endpoint picker.""" + out = _unified_apply() + assert "inference-pool" in out + assert "epp" in out + assert "epp-config" in out + pool = out["inference-pool"].spec.forProvider.manifest + assert pool["kind"] == "InferencePool" + assert pool["spec"]["endpointPickerRef"]["name"] == "r-epp" + + +@pytest.mark.parametrize("copies", [1, 2]) +def test_unified_single_pod_also_pools(copies: int) -> None: + """A single serving pod gets a pool too, as several do.""" + # A single serving pod has nothing to pick between, but still gets the + # pool. Always fronting with one avoids swapping a Service for a pool when a + # second pod appears - a swap that would drop in-flight requests. + out = _unified_apply(copies=copies) + assert "inference-pool" in out + assert "epp" in out + + +def test_unified_fronts_a_leader_worker_set() -> None: + """Unified routing fronts a LeaderWorkerSet.""" + # A Standard multi-node engine composes a LeaderWorkerSet, and unified + # routing must handle that shape too: it reads the engine args for the KV + # block size through _serving_pod_templates, which has to normalize a + # LeaderWorkerSet's leaderTemplate alongside a Deployment's pod template + # and a Grove PodCliqueSet's leader clique. Regression for a shape + # normalization that only knew Deployment and PodCliqueSet and raised + # KeyError on a LeaderWorkerSet. + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + replica = _replica(engines=[engine]) + composed = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") + out = routing.apply(composed, replica, _PC) + assert "inference-pool" in out + assert out["model-serving-main"].spec.forProvider.manifest["kind"] == "LeaderWorkerSet" + + +def test_unified_pool_selects_pods_by_the_serving_label() -> None: + """The pool selects the pods by the serving label they already carry, so no relabeling is needed.""" + pool = _unified_apply()["inference-pool"].spec.forProvider.manifest + assert pool["spec"]["selector"]["matchLabels"] == {base.LABEL_SERVING: "r"} + + +def test_unified_route_targets_inference_pool() -> None: + """Unified routing points the HTTPRoute at the InferencePool.""" + route = _unified_apply()[base.ROUTE_KEY].spec.forProvider.manifest + ref = route["spec"]["rules"][0]["backendRefs"][0] + assert ref["kind"] == "InferencePool" + assert ref["name"] == "r-pool" + + +def test_unified_epp_config_is_unified_not_disaggregated() -> None: + """The unified picker scores by prefix cache and queue depth, with no prefill/decode split.""" + # It scores in a single profile, and still needs the + # approx-prefix-cache-producer that feeds the prefix-cache scorer. + cfg = _unified_apply()["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] + assert "prefix-cache-scorer" in cfg + assert "queue-scorer" in cfg + assert "approx-prefix-cache-producer" in cfg + assert "prefill" not in cfg + assert "decider" not in cfg + + +def test_unified_epp_image_and_config_group_are_pinned() -> None: + """Lock the picker image and the EndpointPickerConfig API group for the unified path too.""" + # A deliberate literal (not routing._EPP_IMAGE) so a wrong + # tag/registry or a stale config group is caught in review, not as a + # deploy-time crashloop. Unified has no sidecar, so only the EPP is checked. + out = _unified_apply() + epp = out["epp"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"] + assert next(c["image"] for c in epp if c["name"] == "epp") == "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0" + cfg = out["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] + assert "apiVersion: llm-d.ai/v1alpha1" in cfg + + +def test_unified_epp_pod_carries_config_checksum() -> None: + """The EPP pod template carries a sha256 of its config, so a config change rolls the pod.""" + # The EPP reads its config once at startup, so a config change must roll + # the pod. The pod template carries a sha256 of the rendered config to drive + # that rollout. + template = _unified_apply()["epp"].spec.forProvider.manifest["spec"]["template"] + checksum = template["metadata"]["annotations"]["modelplane.ai/epp-config-checksum"] + assert len(checksum) == 64 + + +# On a Dynamo cluster the native (Standalone) and Grove (Leader/Worker) +# backends inject the ModelExpress P2P env (MX_SERVER_ADDRESS/MODEL_EXPRESS_URL/ +# MX_MODEL_REVISION/MX_P2P_METADATA/POD_*) and the IPC_LOCK security context +# into every engine container of a replica that references a cache. The env is +# inert unless the engine command opts in with --load-format modelexpress. It's +# gated on the cluster's Dynamo stack: on Standard neither backend injects it +# (the portable engine command falls back), and the llm-d backend never does. +# +# HF_HUB_CACHE is deliberately NOT in this set: it's the cache's own env, on +# every stack (see base.cache_env), and ModelExpress reads it only as a +# fallback for its cache root. Keeping it out here is what makes these +# assertions fail if it ever leaks back into modelexpress_env as a duplicate. + +_MODELEXPRESS_ENV_NAMES = { + "MX_SERVER_ADDRESS", + "MODEL_EXPRESS_URL", + "MX_MODEL_REVISION", + "MX_P2P_METADATA", + "POD_NAME", + "POD_UID", + "POD_NAMESPACE", +} +# What a cache-referencing engine carries on Dynamo: the cache's env plus +# the MX bundle, and nothing else. +_CACHE_ENV_NAME = "HF_HUB_CACHE" - def test_grove_gang_gets_modelexpress_env_on_both_cliques(self) -> None: - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = self._replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - # Grove also gets the leader-address alias, unconditional on a cache, - # ahead of the cache env and the ModelExpress bundle. - want_env_names = self._MODELEXPRESS_ENV_NAMES | {base.LEADER_ADDRESS_ENV, self._CACHE_ENV_NAME} - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - env_names = {e["name"] for e in container["env"]} - self.assertEqual(env_names, want_env_names, f"{clique_name}: {env_names}") - self.assertEqual(container["env"][0], base.grove_leader_address_env()) - server_env = next(e for e in container["env"] if e["name"] == "MX_SERVER_ADDRESS") - # The per-cluster shared server's well-known Service, qualified by - # its namespace because the engine runs in its team's namespace. - self.assertEqual(server_env["value"], "modelexpress-server.default.svc:8001") - mxurl_env = next(e for e in container["env"] if e["name"] == "MODEL_EXPRESS_URL") - self.assertEqual(mxurl_env["value"], server_env["value"]) - self.assertEqual(container["securityContext"], {"capabilities": {"add": ["IPC_LOCK"]}}) - - def test_grove_gang_without_cache_gets_no_modelexpress_env(self) -> None: - # No cache means no ModelExpress env or security context, but the - # leader-address alias is unconditional (it doesn't depend on a cache). - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = self._replica(cache=False, engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - self.assertEqual(container["env"], [base.grove_leader_address_env()]) - self.assertNotIn("securityContext", container) - - def test_native_engine_gets_modelexpress_env_on_dynamo(self) -> None: - # A Standalone engine on a Dynamo cluster with a cache is as valid a P2P - # peer set as a gang, so it gets the full ModelExpress env and the - # IPC_LOCK security context on its engine container. - replica = self._replica() - out = native.NativeBackend().build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo") - container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] - env = {e["name"]: e for e in container["env"]} - self.assertEqual(set(env), self._MODELEXPRESS_ENV_NAMES | {self._CACHE_ENV_NAME}) - self.assertEqual(env["MX_SERVER_ADDRESS"]["value"], "modelexpress-server.default.svc:8001") - self.assertEqual(env["MX_P2P_METADATA"]["value"], "1") - self.assertEqual(env["HF_HUB_CACHE"]["value"], "/mnt/models") - # Isolates this cache's P2P source identity, qualified by the - # Modelplane namespace (like cache_pvc_name) so two namespaces' caches - # of the same name can't collide at the cluster's one shared server. - self.assertEqual(env["MX_MODEL_REVISION"]["value"], base.cache_pvc_name("ml-team", "qwen")) - for name, field in ( - ("POD_NAME", "metadata.name"), - ("POD_UID", "metadata.uid"), - ("POD_NAMESPACE", "metadata.namespace"), - ): - self.assertEqual(env[name]["valueFrom"]["fieldRef"]["fieldPath"], field) - self.assertEqual(container["securityContext"], {"capabilities": {"add": ["IPC_LOCK"]}}) - - def test_native_engine_gets_no_modelexpress_env_on_standard(self) -> None: - # The same cached Standalone engine on a Standard cluster gets no - # ModelExpress env and no security context: the portable engine command - # falls back. It keeps the cache's own HF_HUB_CACHE, which is not part - # of the ModelExpress bundle and applies on every stack. - replica = self._replica() - out = native.NativeBackend().build( - replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Standard" - ) - container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] - self.assertEqual(container["env"], [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}]) - self.assertNotIn("securityContext", container) - self.assertEqual(container["args"], []) - - -class TestKvBlockSize(unittest.TestCase): - """The EPP prefix-cache producer's blockSizeTokens is derived best-effort - from the engine flags (#179) so it matches the engine's KV block size.""" - - def test_defaults_to_16_when_absent(self) -> None: - self.assertEqual(routing._kv_block_size([]), 16) - self.assertEqual(routing._kv_block_size(["--model=/mnt/models"]), 16) - - def test_reads_vllm_block_size(self) -> None: - self.assertEqual(routing._kv_block_size(["--block-size", "32"]), 32) - self.assertEqual(routing._kv_block_size(["--model=/m", "--block-size=8"]), 8) - - def test_reads_sglang_page_size(self) -> None: - self.assertEqual(routing._kv_block_size(["--page-size=64"]), 64) - - def test_non_integer_falls_back_to_default(self) -> None: - self.assertEqual(routing._kv_block_size(["--block-size", "auto"]), 16) - - def test_rendered_config_uses_block_size(self) -> None: - cfg = routing._disaggregated_epp_config_yaml(32) - self.assertIn("blockSizeTokens: 32", cfg) - self.assertNotIn("BLOCK_SIZE_TOKENS", cfg) - - -class TestRemoteNamespace(unittest.TestCase): - """The mirrored namespace a replica's objects land in. The expected names are - spelled out, because compose-inference-cluster creates the namespace and - compose-model-route and compose-model-cache land objects in it by the same - derivation, and all four must agree.""" - - def test_remote_namespace(self) -> None: - cases = [ - ("a short namespace keeps its name, prefixed and hashed", "ml-team", "mp-ml-team-51733"), - ( - # 63 is the longest a namespace can be, so mp- plus it can't be - # used as is. It's truncated to leave room for the hash. - "the longest valid namespace still yields a valid one", - "a" * 63, - "mp-" + "a" * 54 + "-38bfb", - ), - ] - for name, namespace, want in cases: - with self.subTest(name): - got = base.remote_namespace(_replica(namespace=namespace)) - self.assertEqual(got, want) - self.assertLessEqual(len(got), 63) + +def _modelexpress_replica(*, cache: bool = True, engines: list[v1alpha1.Engine] | None = None) -> v1alpha1.ModelReplica: + """A replica of engines, referencing the qwen cache unless cache is False.""" + engines = engines if engines is not None else [_standalone_engine(args=[])] + return v1alpha1.ModelReplica( + metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), + spec=v1alpha1.SpecModel( + clusterName="cluster-a", + modelCacheRef=v1alpha1.ModelCacheRef(name="qwen") if cache else None, + engines=engines, + ), + ) + + +def test_grove_gang_gets_modelexpress_env_on_both_cliques() -> None: + """A cached Grove gang on Dynamo gets the ModelExpress env and IPC_LOCK on both cliques.""" + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + replica = _modelexpress_replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + # Grove also gets the leader-address alias, unconditional on a cache, + # ahead of the cache env and the ModelExpress bundle. + want_env_names = _MODELEXPRESS_ENV_NAMES | {base.LEADER_ADDRESS_ENV, _CACHE_ENV_NAME} + for clique_name in ("leader", "worker"): + container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] + env_names = {e["name"] for e in container["env"]} + assert env_names == want_env_names, f"{clique_name}: {env_names}" + assert container["env"][0] == base.grove_leader_address_env() + server_env = next(e for e in container["env"] if e["name"] == "MX_SERVER_ADDRESS") + # The per-cluster shared server's well-known Service, qualified by + # its namespace because the engine runs in its team's namespace. + assert server_env["value"] == "modelexpress-server.default.svc:8001" + mxurl_env = next(e for e in container["env"] if e["name"] == "MODEL_EXPRESS_URL") + assert mxurl_env["value"] == server_env["value"] + assert container["securityContext"] == {"capabilities": {"add": ["IPC_LOCK"]}} + + +def test_grove_gang_without_cache_gets_no_modelexpress_env() -> None: + """A Grove gang with no cache gets only the leader address alias, and no security context.""" + # No cache means no ModelExpress env or security context, but the + # leader-address alias is unconditional (it doesn't depend on a cache). + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + replica = _modelexpress_replica(cache=False, engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + for clique_name in ("leader", "worker"): + container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] + assert container["env"] == [base.grove_leader_address_env()] + assert "securityContext" not in container + + +def test_native_engine_gets_modelexpress_env_on_dynamo() -> None: + """A cached Standalone engine on Dynamo gets the ModelExpress env and IPC_LOCK.""" + # A Standalone engine on a Dynamo cluster with a cache is as valid a P2P + # peer set as a gang, so it gets the full ModelExpress env and the + # IPC_LOCK security context on its engine container. + replica = _modelexpress_replica() + out = native.NativeBackend().build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo") + container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] + env = {e["name"]: e for e in container["env"]} + assert set(env) == _MODELEXPRESS_ENV_NAMES | {_CACHE_ENV_NAME} + assert env["MX_SERVER_ADDRESS"]["value"] == "modelexpress-server.default.svc:8001" + assert env["MX_P2P_METADATA"]["value"] == "1" + assert env["HF_HUB_CACHE"]["value"] == "/mnt/models" + # Isolates this cache's P2P source identity, qualified by the + # Modelplane namespace (like cache_pvc_name) so two namespaces' caches + # of the same name can't collide at the cluster's one shared server. + assert env["MX_MODEL_REVISION"]["value"] == base.cache_pvc_name("ml-team", "qwen") + for name, field in ( + ("POD_NAME", "metadata.name"), + ("POD_UID", "metadata.uid"), + ("POD_NAMESPACE", "metadata.namespace"), + ): + assert env[name]["valueFrom"]["fieldRef"]["fieldPath"] == field + assert container["securityContext"] == {"capabilities": {"add": ["IPC_LOCK"]}} + + +def test_native_engine_gets_no_modelexpress_env_on_standard() -> None: + """A cached Standalone engine on Standard gets only the cache env, and no security context.""" + # The same cached Standalone engine on a Standard cluster gets no + # ModelExpress env and no security context: the portable engine command + # falls back. It keeps the cache's own HF_HUB_CACHE, which is not part + # of the ModelExpress bundle and applies on every stack. + replica = _modelexpress_replica() + out = native.NativeBackend().build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Standard") + container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] + assert container["env"] == [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}] + assert "securityContext" not in container + assert container["args"] == [] + + +# The EPP prefix-cache producer's blockSizeTokens is derived best-effort +# from the engine flags (#179) so it matches the engine's KV block size. + + +def test_kv_block_size_defaults_to_16_when_absent() -> None: + """The KV block size defaults to 16 when no flag sets it.""" + assert routing._kv_block_size([]) == 16 + assert routing._kv_block_size(["--model=/mnt/models"]) == 16 + + +def test_kv_block_size_reads_vllm_block_size() -> None: + """The KV block size comes from vLLM's --block-size.""" + assert routing._kv_block_size(["--block-size", "32"]) == 32 + assert routing._kv_block_size(["--model=/m", "--block-size=8"]) == 8 + + +def test_kv_block_size_reads_sglang_page_size() -> None: + """The KV block size comes from SGLang's --page-size.""" + assert routing._kv_block_size(["--page-size=64"]) == 64 + + +def test_kv_block_size_non_integer_falls_back_to_default() -> None: + """A non-integer block size falls back to 16.""" + assert routing._kv_block_size(["--block-size", "auto"]) == 16 + + +def test_kv_block_size_rendered_config_uses_block_size() -> None: + """The rendered EPP config carries the block size in place of its placeholder.""" + cfg = routing._disaggregated_epp_config_yaml(32) + assert "blockSizeTokens: 32" in cfg + assert "BLOCK_SIZE_TOKENS" not in cfg + + +# The mirrored namespace a replica's objects land in. The expected names are +# spelled out, because compose-inference-cluster creates the namespace and +# compose-model-route and compose-model-cache land objects in it by the same +# derivation, and all four must agree. + +REMOTE_NAMESPACE_CASES = [ + pytest.param("ml-team", "mp-ml-team-51733", id="a short namespace keeps its name, prefixed and hashed"), + pytest.param( + # 63 is the longest a namespace can be, so mp- plus it can't be + # used as is. It's truncated to leave room for the hash. + "a" * 63, + "mp-" + "a" * 54 + "-38bfb", + id="the longest valid namespace still yields a valid one", + ), +] + + +@pytest.mark.parametrize(("namespace", "want"), REMOTE_NAMESPACE_CASES) +def test_remote_namespace(namespace: str, want: str) -> None: + """A replica's objects land in a namespace mirroring its own.""" + got = base.remote_namespace(_replica(namespace=namespace)) + assert got == want + assert len(got) <= 63 diff --git a/functions/compose-model-replica/tests/test_fn.py b/functions/compose-model-replica/tests/test_fn.py index ad6d524ec..f78010a1c 100644 --- a/functions/compose-model-replica/tests/test_fn.py +++ b/functions/compose-model-replica/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-model-replica function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.modelreplica import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -29,6 +31,21 @@ # A GPU device request CEL selector, as compose-model-deployment stamps it. _GPU_CEL = 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' +# Unified routing fronts the serving pods with an InferencePool + endpoint +# picker; their manifests are asserted in detail in test_backends. Here we +# only check the function wired the whole set in (and dropped the plain +# Service), then drop their manifests so the golden covers the dispatch, +# wiring and readiness the function itself owns. +_ROUTING_KEYS = { + "inference-pool", + "epp", + "epp-config", + "epp-role", + "epp-rolebinding", + "epp-serviceaccount", + "epp-service", +} + @dataclasses.dataclass class Case: @@ -39,10 +56,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - def _observed_object(*, ready: bool) -> fnv1.Resource: """A composed provider-kubernetes Object as observed back, with the Ready condition its readiness policy derives.""" @@ -66,532 +79,515 @@ def _observed_object(*, ready: bool) -> fnv1.Resource: ) -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function dispatches to a backend to compose serving resources on a remote cluster.""" - - xr = v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta( - name="test-replica", - namespace="ml-team", - labels={ - "modelplane.ai/deployment": "my-deployment", - "modelplane.ai/cluster": "cluster-a", - }, - ), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - engines=[ - v1alpha1.Engine( - name="main", - copies=1, - members=[ - v1alpha1.Member( - role="Standalone", - nodePoolName="frontier", - deviceRequests=[ - v1alpha1.DeviceRequest( - name="gpu", - deviceClassName="gpu.nvidia.com", - count=1, - selectors=[v1alpha1.Selector(cel=_GPU_CEL)], - ), - ], - template=v1alpha1.Template( - spec=v1alpha1.Spec( - containers=[ - v1alpha1.Container( - name="engine", - image="vllm/vllm-openai:latest", - args=["--model=Qwen/Qwen3-0.6B"], - ), - ], - ), +def _compose_cases() -> list[Case]: + """The compose cases. Later cases are built from earlier ones.""" + xr = v1alpha1.ModelReplica( + metadata=metav1.ObjectMeta( + name="test-replica", + namespace="ml-team", + labels={ + "modelplane.ai/deployment": "my-deployment", + "modelplane.ai/cluster": "cluster-a", + }, + ), + spec=v1alpha1.SpecModel( + clusterName="cluster-a", + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[v1alpha1.Selector(cel=_GPU_CEL)], + ), + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ), + ], ), ), - ], - ), - ], - ), - ).model_dump(exclude_none=True, mode="json") + ), + ], + ), + ], + ), + ).model_dump(exclude_none=True, mode="json") - cluster_requirement = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceCluster", - match_name="cluster-a", - ) + cluster_requirement = fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_name="cluster-a", + ) - # Case 1: cluster resolved with providerConfigRef — composes native - # Deployment. First reconcile: none of the composed resources are in - # observed yet, so none are marked ready (the function only asserts - # readiness for a resource it can see in observed state). - req1 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), - ), - ) - req1.required_resources["cluster"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceCluster", - "metadata": {"name": "cluster-a"}, - "spec": { - "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k"}}}, - }, - "status": { - "providerConfigRef": {"name": "cluster-a-pc"}, - "gateway": {"address": "10.0.0.1"}, - }, - } - ) + # Case 1: cluster resolved with providerConfigRef — composes native + # Deployment. First reconcile: none of the composed resources are in + # observed yet, so none are marked ready (the function only asserts + # readiness for a resource it can see in observed state). + req1 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), + ), + ) + req1.required_resources["cluster"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "cluster-a"}, + "spec": { + "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k"}}}, + }, + "status": { + "providerConfigRef": {"name": "cluster-a-pc"}, + "gateway": {"address": "10.0.0.1"}, + }, + } ) ) + ) - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - resources={ - "model-serving-main": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": { - "providerConfigRef": { - "kind": "ClusterProviderConfig", - "name": "cluster-a-pc", - }, - "readiness": { - "policy": "DeriveFromCelQuery", - "celQuery": ( - "has(object.status.conditions) && " - "object.status.conditions.exists(" - 'c, c.type == "Available" && c.status == "True")' - ), - }, - "forProvider": { - "manifest": { - "apiVersion": "apps/v1", - "kind": "Deployment", - "metadata": { - "name": resource.child_name("test-replica", "main"), - "namespace": "mp-ml-team-51733", + want1 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + resources={ + "model-serving-main": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "cluster-a-pc", + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status.conditions) && " + "object.status.conditions.exists(" + 'c, c.type == "Available" && c.status == "True")' + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": { + "name": resource.child_name("test-replica", "main"), + "namespace": "mp-ml-team-51733", + }, + "spec": { + "replicas": 1, + "selector": { + "matchLabels": { + "modelplane.ai/workload": resource.child_name( + "test-replica", "main" + ), + }, }, - "spec": { - "replicas": 1, - "selector": { - "matchLabels": { + "template": { + "metadata": { + "labels": { + "modelplane.ai/deployment": "my-deployment", + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "test-replica", "modelplane.ai/workload": resource.child_name( "test-replica", "main" ), }, }, - "template": { - "metadata": { - "labels": { - "modelplane.ai/deployment": "my-deployment", - "modelplane.ai/engine": "main", - "modelplane.ai/role": "Standalone", - "modelplane.ai/serving": "test-replica", - "modelplane.ai/workload": resource.child_name( - "test-replica", "main" + "spec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "resources": {"claims": [{"name": "devices"}]}, + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + ], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + }, + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + ], + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": resource.child_name( + "test-replica", "main", "standalone", "devices" ), }, - }, - "spec": { - "containers": [ - { - "name": "engine", - "image": "vllm/vllm-openai:latest", - "args": ["--model=Qwen/Qwen3-0.6B"], - "ports": [{"name": "http", "containerPort": 8000}], - "resources": {"claims": [{"name": "devices"}]}, - "volumeMounts": [ - {"name": "dshm", "mountPath": "/dev/shm"}, - ], - "readinessProbe": { - "httpGet": {"path": "/health", "port": 8000}, - "initialDelaySeconds": 30, - "periodSeconds": 10, - "timeoutSeconds": 5, - }, - }, - ], - "volumes": [ - {"name": "dshm", "emptyDir": {"medium": "Memory"}}, - ], - "nodeSelector": {"modelplane.ai/pool": "frontier"}, - "resourceClaims": [ - { - "name": "devices", - "resourceClaimTemplateName": resource.child_name( - "test-replica", "main", "standalone", "devices" - ), - }, - ], - "tolerations": [ - { - "key": "nvidia.com/gpu", - "operator": "Exists", - "effect": "NoSchedule", - }, - ], - }, + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + }, + ], }, }, }, }, }, - } - ), + }, + } ), - "model-route": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": { - "providerConfigRef": { - "kind": "ClusterProviderConfig", - "name": "cluster-a-pc", - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "gateway.networking.k8s.io/v1", - "kind": "HTTPRoute", - "metadata": { - "name": "test-replica", - "namespace": "mp-ml-team-51733", - }, - "spec": { - "parentRefs": [ - { - "name": "cluster-gateway", - "namespace": "modelplane-system", - }, - ], - "rules": [ - { - "matches": [ - { - "path": { - "type": "PathPrefix", - "value": "/ml-team/test-replica/", - }, + ), + "model-route": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "cluster-a-pc", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "HTTPRoute", + "metadata": { + "name": "test-replica", + "namespace": "mp-ml-team-51733", + }, + "spec": { + "parentRefs": [ + { + "name": "cluster-gateway", + "namespace": "modelplane-system", + }, + ], + "rules": [ + { + "matches": [ + { + "path": { + "type": "PathPrefix", + "value": "/ml-team/test-replica/", }, - ], - "timeouts": {"request": "0s"}, - "filters": [ - { - "type": "URLRewrite", - "urlRewrite": { - "path": { - "type": "ReplacePrefixMatch", - "replacePrefixMatch": "/", - }, + }, + ], + "timeouts": {"request": "0s"}, + "filters": [ + { + "type": "URLRewrite", + "urlRewrite": { + "path": { + "type": "ReplacePrefixMatch", + "replacePrefixMatch": "/", }, }, - ], - "backendRefs": [ - { - "group": "inference.networking.k8s.io", - "kind": "InferencePool", - "name": "test-replica-pool", - }, - ], - }, - ], - }, + }, + ], + "backendRefs": [ + { + "group": "inference.networking.k8s.io", + "kind": "InferencePool", + "name": "test-replica-pool", + }, + ], + }, + ], }, }, }, - } - ), + }, + } ), - "resource-claim-main-standalone": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": { - "providerConfigRef": { - "kind": "ClusterProviderConfig", - "name": "cluster-a-pc", - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "resource.k8s.io/v1", - "kind": "ResourceClaimTemplate", - "metadata": { - "name": resource.child_name( - "test-replica", "main", "standalone", "devices" - ), - "namespace": "mp-ml-team-51733", - }, + ), + "resource-claim-main-standalone": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "cluster-a-pc", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": { + "name": resource.child_name( + "test-replica", "main", "standalone", "devices" + ), + "namespace": "mp-ml-team-51733", + }, + "spec": { "spec": { - "spec": { - "devices": { - "requests": [ - { - "name": "gpu", - "exactly": { - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "selectors": [ - {"cel": {"expression": _GPU_CEL}}, - ], - }, + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + {"cel": {"expression": _GPU_CEL}}, + ], }, - ], - }, + }, + ], }, }, }, }, }, - } - ), + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ModelAccepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Deploying", - ), - fnv1.Condition( - type="ModelReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForModel", ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Composing vllm/vllm-openai:latest on cluster-a", - ), - ], - context=structpb.Struct(), - ) - want1.requirements.resources["cluster"].CopyFrom(cluster_requirement) - - # Case 2: cluster not resolved — early return with conditions. - req2 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelAccepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Deploying", ), - ) - - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - # Nothing is composed while waiting, so the XR is marked not ready - # rather than left to aggregate to trivially ready. - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - conditions=[ - fnv1.Condition( - type="ModelAccepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - fnv1.Condition( - type="ModelReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForModel", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for cluster to be resolved", - ), - ], - context=structpb.Struct(), - ) - want2.requirements.resources["cluster"].CopyFrom(cluster_requirement) - - # Case 3: cluster resolved but no providerConfigRef — early return. - req3 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), + fnv1.Condition( + type="ModelReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForModel", ), - ) - req3.required_resources["cluster"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceCluster", - "metadata": {"name": "cluster-a"}, - "spec": { - "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k"}}}, - }, - } - ) - ) - ) + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Composing vllm/vllm-openai:latest on cluster-a", + ), + ], + context=structpb.Struct(), + ) + want1.requirements.resources["cluster"].CopyFrom(cluster_requirement) - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - conditions=[ - fnv1.Condition( - type="ModelAccepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - fnv1.Condition( - type="ModelReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForModel", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for cluster providerConfigRef", - ), - ], - context=structpb.Struct(), - ) - want3.requirements.resources["cluster"].CopyFrom(cluster_requirement) + # Case 2: cluster not resolved — early return with conditions. + req2 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), + ), + ) - # Unified routing fronts the serving pods with an InferencePool + endpoint - # picker; their manifests are asserted in detail in test_backends. Here we - # only check the function wired the whole set in (and dropped the plain - # Service), then drop their manifests so the golden covers the dispatch, - # wiring and readiness the function itself owns. - routing_keys = { - "inference-pool", - "epp", - "epp-config", - "epp-role", - "epp-rolebinding", - "epp-serviceaccount", - "epp-service", - } - for key in routing_keys: - want1.desired.resources[key].CopyFrom(fnv1.Resource()) + want2 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + # Nothing is composed while waiting, so the XR is marked not ready + # rather than left to aggregate to trivially ready. + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + conditions=[ + fnv1.Condition( + type="ModelAccepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + fnv1.Condition( + type="ModelReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForModel", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for cluster to be resolved", + ), + ], + context=structpb.Struct(), + ) + want2.requirements.resources["cluster"].CopyFrom(cluster_requirement) - # Case 4: the resources from case 1 now exist in observed, and the - # workload Object reports Available (so its derived Ready is True). The - # function marks each observed resource ready once its Object reports - # Ready: the workload because it's serving and the rest because existing - # is being ready for them. Built from case 1, mutating only what the - # observed-ready transition changes: the three ready flags, the - # acceptance/readiness conditions, and the dropped first-reconcile event. - req4 = fnv1.RunFunctionRequest() - req4.CopyFrom(req1) - # The workload Object as provider-kubernetes observes it back: applied - # (atProvider.manifest populated) and Available (its derived Ready=True). - req4.observed.resources["model-serving-main"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": {"forProvider": {"manifest": {"kind": "Deployment"}}}, - "status": { - "atProvider": {"manifest": {"kind": "Deployment"}}, - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2025-01-01T00:00:00Z", - }, - ], - }, - } - ), + # Case 3: cluster resolved but no providerConfigRef — early return. + req3 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), + ), + ) + req3.required_resources["cluster"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "cluster-a"}, + "spec": { + "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k"}}}, + }, + } ) ) - # The other two have no runtime readiness to wait on, so under their - # SuccessfulCreate policy provider-kubernetes reports them Ready once - # applied. (The InferencePool + endpoint picker resources aren't - # observed here, so they stay unready.) - for key in ("model-route", "resource-claim-main-standalone"): - req4.observed.resources[key].CopyFrom(_observed_object(ready=True)) + ) - want4 = fnv1.RunFunctionResponse() - want4.CopyFrom(want1) - for key in ("model-serving-main", "model-route", "resource-claim-main-standalone"): - want4.desired.resources[key].ready = fnv1.READY_TRUE - del want4.conditions[:] - want4.conditions.extend( - [ - fnv1.Condition(type="ModelAccepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Accepted"), - fnv1.Condition(type="ModelReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Serving"), - ] + want3 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + conditions=[ + fnv1.Condition( + type="ModelAccepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + fnv1.Condition( + type="ModelReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForModel", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for cluster providerConfigRef", + ), + ], + context=structpb.Struct(), + ) + want3.requirements.resources["cluster"].CopyFrom(cluster_requirement) + + # The routing objects' manifests are dropped from the golden (see + # _ROUTING_KEYS). + for key in _ROUTING_KEYS: + want1.desired.resources[key].CopyFrom(fnv1.Resource()) + + # Case 4: the resources from case 1 now exist in observed, and the + # workload Object reports Available (so its derived Ready is True). The + # function marks each observed resource ready once its Object reports + # Ready: the workload because it's serving and the rest because existing + # is being ready for them. Built from case 1, mutating only what the + # observed-ready transition changes: the three ready flags, the + # acceptance/readiness conditions, and the dropped first-reconcile event. + req4 = fnv1.RunFunctionRequest() + req4.CopyFrom(req1) + # The workload Object as provider-kubernetes observes it back: applied + # (atProvider.manifest populated) and Available (its derived Ready=True). + req4.observed.resources["model-serving-main"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": {"forProvider": {"manifest": {"kind": "Deployment"}}}, + "status": { + "atProvider": {"manifest": {"kind": "Deployment"}}, + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2025-01-01T00:00:00Z", + }, + ], + }, + } + ), ) - # The "Composing ..." event fires only the first reconcile (model-serving - # not yet observed), so it's gone now. - del want4.results[:] - - # Case 5: everything from case 4 plus the routing objects is observed, - # but the endpoint picker's Service failed to apply, say because its - # name was invalid, so its Object isn't Ready. Being observed isn't - # being applied, so it stays unready and holds the replica unready with - # it. Built from case 4, mutating only the routing objects' ready flags. - req5 = fnv1.RunFunctionRequest() - req5.CopyFrom(req4) - for key in routing_keys: - req5.observed.resources[key].CopyFrom(_observed_object(ready=key != "epp-service")) - - want5 = fnv1.RunFunctionResponse() - want5.CopyFrom(want4) - for key in routing_keys - {"epp-service"}: - want5.desired.resources[key].ready = fnv1.READY_TRUE - - # Case 6: as case 5, but everything applied and the endpoint picker's - # Deployment isn't Available yet, so its Object's CEL-derived Ready is - # False. The gateway fails closed without a picker, so the replica stays - # unready with it. - req6 = fnv1.RunFunctionRequest() - req6.CopyFrom(req4) - for key in routing_keys: - req6.observed.resources[key].CopyFrom(_observed_object(ready=key != "epp")) - - want6 = fnv1.RunFunctionResponse() - want6.CopyFrom(want4) - for key in routing_keys - {"epp"}: - want6.desired.resources[key].ready = fnv1.READY_TRUE - - cases = [ - Case(name="cluster ready composes native Deployment", req=req1, want=want1), - Case(name="cluster not resolved returns waiting conditions", req=req2, want=want2), - Case(name="cluster without providerConfigRef returns waiting conditions", req=req3, want=want3), - Case(name="observed resources are marked ready", req=req4, want=want4), - Case(name="an object that failed to apply stays unready", req=req5, want=want5), - Case(name="an unavailable endpoint picker stays unready", req=req6, want=want6), + ) + # The other two have no runtime readiness to wait on, so under their + # SuccessfulCreate policy provider-kubernetes reports them Ready once + # applied. (The InferencePool + endpoint picker resources aren't + # observed here, so they stay unready.) + for key in ("model-route", "resource-claim-main-standalone"): + req4.observed.resources[key].CopyFrom(_observed_object(ready=True)) + + want4 = fnv1.RunFunctionResponse() + want4.CopyFrom(want1) + for key in ("model-serving-main", "model-route", "resource-claim-main-standalone"): + want4.desired.resources[key].ready = fnv1.READY_TRUE + del want4.conditions[:] + want4.conditions.extend( + [ + fnv1.Condition(type="ModelAccepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Accepted"), + fnv1.Condition(type="ModelReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Serving"), ] - - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - got_dict = json_format.MessageToDict(got) - resources = got_dict.get("desired", {}).get("resources", {}) - if "model-serving-main" in resources: - self.assertLessEqual(routing_keys, set(resources)) - self.assertNotIn("model-service", resources) - # The routing objects land in the mirrored namespace too, - # before they're dropped from the golden below. - for key in routing_keys: - manifest = resources[key]["resource"]["spec"]["forProvider"]["manifest"] - self.assertEqual(manifest["metadata"]["namespace"], "mp-ml-team-51733", key) - del resources[key]["resource"] - self.assertEqual( - json_format.MessageToDict(case.want), - got_dict, - "-want, +got", - ) + ) + # The "Composing ..." event fires only the first reconcile (model-serving + # not yet observed), so it's gone now. + del want4.results[:] + + # Case 5: everything from case 4 plus the routing objects is observed, + # but the endpoint picker's Service failed to apply, say because its + # name was invalid, so its Object isn't Ready. Being observed isn't + # being applied, so it stays unready and holds the replica unready with + # it. Built from case 4, mutating only the routing objects' ready flags. + req5 = fnv1.RunFunctionRequest() + req5.CopyFrom(req4) + for key in _ROUTING_KEYS: + req5.observed.resources[key].CopyFrom(_observed_object(ready=key != "epp-service")) + + want5 = fnv1.RunFunctionResponse() + want5.CopyFrom(want4) + for key in _ROUTING_KEYS - {"epp-service"}: + want5.desired.resources[key].ready = fnv1.READY_TRUE + + # Case 6: as case 5, but everything applied and the endpoint picker's + # Deployment isn't Available yet, so its Object's CEL-derived Ready is + # False. The gateway fails closed without a picker, so the replica stays + # unready with it. + req6 = fnv1.RunFunctionRequest() + req6.CopyFrom(req4) + for key in _ROUTING_KEYS: + req6.observed.resources[key].CopyFrom(_observed_object(ready=key != "epp")) + + want6 = fnv1.RunFunctionResponse() + want6.CopyFrom(want4) + for key in _ROUTING_KEYS - {"epp"}: + want6.desired.resources[key].ready = fnv1.READY_TRUE + + return [ + Case(name="cluster ready composes native Deployment", req=req1, want=want1), + Case(name="cluster not resolved returns waiting conditions", req=req2, want=want2), + Case(name="cluster without providerConfigRef returns waiting conditions", req=req3, want=want3), + Case(name="observed resources are marked ready", req=req4, want=want4), + Case(name="an object that failed to apply stays unready", req=req5, want=want5), + Case(name="an unavailable endpoint picker stays unready", req=req6, want=want6), + ] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """The function dispatches to a backend to compose serving resources on a remote cluster.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + got_dict = _to_dict(got) + resources = got_dict.get("desired", {}).get("resources", {}) + if "model-serving-main" in resources: + assert set(resources) >= _ROUTING_KEYS + assert "model-service" not in resources + # The routing objects land in the mirrored namespace too, + # before they're dropped from the golden below. + for key in _ROUTING_KEYS: + manifest = resources[key]["resource"]["spec"]["forProvider"]["manifest"] + assert manifest["metadata"]["namespace"] == "mp-ml-team-51733", key + del resources[key]["resource"] + assert got_dict == _to_dict(case.want) diff --git a/functions/compose-model-route/tests/__init__.py b/functions/compose-model-route/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-model-route/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-model-route/tests/test_fn.py b/functions/compose-model-route/tests/test_fn.py index 0a8755aa2..c19c64d14 100644 --- a/functions/compose-model-route/tests/test_fn.py +++ b/functions/compose-model-route/tests/test_fn.py @@ -14,15 +14,17 @@ """Tests for the compose-model-route function.""" +import asyncio import base64 import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 from models.ai.modelplane.inferencegateway import v1alpha1 as igv1alpha1 @@ -248,374 +250,335 @@ def _manifest(rsp: fnv1.RunFunctionResponse, key: str) -> dict: return resource.struct_to_dict(rsp.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) -class TestGates(unittest.IsolatedAsyncioTestCase): - """Passes where a route can't be composed compose nothing and say why. - Asserting the whole response proves nothing is composed against a cluster the - route can't yet reach, rather than a subset being applied.""" - - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_gates(self) -> None: - composed = _endpoint("self", origin="https://gw-eu.example.com", composed=True) - cases = [ - Case( - name="the gateway's client PKI hasn't issued, so nothing can name its certificate", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) - ), - required_resources=_required( - gateway=[_gateway(client_ca=None)], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [composed]}, - ), +# Passes where a route can't be composed compose nothing and say why. Asserting +# the whole response proves nothing is composed against a cluster the route +# can't yet reach, rather than a subset being applied. +def _gates_cases() -> list[Case]: + composed = _endpoint("self", origin="https://gw-eu.example.com", composed=True) + return [ + Case( + name="the gateway's client PKI hasn't issued, so nothing can name its certificate", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) ), - want=_not_ready( - {"model": _MODEL, "endpoints": {"total": 0, "ready": 0}}, - fn.CONDITION_REASON_WAITING_FOR_GATEWAY, - "InferenceGateway eu has not published its client CA", - _requirements([_entry("d")]), + required_resources=_required( + gateway=[_gateway(client_ca=None)], + clusters=[_cluster("gw-eu")], + **{"endpoints-d": [composed]}, ), ), - Case( - name="no selected endpoint is ready", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) - ), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", ready=False)]}, - ), + want=_not_ready( + {"model": _MODEL, "endpoints": {"total": 0, "ready": 0}}, + fn.CONDITION_REASON_WAITING_FOR_GATEWAY, + "InferenceGateway eu has not published its client CA", + _requirements([_entry("d")]), + ), + ), + Case( + name="no selected endpoint is ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) ), - want=_not_ready( - { - "model": _MODEL, - "address": "203.0.113.1", - "endpoints": {"total": 1, "ready": 0}, - }, - fn.CONDITION_REASON_NO_ENDPOINTS, - "None of the 1 selected ModelEndpoints is ready to carry traffic", - _requirements([_entry("d")]), + required_resources=_required( + gateway=[_gateway()], + clusters=[_cluster("gw-eu")], + **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", ready=False)]}, ), ), - Case( - name="a composed endpoint whose cluster withdrew its CA is dropped", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) - ), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu", ca=None)], - **{"endpoints-d": [composed]}, - ), + want=_not_ready( + { + "model": _MODEL, + "address": "203.0.113.1", + "endpoints": {"total": 1, "ready": 0}, + }, + fn.CONDITION_REASON_NO_ENDPOINTS, + "None of the 1 selected ModelEndpoints is ready to carry traffic", + _requirements([_entry("d")]), + ), + ), + Case( + name="a composed endpoint whose cluster withdrew its CA is dropped", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) ), - want=_not_ready( - { - "model": _MODEL, - "address": "203.0.113.1", - "endpoints": {"total": 1, "ready": 0}, - }, - fn.CONDITION_REASON_NO_ENDPOINTS, - "None of the 1 selected ModelEndpoints is ready to carry traffic", - _requirements([_entry("d")]), - warning="Endpoints left out of the route, their cluster has published no gateway CA: self", + required_resources=_required( + gateway=[_gateway()], + clusters=[_cluster("gw-eu", ca=None)], + **{"endpoints-d": [composed]}, ), ), - Case( - name="a credential Secret missing its key drops the endpoint", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("a")]))) - ), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-a": [_endpoint("wrongkey", origin="https://a.example.com", credential="k")], - "credential-wrongkey": [_secret("k", {"token": "sk-1"})], - }, - ), + want=_not_ready( + { + "model": _MODEL, + "address": "203.0.113.1", + "endpoints": {"total": 1, "ready": 0}, + }, + fn.CONDITION_REASON_NO_ENDPOINTS, + "None of the 1 selected ModelEndpoints is ready to carry traffic", + _requirements([_entry("d")]), + warning="Endpoints left out of the route, their cluster has published no gateway CA: self", + ), + ), + Case( + name="a credential Secret missing its key drops the endpoint", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("a")]))) ), - want=_not_ready( - { - "model": _MODEL, - "address": "203.0.113.1", - "endpoints": {"total": 1, "ready": 0}, + required_resources=_required( + gateway=[_gateway()], + clusters=[_cluster("gw-eu")], + **{ + "endpoints-a": [_endpoint("wrongkey", origin="https://a.example.com", credential="k")], + "credential-wrongkey": [_secret("k", {"token": "sk-1"})], }, - fn.CONDITION_REASON_NO_ENDPOINTS, - "None of the 1 selected ModelEndpoints is ready to carry traffic", - _requirements([_entry("a")], credentials={"wrongkey": "k"}), - warning=( - "Endpoints left out of the route, their credential Secret missing or missing its key: wrongkey" - ), ), ), - ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) - - -class TestCompose(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """A composed self-hosted endpoint at priority 0 and a third-party - provider at priority 1: backends, credential, cluster CA and route.""" - entries = [_entry("d", priority=0), _entry("together", priority=1)] - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-d": [ - _endpoint("self", origin="https://gw-eu.example.com", model="d", composed=True), - ], - "endpoints-together": [ - _endpoint( - "together", - origin="https://api.together.xyz", - model="Qwen/Qwen2.5", - credential="together-key", - ), - ], - "credential-together": [_secret("together-key", {"apiKey": "sk-tog"})], + want=_not_ready( + { + "model": _MODEL, + "address": "203.0.113.1", + "endpoints": {"total": 1, "ready": 0}, }, + fn.CONDITION_REASON_NO_ENDPOINTS, + "None of the 1 selected ModelEndpoints is ready to carry traffic", + _requirements([_entry("a")], credentials={"wrongkey": "k"}), + warning=( + "Endpoints left out of the route, their credential Secret missing or missing its key: wrongkey" + ), ), - ) - got = await self.runner.RunFunction(req, None) - - # The exact set, so an unexpected extra object fails the test. The - # mirrored namespace isn't here: compose-inference-cluster composes it. - self.assertEqual( - set(got.desired.resources), - { - "client-certificate", - "backend-self", - "aibackend-self", - "backend-together", - "aibackend-together", - "credential-together", - "credpolicy-together", - "cluster-ca-gw-eu", - "route", + ), + ] + + +@pytest.mark.parametrize("case", _gates_cases(), ids=lambda case: case.name) +def test_gates(case: Case) -> None: + """A pass where a route can't be composed composes nothing and says why.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) + + +def test_compose() -> None: + """A composed endpoint and a third-party provider get backends, a credential, a cluster CA and a route.""" + # A composed self-hosted endpoint at priority 0 and a third-party provider + # at priority 1. + entries = [_entry("d", priority=0), _entry("together", priority=1)] + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), + required_resources=_required( + gateway=[_gateway()], + clusters=[_cluster("gw-eu")], + **{ + "endpoints-d": [ + _endpoint("self", origin="https://gw-eu.example.com", model="d", composed=True), + ], + "endpoints-together": [ + _endpoint( + "together", + origin="https://api.together.xyz", + model="Qwen/Qwen2.5", + credential="together-key", + ), + ], + "credential-together": [_secret("together-key", {"apiKey": "sk-tog"})], }, - ) + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + # The exact set, so an unexpected extra object fails the test. The + # mirrored namespace isn't here: compose-inference-cluster composes it. + assert set(got.desired.resources) == { + "client-certificate", + "backend-self", + "aibackend-self", + "backend-together", + "aibackend-together", + "credential-together", + "credpolicy-together", + "cluster-ca-gw-eu", + "route", + } - route = _manifest(got, "route") - # The timeouts are the ModelRoute's, which _route_xr sets to values - # other than the ModelService's defaults. - self.assertEqual( - route["spec"]["rules"], - [ + route = _manifest(got, "route") + # The timeouts are the ModelRoute's, which _route_xr sets to values + # other than the ModelService's defaults. + assert route["spec"]["rules"] == [ + { + "matches": [{"headers": [{"type": "Exact", "name": "x-ai-eg-model", "value": _MODEL}]}], + "backendRefs": [ + {"name": _be("self"), "weight": 1, "priority": 0, "modelNameOverride": "d"}, { - "matches": [{"headers": [{"type": "Exact", "name": "x-ai-eg-model", "value": _MODEL}]}], - "backendRefs": [ - {"name": _be("self"), "weight": 1, "priority": 0, "modelNameOverride": "d"}, - { - "name": _be("together"), - "weight": 1, - "priority": 1, - "modelNameOverride": "Qwen/Qwen2.5", - }, - ], - "timeouts": {"request": "600s"}, - "streamIdleTimeout": "0s", - "modelsOwnedBy": _NS, - } - ], - ) - # Declaring the token costs is what makes the ext-proc ask a backend for - # usage on a streamed response, which otherwise reports none, and is - # where the metered counts in the access log come from. - self.assertEqual( - route["spec"]["llmRequestCosts"], - [ - {"metadataKey": "llm_input_token", "type": "InputToken"}, - {"metadataKey": "llm_output_token", "type": "OutputToken"}, - {"metadataKey": "llm_total_token", "type": "TotalToken"}, + "name": _be("together"), + "weight": 1, + "priority": 1, + "modelNameOverride": "Qwen/Qwen2.5", + }, ], - ) + "timeouts": {"request": "600s"}, + "streamIdleTimeout": "0s", + "modelsOwnedBy": _NS, + } + ] + # Declaring the token costs is what makes the ext-proc ask a backend for + # usage on a streamed response, which otherwise reports none, and is + # where the metered counts in the access log come from. + assert route["spec"]["llmRequestCosts"] == [ + {"metadataKey": "llm_input_token", "type": "InputToken"}, + {"metadataKey": "llm_output_token", "type": "OutputToken"}, + {"metadataKey": "llm_total_token", "type": "TotalToken"}, + ] + + # The composed backend pins its cluster's CA and presents the client + # certificate, both this route's own; the third-party one uses the + # system trust store. + assert _manifest(got, "backend-self")["spec"]["tls"] == { + "caCertificateRefs": [{"kind": "ConfigMap", "group": "", "name": "assistant-eu-gw-eu-ca-3dd16"}], + "sni": "gw-eu.example.com", + "clientCertificateRef": {"kind": "Secret", "group": "", "name": "assistant-eu-client-08324"}, + } + assert _manifest(got, "backend-together")["spec"]["tls"] == { + "wellKnownCACertificates": "System", + "sni": "api.together.xyz", + } - # The composed backend pins its cluster's CA and presents the client - # certificate, both this route's own; the third-party one uses the - # system trust store. - self.assertEqual( - _manifest(got, "backend-self")["spec"]["tls"], - { - "caCertificateRefs": [{"kind": "ConfigMap", "group": "", "name": "assistant-eu-gw-eu-ca-3dd16"}], - "sni": "gw-eu.example.com", - "clientCertificateRef": {"kind": "Secret", "group": "", "name": "assistant-eu-client-08324"}, - }, - ) - self.assertEqual( - _manifest(got, "backend-together")["spec"]["tls"], - {"wellKnownCACertificates": "System", "sni": "api.together.xyz"}, - ) + # The caller header is stripped only for the backend we don't operate. + assert "headerMutation" not in _manifest(got, "aibackend-self")["spec"] + assert _manifest(got, "aibackend-together")["spec"]["headerMutation"] == {"remove": ["x-modelplane-caller"]} - # The caller header is stripped only for the backend we don't operate. - self.assertNotIn("headerMutation", _manifest(got, "aibackend-self")["spec"]) - self.assertEqual( - _manifest(got, "aibackend-together")["spec"]["headerMutation"], - {"remove": ["x-modelplane-caller"]}, - ) + # The credential is republished under the fixed apiKey key. + assert _manifest(got, "credential-together")["data"] == {"apiKey": base64.b64encode(b"sk-tog").decode()} + # Named for this route, so no other route in the namespace composes it. + assert _manifest(got, "cluster-ca-gw-eu") == { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "assistant-eu-gw-eu-ca-3dd16", "namespace": "mp-ml-team-51733"}, + "data": {"ca.crt": _CLUSTER_CA}, + } - # The credential is republished under the fixed apiKey key. - self.assertEqual( - _manifest(got, "credential-together")["data"], - {"apiKey": base64.b64encode(b"sk-tog").decode()}, - ) - # Named for this route, so no other route in the namespace composes it. - self.assertEqual( - _manifest(got, "cluster-ca-gw-eu"), - { - "apiVersion": "v1", - "kind": "ConfigMap", - "metadata": {"name": "assistant-eu-gw-eu-ca-3dd16", "namespace": "mp-ml-team-51733"}, - "data": {"ca.crt": _CLUSTER_CA}, - }, - ) + # Every composed object lands in the namespace mirroring the route's own, + # which compose-inference-cluster composes. + for key in ("backend-self", "backend-together", "credential-together", "cluster-ca-gw-eu", "route"): + assert _manifest(got, key)["metadata"]["namespace"] == "mp-ml-team-51733", key - # Every composed object lands in the namespace mirroring the route's own, - # which compose-inference-cluster composes. - for key in ("backend-self", "backend-together", "credential-together", "cluster-ca-gw-eu", "route"): - self.assertEqual(_manifest(got, key)["metadata"]["namespace"], "mp-ml-team-51733", key) + # The route lives in the team namespace but attaches across to the gateway. + assert route["spec"]["parentRefs"][0]["namespace"] == "modelplane-system" - # The route lives in the team namespace but attaches across to the gateway. - self.assertEqual(route["spec"]["parentRefs"][0]["namespace"], "modelplane-system") + # The client certificate the backends present, issued from the gateway's + # CA ClusterIssuer into this namespace. Named for this route, so no other + # route in the namespace composes it, and deleted with the route. + assert ( + "managementPolicies" + not in resource.struct_to_dict(got.desired.resources["client-certificate"].resource)["spec"] + ) + assert _manifest(got, "client-certificate") == { + "apiVersion": "cert-manager.io/v1", + "kind": "Certificate", + "metadata": {"name": "assistant-eu-client-08324", "namespace": "mp-ml-team-51733"}, + "spec": { + "secretName": "assistant-eu-client-08324", + "commonName": "inference-gateway-eu", + "usages": ["client auth", "digital signature", "key encipherment"], + "duration": "2160h", + "renewBefore": "720h", + "privateKey": {"algorithm": "ECDSA", "size": 256, "rotationPolicy": "Always"}, + "issuerRef": {"name": "inference-gateway-ca", "kind": "ClusterIssuer", "group": "cert-manager.io"}, + }, + } - # The client certificate the backends present, issued from the gateway's - # CA ClusterIssuer into this namespace. Named for this route, so no other - # route in the namespace composes it, and deleted with the route. - self.assertNotIn( - "managementPolicies", - resource.struct_to_dict(got.desired.resources["client-certificate"].resource)["spec"], - ) - self.assertEqual( - _manifest(got, "client-certificate"), - { - "apiVersion": "cert-manager.io/v1", - "kind": "Certificate", - "metadata": {"name": "assistant-eu-client-08324", "namespace": "mp-ml-team-51733"}, - "spec": { - "secretName": "assistant-eu-client-08324", - "commonName": "inference-gateway-eu", - "usages": ["client auth", "digital signature", "key encipherment"], - "duration": "2160h", - "renewBefore": "720h", - "privateKey": {"algorithm": "ECDSA", "size": 256, "rotationPolicy": "Always"}, - "issuerRef": {"name": "inference-gateway-ca", "kind": "ClusterIssuer", "group": "cert-manager.io"}, - }, - }, - ) - async def test_route_binds_to_the_listener_matching_the_gateways_tls(self) -> None: - """A TLS gateway serves inference on its HTTPS listener alone, so the - route binds there; without TLS there's only the HTTP listener. Binding to - :80 on a TLS gateway would carry credentials in the clear.""" - for tls, want in ((False, "http"), (True, "https")): - with self.subTest(tls=tls): - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) - ), - required_resources=_required( - gateway=[_gateway(tls=tls)], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", composed=True)]}, - ), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual(_manifest(got, "route")["spec"]["parentRefs"][0]["sectionName"], want) - - async def test_status_reports_address_and_counts(self) -> None: - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")])))), - required_resources=_required( - gateway=[_gateway(address="203.0.113.9")], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", composed=True)]}, - ), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual( - resource.struct_to_dict(got.desired.composite.resource)["status"], - { - "model": _MODEL, - "address": "203.0.113.9", - "endpoints": {"total": 1, "ready": 1}, - }, - ) +@pytest.mark.parametrize(("tls", "want"), [(False, "http"), (True, "https")]) +def test_route_binds_to_the_listener_matching_the_gateways_tls(tls: bool, want: str) -> None: + """The route binds to the HTTPS listener on a TLS gateway, and to the HTTP listener otherwise.""" + # A TLS gateway serves inference on its HTTPS listener alone, so the route + # binds there; without TLS there's only the HTTP listener. Binding to :80 on + # a TLS gateway would carry credentials in the clear. + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")])))), + required_resources=_required( + gateway=[_gateway(tls=tls)], + clusters=[_cluster("gw-eu")], + **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", composed=True)]}, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert _manifest(got, "route")["spec"]["parentRefs"][0]["sectionName"] == want + + +def test_status_reports_address_and_counts() -> None: + """The status reports the model name, the gateway's address, and the endpoint counts.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")])))), + required_resources=_required( + gateway=[_gateway(address="203.0.113.9")], + clusters=[_cluster("gw-eu")], + **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", composed=True)]}, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert resource.struct_to_dict(got.desired.composite.resource)["status"] == { + "model": _MODEL, + "address": "203.0.113.9", + "endpoints": {"total": 1, "ready": 1}, + } - async def test_an_endpoint_matched_twice_belongs_to_the_first_entry(self) -> None: - """A canary entry and a catch-all entry must not both weight one - endpoint; the first that matches it wins.""" - entries = [_entry("kimi", name="canary", priority=0), _entry("kimi", name="catchall", priority=1)] - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-canary": [_endpoint("kimi-a", origin="https://a.example.com")], - "endpoints-catchall": [_endpoint("kimi-a", origin="https://a.example.com")], - }, - ), - ) - got = await self.runner.RunFunction(req, None) - # Neither endpoint is Modelplane-composed, so no client certificate is - # issued. - self.assertNotIn("client-certificate", got.desired.resources) - refs = _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] - self.assertEqual(refs, [{"name": _be("kimi-a"), "weight": 1, "priority": 0}]) - self.assertEqual( - resource.struct_to_dict(got.desired.composite.resource)["status"]["endpoints"], - {"total": 1, "ready": 1}, - ) - async def test_priorities_are_renumbered_without_gaps(self) -> None: - """A ModelService's priorities are an ordering; Envoy's are levels it - walks from 0. A user writing 0 and 5, or a tier gone unready during a - roll, would otherwise leave gaps in what Envoy gets.""" - entries = [_entry("a", priority=0), _entry("b", priority=5), _entry("c", priority=9)] - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - # The middle tier has no ready endpoint, so it drops out and - # must not leave a hole behind it. - "endpoints-a": [_endpoint("a-0", origin="https://a.example.com")], - "endpoints-b": [_endpoint("b-0", origin="https://b.example.com", ready=False)], - "endpoints-c": [_endpoint("c-0", origin="https://c.example.com")], - }, - ), - ) - got = await self.runner.RunFunction(req, None) - refs = _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] - self.assertEqual([r["priority"] for r in refs], [0, 1], "two tiers survive, renumbered 0 and 1") +def test_an_endpoint_matched_twice_belongs_to_the_first_entry() -> None: + """An endpoint two entries match belongs to the first of them.""" + # A canary entry and a catch-all entry must not both weight one endpoint; + # the first that matches it wins. + entries = [_entry("kimi", name="canary", priority=0), _entry("kimi", name="catchall", priority=1)] + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), + required_resources=_required( + gateway=[_gateway()], + clusters=[_cluster("gw-eu")], + **{ + "endpoints-canary": [_endpoint("kimi-a", origin="https://a.example.com")], + "endpoints-catchall": [_endpoint("kimi-a", origin="https://a.example.com")], + }, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + # Neither endpoint is Modelplane-composed, so no client certificate is + # issued. + assert "client-certificate" not in got.desired.resources + refs = _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] + assert refs == [{"name": _be("kimi-a"), "weight": 1, "priority": 0}] + assert resource.struct_to_dict(got.desired.composite.resource)["status"]["endpoints"] == {"total": 1, "ready": 1} + + +def test_priorities_are_renumbered_without_gaps() -> None: + """The priorities of the tiers that have a ready endpoint are renumbered from 0, without gaps.""" + # A ModelService's priorities are an ordering; Envoy's are levels it walks + # from 0. A user writing 0 and 5, or a tier gone unready during a roll, + # would otherwise leave gaps in what Envoy gets. + entries = [_entry("a", priority=0), _entry("b", priority=5), _entry("c", priority=9)] + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), + required_resources=_required( + gateway=[_gateway()], + clusters=[_cluster("gw-eu")], + **{ + # The middle tier has no ready endpoint, so it drops out and + # must not leave a hole behind it. + "endpoints-a": [_endpoint("a-0", origin="https://a.example.com")], + "endpoints-b": [_endpoint("b-0", origin="https://b.example.com", ready=False)], + "endpoints-c": [_endpoint("c-0", origin="https://c.example.com")], + }, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + refs = _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] + assert [r["priority"] for r in refs] == [0, 1], "two tiers survive, renumbered 0 and 1" @dataclasses.dataclass @@ -629,77 +592,71 @@ class WeightCase: want_refs: list[dict] -class TestWeights(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_weight_distribution(self) -> None: - def _origins(*names_: str) -> list[dict]: - return [_endpoint(n, origin=f"https://{n}.example.com") for n in names_] +def _weight_distribution_cases() -> list[WeightCase]: + def _origins(*names_: str) -> list[dict]: + return [_endpoint(n, origin=f"https://{n}.example.com") for n in names_] + + def _ref(ep: str, weight: int, priority: int = 0) -> dict: + return {"name": _be(ep), "weight": weight, "priority": priority} + + return [ + WeightCase( + # An entry's weight is written once but applied per backend, so it + # spreads over the endpoints it matched while the ratio between + # entries survives: 90 over three is 30 each, 10 over one is 10, + # reduced by the gcd to the smallest equivalent integers. + name="a weight spreads across a tier's endpoints, ratio preserved", + entries=[_entry("big", weight=90), _entry("small", weight=10)], + endpoints={ + "endpoints-big": _origins("big-0", "big-1", "big-2"), + "endpoints-small": _origins("small-0"), + }, + want_refs=[ + _ref("big-0", 3), + _ref("big-1", 3), + _ref("big-2", 3), + _ref("small-0", 1), + ], + ), + WeightCase( + # Weight 1 over five endpoints must floor none of them to 0, which + # would drop them from the load assignment rather than share. + name="a weight below its endpoint count floors no endpoint", + entries=[_entry("many", weight=1)], + endpoints={"endpoints-many": _origins("many-0", "many-1", "many-2", "many-3", "many-4")}, + want_refs=[_ref(f"many-{i}", 1) for i in range(5)], + ), + WeightCase( + # A max-weight entry beside a tiny one spread over two endpoints + # scales past the per-backendRef limit even though every weight is + # in bounds, so it rescales to the limit rather than composing a + # route the API server rejects. + name="an extreme but valid ratio is clamped to the limit", + entries=[_entry("big", weight=1000000, priority=0), _entry("small", weight=1, priority=0)], + endpoints={"endpoints-big": _origins("big-0"), "endpoints-small": _origins("small-0", "small-1")}, + want_refs=[_ref("big-0", 1000000), _ref("small-0", 1), _ref("small-1", 1)], + ), + WeightCase( + # The remainder is handed to the first endpoints of a tier, so the + # order must be the endpoints' names rather than the API server's + # unspecified list order, or the composed weights churn. + name="endpoints are ordered by name for a stable split", + entries=[_entry("d")], + endpoints={"endpoints-d": _origins("z", "a", "m")}, + want_refs=[_ref("a", 1), _ref("m", 1), _ref("z", 1)], + ), + ] - def _ref(ep: str, weight: int, priority: int = 0) -> dict: - return {"name": _be(ep), "weight": weight, "priority": priority} - cases = [ - WeightCase( - # An entry's weight is written once but applied per backend, so it - # spreads over the endpoints it matched while the ratio between - # entries survives: 90 over three is 30 each, 10 over one is 10, - # reduced by the gcd to the smallest equivalent integers. - name="a weight spreads across a tier's endpoints, ratio preserved", - entries=[_entry("big", weight=90), _entry("small", weight=10)], - endpoints={ - "endpoints-big": _origins("big-0", "big-1", "big-2"), - "endpoints-small": _origins("small-0"), - }, - want_refs=[ - _ref("big-0", 3), - _ref("big-1", 3), - _ref("big-2", 3), - _ref("small-0", 1), - ], - ), - WeightCase( - # Weight 1 over five endpoints must floor none of them to 0, which - # would drop them from the load assignment rather than share. - name="a weight below its endpoint count floors no endpoint", - entries=[_entry("many", weight=1)], - endpoints={"endpoints-many": _origins("many-0", "many-1", "many-2", "many-3", "many-4")}, - want_refs=[_ref(f"many-{i}", 1) for i in range(5)], - ), - WeightCase( - # A max-weight entry beside a tiny one spread over two endpoints - # scales past the per-backendRef limit even though every weight is - # in bounds, so it rescales to the limit rather than composing a - # route the API server rejects. - name="an extreme but valid ratio is clamped to the limit", - entries=[_entry("big", weight=1000000, priority=0), _entry("small", weight=1, priority=0)], - endpoints={"endpoints-big": _origins("big-0"), "endpoints-small": _origins("small-0", "small-1")}, - want_refs=[_ref("big-0", 1000000), _ref("small-0", 1), _ref("small-1", 1)], - ), - WeightCase( - # The remainder is handed to the first endpoints of a tier, so the - # order must be the endpoints' names rather than the API server's - # unspecified list order, or the composed weights churn. - name="endpoints are ordered by name for a stable split", - entries=[_entry("d")], - endpoints={"endpoints-d": _origins("z", "a", "m")}, - want_refs=[_ref("a", 1), _ref("m", 1), _ref("z", 1)], - ), - ] - for case in cases: - with self.subTest(case.name): - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(case.entries))) - ), - required_resources=_required(gateway=[_gateway()], clusters=[_cluster("gw-eu")], **case.endpoints), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual(_manifest(got, "route")["spec"]["rules"][0]["backendRefs"], case.want_refs) +@pytest.mark.parametrize("case", _weight_distribution_cases(), ids=lambda case: case.name) +def test_weight_distribution(case: WeightCase) -> None: + """Each entry's weight is distributed across the endpoints it matched.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(case.entries)))), + required_resources=_required(gateway=[_gateway()], clusters=[_cluster("gw-eu")], **case.endpoints), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] == case.want_refs @dataclasses.dataclass @@ -712,67 +669,61 @@ class CredentialCase: want: dict -class TestCredentials(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_credential_policy(self) -> None: - secret = resource.child_name(f"{_SVC}-{_GW}", "provider", "credential") - target = {"group": "aigateway.envoyproxy.io", "kind": "AIServiceBackend", "name": _be("provider")} - cases = [ - CredentialCase( - name="a backend speaking OpenAI's API gets the key as a bearer token", - api=None, - want={ - "apiVersion": "aigateway.envoyproxy.io/v1beta1", - "kind": "BackendSecurityPolicy", - "metadata": {"name": _be("provider"), "namespace": "mp-ml-team-51733"}, - "spec": { - "type": "APIKey", - "apiKey": {"secretRef": {"name": secret}}, - "targetRefs": [target], - }, +def _credential_policy_cases() -> list[CredentialCase]: + secret = resource.child_name(f"{_SVC}-{_GW}", "provider", "credential") + target = {"group": "aigateway.envoyproxy.io", "kind": "AIServiceBackend", "name": _be("provider")} + return [ + CredentialCase( + name="a backend speaking OpenAI's API gets the key as a bearer token", + api=None, + want={ + "apiVersion": "aigateway.envoyproxy.io/v1beta1", + "kind": "BackendSecurityPolicy", + "metadata": {"name": _be("provider"), "namespace": "mp-ml-team-51733"}, + "spec": { + "type": "APIKey", + "apiKey": {"secretRef": {"name": secret}}, + "targetRefs": [target], }, - ), - CredentialCase( - name="a backend speaking Anthropic's API gets the key in x-api-key", - api=mev1alpha1.Api(schema="Anthropic"), - want={ - "apiVersion": "aigateway.envoyproxy.io/v1beta1", - "kind": "BackendSecurityPolicy", - "metadata": {"name": _be("provider"), "namespace": "mp-ml-team-51733"}, - "spec": { - "type": "AnthropicAPIKey", - "anthropicAPIKey": {"secretRef": {"name": secret}}, - "targetRefs": [target], - }, + }, + ), + CredentialCase( + name="a backend speaking Anthropic's API gets the key in x-api-key", + api=mev1alpha1.Api(schema="Anthropic"), + want={ + "apiVersion": "aigateway.envoyproxy.io/v1beta1", + "kind": "BackendSecurityPolicy", + "metadata": {"name": _be("provider"), "namespace": "mp-ml-team-51733"}, + "spec": { + "type": "AnthropicAPIKey", + "anthropicAPIKey": {"secretRef": {"name": secret}}, + "targetRefs": [target], }, - ), - ] - for case in cases: - with self.subTest(case.name): - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) - ), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-d": [ - _endpoint( - "provider", - origin="https://api.example.com", - api=case.api, - credential="provider-key", - ) - ], - "credential-provider": [_secret("provider-key", {"apiKey": "sk-provider"})], - }, - ), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual(_manifest(got, "credpolicy-provider"), case.want) + }, + ), + ] + + +@pytest.mark.parametrize("case", _credential_policy_cases(), ids=lambda case: case.name) +def test_credential_policy(case: CredentialCase) -> None: + """A keyed backend's BackendSecurityPolicy sends the key the way its API expects.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")])))), + required_resources=_required( + gateway=[_gateway()], + clusters=[_cluster("gw-eu")], + **{ + "endpoints-d": [ + _endpoint( + "provider", + origin="https://api.example.com", + api=case.api, + credential="provider-key", + ) + ], + "credential-provider": [_secret("provider-key", {"apiKey": "sk-provider"})], + }, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert _manifest(got, "credpolicy-provider") == case.want diff --git a/functions/compose-model-service/tests/__init__.py b/functions/compose-model-service/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-model-service/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-model-service/tests/test_fn.py b/functions/compose-model-service/tests/test_fn.py index 7b10eab6b..4fb2975b9 100644 --- a/functions/compose-model-service/tests/test_fn.py +++ b/functions/compose-model-service/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-model-service function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.inferencegateway import v1alpha1 as igv1alpha1 from models.ai.modelplane.modelroute import v1alpha1 as mrtv1alpha1 @@ -146,277 +148,270 @@ def _required(**resources) -> dict: # noqa: ANN003 } -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - entries = [_entry("kimi-k2")] - cases = [ - Case( - name="gateways not resolved yet: require them and wait", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries)))), +def _compose_cases() -> list[Case]: + entries = [_entry("kimi-k2")] + return [ + Case( + name="gateways not resolved yet: require them and wait", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries)))), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"model": _MODEL, "routes": {"total": 0, "ready": 0}}} + ), + ready=fnv1.READY_FALSE, + ) ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 0, "ready": 0}}} - ), - ready=fnv1.READY_FALSE, - ) - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" - ), - } - ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_WAITING_FOR_GATEWAYS, - message="Waiting for the gateways to resolve", - ) - ], - results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the gateways to resolve")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" + ), + } ), + conditions=[ + fnv1.Condition( + type=fn.CONDITION_TYPE_ROUTING_READY, + status=fnv1.STATUS_CONDITION_FALSE, + reason=fn.CONDITION_REASON_WAITING_FOR_GATEWAYS, + message="Waiting for the gateways to resolve", + ) + ], + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the gateways to resolve")], ), - Case( - name="no gateway selects the service: unreachable, and say so", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_service(entries, labels={"region": "us"})) - ) - ), - required_resources=_required(gateways=[_gateway("eu", "gw-eu", selector={"region": "eu"})]), + ), + Case( + name="no gateway selects the service: unreachable, and say so", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct(_service(entries, labels={"region": "us"})) + ) ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 0, "ready": 0}}} - ), - ready=fnv1.READY_FALSE, - ) - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" - ), - } - ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_NO_GATEWAY, - message=( - "No InferenceGateway's serviceSelector matches this service's labels, " - "so no caller can reach it" - ), - ) - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message=( - "No InferenceGateway's serviceSelector matches this service's labels, " - "so no caller can reach it" - ), - ) - ], + required_resources=_required(gateways=[_gateway("eu", "gw-eu", selector={"region": "eu"})]), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"model": _MODEL, "routes": {"total": 0, "ready": 0}}} + ), + ready=fnv1.READY_FALSE, + ) ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" + ), + } + ), + conditions=[ + fnv1.Condition( + type=fn.CONDITION_TYPE_ROUTING_READY, + status=fnv1.STATUS_CONDITION_FALSE, + reason=fn.CONDITION_REASON_NO_GATEWAY, + message=( + "No InferenceGateway's serviceSelector matches this service's labels, " + "so no caller can reach it" + ), + ) + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message=( + "No InferenceGateway's serviceSelector matches this service's labels, " + "so no caller can reach it" + ), + ) + ], ), - Case( - # A gateway with no address is left out of readiness, but with no - # other gateway there is nowhere a caller could reach the service. - name="its only gateway has no address yet: not RoutingReady", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - ), - required_resources=_required(gateways=[_gateway("eu", "gw-eu")]), + ), + Case( + # A gateway with no address is left out of readiness, but with no + # other gateway there is nowhere a caller could reach the service. + name="its only gateway has no address yet: not RoutingReady", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 1, "ready": 0}}} - ), - ready=fnv1.READY_FALSE, + required_resources=_required(gateways=[_gateway("eu", "gw-eu")]), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"model": _MODEL, "routes": {"total": 1, "ready": 0}}} ), - resources={ - "route-eu": fnv1.Resource(resource=resource.dict_to_struct(_route("eu", "gw-eu"))), - }, + ready=fnv1.READY_FALSE, ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" - ), - } - ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_WAITING_FOR_GATEWAYS, - message="Waiting for gateways to come up: eu", - ) + resources={ + "route-eu": fnv1.Resource(resource=resource.dict_to_struct(_route("eu", "gw-eu"))), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" + ), + } + ), + conditions=[ + fnv1.Condition( + type=fn.CONDITION_TYPE_ROUTING_READY, + status=fnv1.STATUS_CONDITION_FALSE, + reason=fn.CONDITION_REASON_WAITING_FOR_GATEWAYS, + message="Waiting for gateways to come up: eu", + ) + ], + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for gateways to come up: eu")], + ), + ), + Case( + name="two gateways serve it: a ModelRoute each, waiting for both routes", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), + ), + required_resources=_required( + gateways=[ + _gateway("eu", "gw-eu", address="203.0.113.1"), + _gateway("us", "gw-us", address="203.0.113.2"), ], - results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for gateways to come up: eu")], ), ), - Case( - name="two gateways serve it: a ModelRoute each, waiting for both routes", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - ), - required_resources=_required( - gateways=[ - _gateway("eu", "gw-eu", address="203.0.113.1"), - _gateway("us", "gw-us", address="203.0.113.2"), - ], + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"model": _MODEL, "routes": {"total": 2, "ready": 0}}} + ), + ready=fnv1.READY_FALSE, ), + resources={ + "route-eu": fnv1.Resource(resource=resource.dict_to_struct(_route("eu", "gw-eu"))), + "route-us": fnv1.Resource(resource=resource.dict_to_struct(_route("us", "gw-us"))), + }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 2, "ready": 0}}} - ), - ready=fnv1.READY_FALSE, + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" ), - resources={ - "route-eu": fnv1.Resource(resource=resource.dict_to_struct(_route("eu", "gw-eu"))), - "route-us": fnv1.Resource(resource=resource.dict_to_struct(_route("us", "gw-us"))), - }, - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" - ), - } - ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_WAITING_FOR_ROUTES, - message="Waiting for routes on gateways: eu, us", - ) - ], - results=[ - fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for routes on gateways: eu, us") + } + ), + conditions=[ + fnv1.Condition( + type=fn.CONDITION_TYPE_ROUTING_READY, + status=fnv1.STATUS_CONDITION_FALSE, + reason=fn.CONDITION_REASON_WAITING_FOR_ROUTES, + message="Waiting for routes on gateways: eu, us", + ) + ], + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for routes on gateways: eu, us")], + ), + ), + Case( + name="both routes accepted: ModelRoutes ready, service RoutingReady", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), + resources={ + "route-eu": _observed_route("eu", ready=True), + "route-us": _observed_route("us", ready=True), + }, + ), + required_resources=_required( + gateways=[ + _gateway("eu", "gw-eu", address="203.0.113.1"), + _gateway("us", "gw-us", address="203.0.113.2"), ], ), ), - Case( - name="both routes accepted: ModelRoutes ready, service RoutingReady", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - resources={ - "route-eu": _observed_route("eu", ready=True), - "route-us": _observed_route("us", ready=True), - }, - ), - required_resources=_required( - gateways=[ - _gateway("eu", "gw-eu", address="203.0.113.1"), - _gateway("us", "gw-us", address="203.0.113.2"), - ], + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"model": _MODEL, "routes": {"total": 2, "ready": 2}}} + ), + ready=fnv1.READY_TRUE, ), + resources={ + "route-eu": fnv1.Resource( + resource=resource.dict_to_struct(_route("eu", "gw-eu")), ready=fnv1.READY_TRUE + ), + "route-us": fnv1.Resource( + resource=resource.dict_to_struct(_route("us", "gw-us")), ready=fnv1.READY_TRUE + ), + }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 2, "ready": 2}}} - ), - ready=fnv1.READY_TRUE, + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" ), - resources={ - "route-eu": fnv1.Resource( - resource=resource.dict_to_struct(_route("eu", "gw-eu")), ready=fnv1.READY_TRUE - ), - "route-us": fnv1.Resource( - resource=resource.dict_to_struct(_route("us", "gw-us")), ready=fnv1.READY_TRUE - ), - }, - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" - ), - } - ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_TRUE, - reason=fn.CONDITION_REASON_ROUTES_ACCEPTED, - ) - ], + } ), - ), - ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) - - async def test_absent_selector_serves_every_service(self) -> None: - """A gateway with no serviceSelector serves the service, and one still - coming up (no address) is excluded from readiness rather than failing - it: both RoutingReady and the service's own Ready ignore its unready - ModelRoute.""" - entries = [_entry("kimi-k2")] - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - resources={"route-eu": _observed_route("eu", ready=True)}, - ), - required_resources=_required( - gateways=[ - _gateway("eu", "gw-eu", address="203.0.113.1"), - _gateway("us", "gw-us"), # no address: still coming up + conditions=[ + fnv1.Condition( + type=fn.CONDITION_TYPE_ROUTING_READY, + status=fnv1.STATUS_CONDITION_TRUE, + reason=fn.CONDITION_REASON_ROUTES_ACCEPTED, + ) ], ), - ) - got = await self.runner.RunFunction(req, None) - self.assertIn("route-eu", got.desired.resources) - self.assertIn("route-us", got.desired.resources) - cond = next(c for c in got.conditions if c.type == fn.CONDITION_TYPE_ROUTING_READY) - self.assertEqual(cond.status, fnv1.STATUS_CONDITION_TRUE, "us has no address, so it doesn't block") - self.assertEqual(got.desired.composite.ready, fnv1.READY_TRUE, "nor does its unready ModelRoute") + ), + ] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes a ModelRoute per serving gateway and reports routing readiness.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) + + +def test_absent_selector_serves_every_service() -> None: + """A gateway with no serviceSelector serves the service, and one with no address doesn't block readiness.""" + # A gateway with no serviceSelector serves the service, and one still + # coming up (no address) is excluded from readiness rather than failing + # it: both RoutingReady and the service's own Ready ignore its unready + # ModelRoute. + entries = [_entry("kimi-k2")] + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), + resources={"route-eu": _observed_route("eu", ready=True)}, + ), + required_resources=_required( + gateways=[ + _gateway("eu", "gw-eu", address="203.0.113.1"), + _gateway("us", "gw-us"), # no address: still coming up + ], + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert "route-eu" in got.desired.resources + assert "route-us" in got.desired.resources + cond = next(c for c in got.conditions if c.type == fn.CONDITION_TYPE_ROUTING_READY) + assert cond.status == fnv1.STATUS_CONDITION_TRUE, "us has no address, so it doesn't block" + assert got.desired.composite.ready == fnv1.READY_TRUE, "nor does its unready ModelRoute" diff --git a/functions/compose-nebius-cluster/tests/test_fn.py b/functions/compose-nebius-cluster/tests/test_fn.py index 0561ac006..74b131657 100644 --- a/functions/compose-nebius-cluster/tests/test_fn.py +++ b/functions/compose-nebius-cluster/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-nebius-cluster function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.nebiuscluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -36,10 +38,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - # The Nebius ClusterProviderConfig the function reads the credentials # Secret off. _NEBIUS_PROVIDER_CONFIG = { @@ -414,405 +412,402 @@ def _observed_ready(desired: dict) -> fnv1.Resource: ) -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function composes Nebius mk8s cluster infrastructure.""" - cases = [ - Case( - name="first pass composes infra resources; autoscaling from maxNodeCount", - req=_req([_GPU_POOL]), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - # The CSI driver release and StorageClass aren't - # composed yet: the cluster isn't observed, so the - # ProviderConfigs can't reach it. - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - # nodeCount defaults to 1 and minNodeCount is - # unset, so autoscaling starts at the node count. - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), +def _compose_cases() -> list[Case]: + """The cases for test_compose, with requirements patched onto their wants by position.""" + cases = [ + Case( + name="first pass composes infra resources; autoscaling from maxNodeCount", + req=_req([_GPU_POOL]), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), + "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), + "cloud-init": fnv1.Resource( + resource=resource.dict_to_struct(_cloud_init_secret()), + ready=fnv1.READY_TRUE, + ), + # The CSI driver release and StorageClass aren't + # composed yet: the cluster isn't observed, so the + # ProviderConfigs can't reach it. + "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), + # nodeCount defaults to 1 and minNodeCount is + # unset, so autoscaling starts at the node count. + "nodegroup-gpu-h100": fnv1.Resource( + resource=resource.dict_to_struct( + _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), ), - }, - ), - context=structpb.Struct(), + ready=fnv1.READY_TRUE, + ), + }, ), + context=structpb.Struct(), ), - Case( - name="provider config not yet fetched gates provider configs, not infra", - req=_req([_GPU_POOL], with_provider_config=False), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_status(with_credentials=False)), - ), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), - ), - }, + ), + Case( + name="provider config not yet fetched gates provider configs, not infra", + req=_req([_GPU_POOL], with_provider_config=False), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct(_status(with_credentials=False)), ), - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for Nebius ClusterProviderConfig default", + resources={ + "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), + "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), + "cloud-init": fnv1.Resource( + resource=resource.dict_to_struct(_cloud_init_secret()), + ready=fnv1.READY_TRUE, ), - ], - context=structpb.Struct(), + "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), + "nodegroup-gpu-h100": fnv1.Resource( + resource=resource.dict_to_struct( + _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), + ), + ), + }, ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for Nebius ClusterProviderConfig default", + ), + ], + context=structpb.Struct(), ), - Case( - name="deleted provider config keeps credentials from the observed ProviderConfig", - req=_req( - [_GPU_POOL], - observed_resources={ + ), + Case( + name="deleted provider config keeps credentials from the observed ProviderConfig", + req=_req( + [_GPU_POOL], + observed_resources={ + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + ), + ), + }, + with_provider_config=False, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), + "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), + "cloud-init": fnv1.Resource( + resource=resource.dict_to_struct(_cloud_init_secret()), + ready=fnv1.READY_TRUE, + ), + "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), + "nodegroup-gpu-h100": fnv1.Resource( + resource=resource.dict_to_struct( + _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), + ), + ), "provider-config-kubernetes": fnv1.Resource( resource=resource.dict_to_struct( _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), ), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, ), }, - with_provider_config=False, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Nebius ClusterProviderConfig default not found; keeping the " + "credentials the composed ProviderConfig already carries", ), - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Nebius ClusterProviderConfig default not found; keeping the " - "credentials the composed ProviderConfig already carries", + ], + context=structpb.Struct(), + ), + ), + Case( + name="fixed-size fabric pool composes a GPU cluster and fixedNodeCount", + req=_req( + [ + v1alpha1.NodePool( + name="gpu-h100", + role="GPU", + platform="gpu-h100-sxm", + preset="8gpu-128vcpu-1600gb", + diskSizeGb=200, + nodeCount=2, + fabric=v1alpha1.Fabric( + type="InfiniBand", + infiniband=v1alpha1.Infiniband(fabric="fabric-2"), ), - ], - context=structpb.Struct(), - ), + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + ), + ] ), - Case( - name="fixed-size fabric pool composes a GPU cluster and fixedNodeCount", - req=_req( - [ - v1alpha1.NodePool( - name="gpu-h100", - role="GPU", - platform="gpu-h100-sxm", - preset="8gpu-128vcpu-1600gb", - diskSizeGb=200, - nodeCount=2, - fabric=v1alpha1.Fabric( - type="InfiniBand", - infiniband=v1alpha1.Infiniband(fabric="fabric-2"), - ), - gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), + "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), + "cloud-init": fnv1.Resource( + resource=resource.dict_to_struct(_cloud_init_secret()), + ready=fnv1.READY_TRUE, ), - ] - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - "gpu-cluster-fabric-2": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "compute.nebius.m.upbound.io/v1beta1", - "kind": "GpuCluster", - "metadata": {"labels": {"modelplane.ai/fabric": "fabric-2"}}, - "spec": { - "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, - "forProvider": { - "name": "test-cluster-fabric-2", - "infinibandFabric": "fabric-2", - }, + "gpu-cluster-fabric-2": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.nebius.m.upbound.io/v1beta1", + "kind": "GpuCluster", + "metadata": {"labels": {"modelplane.ai/fabric": "fabric-2"}}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "name": "test-cluster-fabric-2", + "infinibandFabric": "fabric-2", }, - } - ), + }, + } ), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu( - { - "gpuCluster": { - "idSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/fabric": "fabric-2"}, - }, + ), + "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), + "nodegroup-gpu-h100": fnv1.Resource( + resource=resource.dict_to_struct( + _nodegroup_gpu( + { + "gpuCluster": { + "idSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/fabric": "fabric-2"}, }, }, - fixedNodeCount=2, - ), + }, + fixedNodeCount=2, ), ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), ), - }, - ), - context=structpb.Struct(), - ), - ), - Case( - name="marks managed resources ready from observed conditions", - req=_req( - [_GPU_POOL], - observed_resources={ - "network": _observed_ready(_network()), - "subnet": _observed_ready(_subnet()), - "cluster": _observed_ready(_cluster()), - "filesystem": _observed_ready(_filesystem()), - "release-csi-mounted-fs-path": _observed_ready(_csi_release()), - "nodegroup-system": _observed_ready(_nodegroup_system()), - "nodegroup-gpu-h100": _observed_ready( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), + ready=fnv1.READY_TRUE, ), }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ready=fnv1.READY_TRUE, - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ready=fnv1.READY_TRUE, - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "filesystem": fnv1.Resource( - resource=resource.dict_to_struct(_filesystem()), - ready=fnv1.READY_TRUE, - ), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - # The cluster is observed, so the CSI driver - # release and StorageClass are composed too. - "release-csi-mounted-fs-path": fnv1.Resource( - resource=resource.dict_to_struct(_csi_release()), - ready=fnv1.READY_TRUE, - ), - "storage-class-rwx-fs": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource( - resource=resource.dict_to_struct(_nodegroup_system()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), - ready=fnv1.READY_TRUE, + context=structpb.Struct(), + ), + ), + Case( + name="marks managed resources ready from observed conditions", + req=_req( + [_GPU_POOL], + observed_resources={ + "network": _observed_ready(_network()), + "subnet": _observed_ready(_subnet()), + "cluster": _observed_ready(_cluster()), + "filesystem": _observed_ready(_filesystem()), + "release-csi-mounted-fs-path": _observed_ready(_csi_release()), + "nodegroup-system": _observed_ready(_nodegroup_system()), + "nodegroup-gpu-h100": _observed_ready( + _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct(_network()), + ready=fnv1.READY_TRUE, + ), + "subnet": fnv1.Resource( + resource=resource.dict_to_struct(_subnet()), + ready=fnv1.READY_TRUE, + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, + ), + "filesystem": fnv1.Resource( + resource=resource.dict_to_struct(_filesystem()), + ready=fnv1.READY_TRUE, + ), + "cloud-init": fnv1.Resource( + resource=resource.dict_to_struct(_cloud_init_secret()), + ready=fnv1.READY_TRUE, + ), + # The cluster is observed, so the CSI driver + # release and StorageClass are composed too. + "release-csi-mounted-fs-path": fnv1.Resource( + resource=resource.dict_to_struct(_csi_release()), + ready=fnv1.READY_TRUE, + ), + "storage-class-rwx-fs": fnv1.Resource( + resource=resource.dict_to_struct(_storage_class()), + ready=fnv1.READY_TRUE, + ), + "nodegroup-system": fnv1.Resource( + resource=resource.dict_to_struct(_nodegroup_system()), + ready=fnv1.READY_TRUE, + ), + "nodegroup-gpu-h100": fnv1.Resource( + resource=resource.dict_to_struct( + _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), ), - }, - ), - context=structpb.Struct(), + ready=fnv1.READY_TRUE, + ), + }, ), + context=structpb.Struct(), ), - Case( - name="custom credentials flow through to all cloud MRs", - req=_req( - [_GPU_POOL], - credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-nebius-account"), - provider_config_resource={ - "apiVersion": "nebius.m.upbound.io/v1beta1", - "kind": "ProviderConfig", - "metadata": {"name": "my-nebius-account", "namespace": "crossplane-system"}, - "spec": { - "identity": {"type": "ServiceAccount"}, - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "crossplane-system", - "name": "nebius-credentials", - "key": "credentials.json", - }, + ), + Case( + name="custom credentials flow through to all cloud MRs", + req=_req( + [_GPU_POOL], + credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-nebius-account"), + provider_config_resource={ + "apiVersion": "nebius.m.upbound.io/v1beta1", + "kind": "ProviderConfig", + "metadata": {"name": "my-nebius-account", "namespace": "crossplane-system"}, + "spec": { + "identity": {"type": "ServiceAccount"}, + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "crossplane-system", + "name": "nebius-credentials", + "key": "credentials.json", }, - "projectID": "project-e00test", }, + "projectID": "project-e00test", }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network("ProviderConfig", "my-nebius-account")), - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet("ProviderConfig", "my-nebius-account")), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster("ProviderConfig", "my-nebius-account")), - ), - "filesystem": fnv1.Resource( - resource=resource.dict_to_struct(_filesystem("ProviderConfig", "my-nebius-account")), - ), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_system("ProviderConfig", "my-nebius-account"), - ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct(_network("ProviderConfig", "my-nebius-account")), + ), + "subnet": fnv1.Resource( + resource=resource.dict_to_struct(_subnet("ProviderConfig", "my-nebius-account")), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster("ProviderConfig", "my-nebius-account")), + ), + "filesystem": fnv1.Resource( + resource=resource.dict_to_struct(_filesystem("ProviderConfig", "my-nebius-account")), + ), + "cloud-init": fnv1.Resource( + resource=resource.dict_to_struct(_cloud_init_secret()), + ready=fnv1.READY_TRUE, + ), + "nodegroup-system": fnv1.Resource( + resource=resource.dict_to_struct( + _nodegroup_system("ProviderConfig", "my-nebius-account"), ), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu( - {}, - "ProviderConfig", - "my-nebius-account", - autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, - ), + ), + "nodegroup-gpu-h100": fnv1.Resource( + resource=resource.dict_to_struct( + _nodegroup_gpu( + {}, + "ProviderConfig", + "my-nebius-account", + autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, ), ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), ), - }, - ), - context=structpb.Struct(), + ready=fnv1.READY_TRUE, + ), + }, ), + context=structpb.Struct(), ), - ] - - # Every compose path declares the provider config requirement; the - # selector kind and name vary by credentials. - custom_creds_selector = fnv1.ResourceSelector( - api_version="nebius.m.upbound.io/v1beta1", - kind="ProviderConfig", - match_name="my-nebius-account", - namespace="modelplane-system", - ) - for case in cases[:-1]: - case.want.requirements.resources["nebius-provider-config"].CopyFrom(_PROVIDER_CONFIG_SELECTOR) - cases[-1].want.requirements.resources["nebius-provider-config"].CopyFrom(custom_creds_selector) - - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) + ), + ] + + # Every compose path declares the provider config requirement; the + # selector kind and name vary by credentials. + custom_creds_selector = fnv1.ResourceSelector( + api_version="nebius.m.upbound.io/v1beta1", + kind="ProviderConfig", + match_name="my-nebius-account", + namespace="modelplane-system", + ) + for case in cases[:-1]: + case.want.requirements.resources["nebius-provider-config"].CopyFrom(_PROVIDER_CONFIG_SELECTOR) + cases[-1].want.requirements.resources["nebius-provider-config"].CopyFrom(custom_creds_selector) + return cases + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes a NebiusCluster's network, mk8s cluster, node groups and ProviderConfigs.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-serving-stack/tests/__init__.py b/functions/compose-serving-stack/tests/__init__.py deleted file mode 100644 index 5d373016d..000000000 --- a/functions/compose-serving-stack/tests/__init__.py +++ /dev/null @@ -1,15 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - - diff --git a/functions/compose-serving-stack/tests/test_collector.py b/functions/compose-serving-stack/tests/test_collector.py index f1b8217c6..d6692f3ba 100644 --- a/functions/compose-serving-stack/tests/test_collector.py +++ b/functions/compose-serving-stack/tests/test_collector.py @@ -16,8 +16,8 @@ import re import typing -import unittest +import pytest import yaml from function import collector, stacks from models.ai.modelplane.metricmapping import v1alpha1 as mmv1alpha1 @@ -59,435 +59,455 @@ def _config(*, extensions: dict | None = None, sinks: list | None = None) -> dic ) -class TestConfig(unittest.TestCase): - """The collector configuration this renders.""" +def _objects(secret: str | None = None) -> dict: + sinks = [_sink(secret=secret)] if secret else _SINKS + return {k: m for k, m, _ in collector.objects("prod-us-east", list(stacks.BUILTIN_MAPPINGS), sinks, _EXTENSIONS)} - def test_pipeline_order(self) -> None: - """The rename runs before the identity is lifted onto the resource. - Discovery writes the identity onto each datapoint and groupbyattrs - lifts it; a statement matching on a metric's name has to run while the - datapoints are still where the rename can reach them. - """ - procs = _config()["service"]["pipelines"]["metrics"]["processors"] - self.assertLess(procs.index("transform/modelplane"), procs.index("groupbyattrs/identity")) - self.assertEqual(procs[-1], "batch") +def test_pipeline_order() -> None: + """The rename runs before the identity is lifted onto the resource. - def test_only_modelplane_leaves_the_cluster(self) -> None: - """A series the statements didn't rename is dropped.""" - self.assertIn("filter/modelplane", _config()["processors"]) - self.assertIn("filter/modelplane", _config()["service"]["pipelines"]["metrics"]["processors"]) + Discovery writes the identity onto each datapoint and groupbyattrs + lifts it; a statement matching on a metric's name has to run while the + datapoints are still where the rename can reach them. + """ + procs = _config()["service"]["pipelines"]["metrics"]["processors"] + assert procs.index("transform/modelplane") < procs.index("groupbyattrs/identity") + assert procs[-1] == "batch" - def test_cluster_is_stamped_here(self) -> None: - """One receiver downstream sees a merged stream and can't tell senders apart.""" - attrs = _config()["processors"]["resource/cluster"]["attributes"] - self.assertEqual(attrs, [{"key": "cluster", "value": "prod-us-east", "action": "upsert"}]) - def test_the_jobs_cover_disjoint_pods(self) -> None: - """A pod two jobs both collect arrives twice, under two job names.""" - jobs = { - j["job_name"]: j["relabel_configs"] - for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"] - } - substrate = jobs["modelplane-substrate"] - - def predicate(rules: list[dict], action: str) -> set[tuple]: - return {(tuple(r["source_labels"]), r["regex"]) for r in rules if r.get("action") == action} - - # Everything another job keeps, the substrate job drops on the same terms. - for job in ("modelplane-engines", "modelplane-gateway", "modelplane-gpu"): - for kept in predicate(jobs[job], "keep"): - if kept[0] == ("__meta_kubernetes_pod_container_port_name",): - continue # a port filter, not a pod filter - self.assertIn(kept, predicate(substrate, "drop"), f"{job} keeps {kept}, substrate does not drop it") - - def test_only_the_identity_survives_to_the_exporter(self) -> None: - """Discovery attaches the pod's name and uid; neither is the deployment's.""" - blocks = _config()["processors"]["transform/identity"]["metric_statements"] - statement = blocks[0]["statements"][0] - # OTTL quotes with double quotes. A Python list renders single ones and - # the collector refuses to start, which a unit test on shape won't catch. - self.assertNotIn("'", statement) - self.assertIn('keep_keys(resource.attributes, ["cluster"', statement) - pipeline = _config()["service"]["pipelines"]["metrics"]["processors"] - self.assertLess(pipeline.index("transform/identity"), pipeline.index("groupbyattrs/identity")) - - def test_the_identity_is_lifted_onto_the_resource(self) -> None: - """Without this a series arrives carrying only the cluster. - - Discovery writes the identity onto each datapoint. An exporter that - flattens a series into labels reads the resource, so something has to - move it, and this is the only processor that does. Removing it as a - no-op strips every series of what says who it belongs to - verified on - a cluster, where the resource came back carrying `cluster` alone. - """ - cfg = _config() - self.assertEqual(cfg["processors"]["groupbyattrs/identity"]["keys"], list(collector._IDENTITY)) - self.assertIn("groupbyattrs/identity", cfg["service"]["pipelines"]["metrics"]["processors"]) - - def test_a_part_is_extracted_before_anything_selects_on_it(self) -> None: - """A label or a unit for an extracted part names a metric that must exist. - - The extraction mints `_count`, and a datapoint statement for it - selects on that name. Run the datapoint block first and it matches - nothing, silently. - """ - mapping = mmv1alpha1.MetricMapping.model_validate( - { - "spec": { - "metrics": [ - { - "from": "my_engine_duration_ms", - "to": "modelplane_requests_total", - "part": "Count", - "fromUnit": "Milliseconds", - "labels": [{"name": "status", "value": "ok"}], - } - ] - } - } - ) - blocks = collector._transform([mapping])["metric_statements"] - contexts = [b["context"] for b in blocks] - self.assertEqual(contexts, ["metric", "metric", "datapoint", "metric"]) - self.assertIn("extract_count_metric", blocks[0]["statements"][0]) - # Everything selecting on the extracted name comes after the extraction. - for block in blocks[1:]: - for statement in block["statements"]: - self.assertIn("my_engine_duration_ms_count", statement) - - def test_every_job_carries_something_unique_to_its_target(self) -> None: - """Two producers whose series are identical are one series, and one is lost. - - The modelplane identity names an engine and nothing else: a gateway pod - carries none of it, two replicas of a substrate controller share a - namespace, and a ModelReplica with copies > 1 runs several pods under - one replica index. - """ - self.assertIn("service.instance.id", collector._IDENTITY) - statement = _config()["processors"]["transform/identity"]["metric_statements"][0]["statements"][0] - self.assertIn('"service.instance.id"', statement) - - def test_a_scrape_spike_cannot_take_the_collector_down(self) -> None: - """Nothing bounds what one interval brings off a fleet of engines.""" - cfg = _config() - self.assertIn("memory_limiter", cfg["processors"]) - self.assertEqual(cfg["service"]["pipelines"]["metrics"]["processors"][0], "memory_limiter") - - def test_the_port_rewrite_matches_an_ipv6_pod(self) -> None: - """__address__ is [2001:db8::1]:9090 there, which [^:]+ never matches.""" - rule = next( - r - for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"] - if j["job_name"] == "modelplane-substrate" - for r in j["relabel_configs"] - if r.get("target_label") == "__address__" - ) - for address in ("10.1.0.5:8000", "[2001:db8::1]:9090"): - matched = re.fullmatch(rule["regex"], f"{address};9402") - assert matched is not None, address - self.assertTrue(matched.expand(r"\1:\2").endswith(":9402")) - - def test_engine_scrape_selects_the_port_by_name(self) -> None: - """Matching by number would find the pd-sidecar on a disaggregated pod.""" - jobs = {j["job_name"]: j for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"]} - keeps = [r for r in jobs["modelplane-engines"]["relabel_configs"] if r.get("action") == "keep"] - self.assertIn("__meta_kubernetes_pod_container_port_name", [k["source_labels"][0] for k in keeps]) - - def test_gateway_has_a_target_of_its_own(self) -> None: - """Its GenAI metrics are on the ext-proc sidecar, not the proxy's port.""" - jobs = [j["job_name"] for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"]] - self.assertIn("modelplane-gateway", jobs) - - def test_both_spellings_of_remote_write_keep_their_identity(self) -> None: - """The exporter registers as prometheus_remote_write in 0.161.0. - - prometheusremotewrite is the older name it still answers to. A sink - writing the one the collector's own documentation gives would - otherwise match no default here and export every series stripped of - the cluster, deployment, engine and role it belongs to - silently, - because the sink itself works. - """ - for type_ in ("prometheus_remote_write", "prometheusremotewrite", "prometheus"): - with self.subTest(type=type_): - exporters = _config(sinks=[_sink(type_=type_)])["exporters"] - exporter = next(v for k, v in exporters.items() if k.startswith(f"{type_}/")) - self.assertTrue(exporter["resource_to_telemetry_conversion"]["enabled"]) - - def test_extensions_are_declared_to_the_service(self) -> None: - """An authenticator the service doesn't list is one the collector won't load.""" - self.assertEqual(_config()["service"]["extensions"], ["oauth2client/acme"]) - self.assertNotIn("extensions", _config(extensions={})["service"]) - - def test_energy_is_scaled_before_it_is_renamed(self) -> None: - """DCGM counts millijoules, and the name says joules. - - The scale is a block ahead of the renames, not a line ahead. The - processor finishes a block over every metric before the next one - starts, so a rename sharing the block would strand every metric - after the first at millijoules. - """ - blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] - self.assertEqual([b["context"] for b in blocks], ["metric", "metric"]) - self.assertTrue(any("scale_metric(0.001)" in st for st in blocks[0]["statements"])) - self.assertTrue(any("modelplane_energy_joules_total" in st for st in blocks[-1]["statements"])) - - def test_a_conversion_reaches_a_histogram_bucket(self) -> None: - """Setting value_double converts a gauge and leaves a histogram lying. - - A histogram holds its measurements in its sum, its minimum and maximum - and every bucket boundary, none of which is value_double. Renaming one - to seconds with its buckets still at milliseconds puts every quantile - a thousand times out, and nothing says so. - """ - blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] - for block in blocks: - for statement in block["statements"]: - self.assertNotIn("value_double", statement) - scales = next(b for b in blocks if any("scale_metric" in st for st in b["statements"])) - self.assertEqual(scales["context"], "metric") - - def test_every_conversion_factor_is_a_float_literal(self) -> None: - """scale_metric takes a float, and 1048576 is an integer to OTTL. - - The collector refuses to start on it - "must be a float" - which - takes the whole cluster's telemetry down, and nothing short of - running the collector catches it. - """ - for unit, factor in collector._UNIT_FACTOR.items(): - with self.subTest(unit=unit): - self.assertIn(".", factor, "an OTTL float literal needs a decimal point") - float(factor) - - def test_a_conversion_cannot_drop_the_batch_it_rides_in(self) -> None: - """scale_metric refuses an exponential histogram. - - Under the default error mode that one refusal fails the whole batch: - every metric from every pod in the scrape is lost, not the one it - could not convert. - """ - blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] - scales = next(b for b in blocks if any("scale_metric" in st for st in b["statements"])) - self.assertEqual(scales["error_mode"], "ignore") - - def test_dcgm_units_are_converted_to_the_unit_the_name_claims(self) -> None: - """DCGM reports mJ and MiB; the names say joules and bytes.""" - blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] - scales = " ".join(blocks[0]["statements"]) - self.assertIn("DCGM_FI_DEV_TOTAL_ENERGY_CONSUMPTION", scales) - self.assertIn("DCGM_FI_DEV_FB_USED", scales) - - def test_every_unit_the_api_offers_has_a_conversion(self) -> None: - """A unit the API accepts with no conversion here is a KeyError at render time.""" - annotation = mmv1alpha1.Metric.model_fields["fromUnit"].annotation - literal = next(a for a in typing.get_args(annotation) if typing.get_origin(a) is typing.Literal) - self.assertEqual(set(typing.get_args(literal)), set(collector._UNIT_FACTOR)) - - def test_a_percentage_is_divided_into_a_ratio(self) -> None: - """A component counting 0 to 100 under a name that says a ratio is 100x out. - - vLLM and SGLang both publish a fraction, so no built-in needs this, but - vLLM's is called kv_cache_usage_perc - the name is no guide, and an - engine that means it has to be able to say so. - """ - mapping = mmv1alpha1.MetricMapping.model_validate( - { - "spec": { - "metrics": [ - { - "from": "my_engine_cache_percent", - "to": "modelplane_kv_cache_utilization_ratio", - "fromUnit": "Percent", - } - ] - } - } - ) - _, scale, _, _ = collector.statements([mapping]) - self.assertEqual(scale, ['scale_metric(0.01) where metric.name == "my_engine_cache_percent"']) - - def test_a_metric_name_cannot_end_the_comparison_early(self) -> None: - """A quote in `from` would rename whatever the rest of the line matched.""" - with self.assertRaises(ValidationError): - mmv1alpha1.Metric.model_validate({"from": 'x" or true or name == "y', "to": "modelplane_x"}) - for mapping in stacks.BUILTIN_MAPPINGS: - for m in mapping.spec.metrics: - round_tripped = mmv1alpha1.Metric.model_validate({"from": m.from_, "to": m.to}) - self.assertEqual(round_tripped.from_, m.from_) - - def test_a_label_value_cannot_end_the_string_it_sits_in(self) -> None: - """`from` is pattern-constrained; a label's value cannot be. - - A value and a `values` remap carry whatever vocabulary the component - already writes, so the schema has to take free text. A quote in one - would close the OTTL literal early and leave the remainder of the - value as OTTL - at best the collector refuses to start. - """ - mapping = mmv1alpha1.MetricMapping.model_validate( - { - "spec": { - "metrics": [ - { - "from": "my_engine_finish", - "to": "modelplane_requests_total", - "labels": [{"name": "reason", "from": "finish", "values": {'ab"c': 'x"y'}}], - } - ] - } +def test_only_modelplane_leaves_the_cluster() -> None: + """A series the statements didn't rename is dropped.""" + assert "filter/modelplane" in _config()["processors"] + assert "filter/modelplane" in _config()["service"]["pipelines"]["metrics"]["processors"] + + +def test_cluster_is_stamped_here() -> None: + """One receiver downstream sees a merged stream and can't tell senders apart.""" + attrs = _config()["processors"]["resource/cluster"]["attributes"] + assert attrs == [{"key": "cluster", "value": "prod-us-east", "action": "upsert"}] + + +def test_the_jobs_cover_disjoint_pods() -> None: + """A pod two jobs both collect arrives twice, under two job names.""" + jobs = { + j["job_name"]: j["relabel_configs"] for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"] + } + substrate = jobs["modelplane-substrate"] + + def predicate(rules: list[dict], action: str) -> set[tuple]: + return {(tuple(r["source_labels"]), r["regex"]) for r in rules if r.get("action") == action} + + # Everything another job keeps, the substrate job drops on the same terms. + for job in ("modelplane-engines", "modelplane-gateway", "modelplane-gpu"): + for kept in predicate(jobs[job], "keep"): + if kept[0] == ("__meta_kubernetes_pod_container_port_name",): + continue # a port filter, not a pod filter + assert kept in predicate(substrate, "drop"), f"{job} keeps {kept}, substrate does not drop it" + + +def test_only_the_identity_survives_to_the_exporter() -> None: + """Discovery attaches the pod's name and uid; neither is the deployment's.""" + blocks = _config()["processors"]["transform/identity"]["metric_statements"] + statement = blocks[0]["statements"][0] + # OTTL quotes with double quotes. A Python list renders single ones and + # the collector refuses to start, which a unit test on shape won't catch. + assert "'" not in statement + assert 'keep_keys(resource.attributes, ["cluster"' in statement + pipeline = _config()["service"]["pipelines"]["metrics"]["processors"] + assert pipeline.index("transform/identity") < pipeline.index("groupbyattrs/identity") + + +def test_the_identity_is_lifted_onto_the_resource() -> None: + """Without this a series arrives carrying only the cluster. + + Discovery writes the identity onto each datapoint. An exporter that + flattens a series into labels reads the resource, so something has to + move it, and this is the only processor that does. Removing it as a + no-op strips every series of what says who it belongs to - verified on + a cluster, where the resource came back carrying `cluster` alone. + """ + cfg = _config() + assert cfg["processors"]["groupbyattrs/identity"]["keys"] == list(collector._IDENTITY) + assert "groupbyattrs/identity" in cfg["service"]["pipelines"]["metrics"]["processors"] + + +def test_a_part_is_extracted_before_anything_selects_on_it() -> None: + """A label or a unit for an extracted part names a metric that must exist. + + The extraction mints `_count`, and a datapoint statement for it + selects on that name. Run the datapoint block first and it matches + nothing, silently. + """ + mapping = mmv1alpha1.MetricMapping.model_validate( + { + "spec": { + "metrics": [ + { + "from": "my_engine_duration_ms", + "to": "modelplane_requests_total", + "part": "Count", + "fromUnit": "Milliseconds", + "labels": [{"name": "status", "value": "ok"}], + } + ] } - ) - _, _, datapoint, _ = collector.statements([mapping]) - joined = " ".join(datapoint) - self.assertIn(r'"ab\"c"', joined) - self.assertIn(r'"x\"y"', joined) - - def test_carrying_a_label_onto_itself_keeps_it(self) -> None: - """`from` equal to `name` is how a mapping remaps values in place. - - The delete that stops a carried label costing twice the cardinality - would otherwise take the label the statements before it just set, and - the series would lose the label entirely. - """ - mapping = mmv1alpha1.MetricMapping.model_validate( - { - "spec": { - "metrics": [ - { - "from": "my_engine_finish", - "to": "modelplane_requests_total", - "labels": [{"name": "reason", "from": "reason", "values": {"eos": "stop"}}], - } - ] - } + } + ) + blocks = collector._transform([mapping])["metric_statements"] + contexts = [b["context"] for b in blocks] + assert contexts == ["metric", "metric", "datapoint", "metric"] + assert "extract_count_metric" in blocks[0]["statements"][0] + # Everything selecting on the extracted name comes after the extraction. + for block in blocks[1:]: + for statement in block["statements"]: + assert "my_engine_duration_ms_count" in statement + + +def test_every_job_carries_something_unique_to_its_target() -> None: + """Two producers whose series are identical are one series, and one is lost. + + The modelplane identity names an engine and nothing else: a gateway pod + carries none of it, two replicas of a substrate controller share a + namespace, and a ModelReplica with copies > 1 runs several pods under + one replica index. + """ + assert "service.instance.id" in collector._IDENTITY + statement = _config()["processors"]["transform/identity"]["metric_statements"][0]["statements"][0] + assert '"service.instance.id"' in statement + + +def test_a_scrape_spike_cannot_take_the_collector_down() -> None: + """Nothing bounds what one interval brings off a fleet of engines.""" + cfg = _config() + assert "memory_limiter" in cfg["processors"] + assert cfg["service"]["pipelines"]["metrics"]["processors"][0] == "memory_limiter" + + +def test_the_port_rewrite_matches_an_ipv6_pod() -> None: + """__address__ is [2001:db8::1]:9090 there, which [^:]+ never matches.""" + rule = next( + r + for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"] + if j["job_name"] == "modelplane-substrate" + for r in j["relabel_configs"] + if r.get("target_label") == "__address__" + ) + for address in ("10.1.0.5:8000", "[2001:db8::1]:9090"): + matched = re.fullmatch(rule["regex"], f"{address};9402") + assert matched is not None, address + assert matched.expand(r"\1:\2").endswith(":9402") + + +def test_engine_scrape_selects_the_port_by_name() -> None: + """Matching by number would find the pd-sidecar on a disaggregated pod.""" + jobs = {j["job_name"]: j for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"]} + keeps = [r for r in jobs["modelplane-engines"]["relabel_configs"] if r.get("action") == "keep"] + assert "__meta_kubernetes_pod_container_port_name" in [k["source_labels"][0] for k in keeps] + + +def test_gateway_has_a_target_of_its_own() -> None: + """Its GenAI metrics are on the ext-proc sidecar, not the proxy's port.""" + jobs = [j["job_name"] for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"]] + assert "modelplane-gateway" in jobs + + +@pytest.mark.parametrize("type_", ["prometheus_remote_write", "prometheusremotewrite", "prometheus"]) +def test_both_spellings_of_remote_write_keep_their_identity(type_: str) -> None: + """The exporter registers as prometheus_remote_write in 0.161.0. + + prometheusremotewrite is the older name it still answers to. A sink + writing the one the collector's own documentation gives would + otherwise match no default here and export every series stripped of + the cluster, deployment, engine and role it belongs to - silently, + because the sink itself works. + """ + exporters = _config(sinks=[_sink(type_=type_)])["exporters"] + exporter = next(v for k, v in exporters.items() if k.startswith(f"{type_}/")) + assert exporter["resource_to_telemetry_conversion"]["enabled"] + + +def test_extensions_are_declared_to_the_service() -> None: + """An authenticator the service doesn't list is one the collector won't load.""" + assert _config()["service"]["extensions"] == ["oauth2client/acme"] + assert "extensions" not in _config(extensions={})["service"] + + +def test_energy_is_scaled_before_it_is_renamed() -> None: + """DCGM counts millijoules, and the name says joules. + + The scale is a block ahead of the renames, not a line ahead. The + processor finishes a block over every metric before the next one + starts, so a rename sharing the block would strand every metric + after the first at millijoules. + """ + blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] + assert [b["context"] for b in blocks] == ["metric", "metric"] + assert any("scale_metric(0.001)" in st for st in blocks[0]["statements"]) + assert any("modelplane_energy_joules_total" in st for st in blocks[-1]["statements"]) + + +def test_a_conversion_reaches_a_histogram_bucket() -> None: + """Setting value_double converts a gauge and leaves a histogram lying. + + A histogram holds its measurements in its sum, its minimum and maximum + and every bucket boundary, none of which is value_double. Renaming one + to seconds with its buckets still at milliseconds puts every quantile + a thousand times out, and nothing says so. + """ + blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] + for block in blocks: + for statement in block["statements"]: + assert "value_double" not in statement + scales = next(b for b in blocks if any("scale_metric" in st for st in b["statements"])) + assert scales["context"] == "metric" + + +def test_every_conversion_factor_is_a_float_literal() -> None: + """scale_metric takes a float, and 1048576 is an integer to OTTL. + + The collector refuses to start on it - "must be a float" - which + takes the whole cluster's telemetry down, and nothing short of + running the collector catches it. + """ + for unit, factor in collector._UNIT_FACTOR.items(): + assert "." in factor, f"{unit}: an OTTL float literal needs a decimal point" + float(factor) + + +def test_a_conversion_cannot_drop_the_batch_it_rides_in() -> None: + """scale_metric refuses an exponential histogram. + + Under the default error mode that one refusal fails the whole batch: + every metric from every pod in the scrape is lost, not the one it + could not convert. + """ + blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] + scales = next(b for b in blocks if any("scale_metric" in st for st in b["statements"])) + assert scales["error_mode"] == "ignore" + + +def test_dcgm_units_are_converted_to_the_unit_the_name_claims() -> None: + """DCGM reports mJ and MiB; the names say joules and bytes.""" + blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] + scales = " ".join(blocks[0]["statements"]) + assert "DCGM_FI_DEV_TOTAL_ENERGY_CONSUMPTION" in scales + assert "DCGM_FI_DEV_FB_USED" in scales + + +def test_every_unit_the_api_offers_has_a_conversion() -> None: + """A unit the API accepts with no conversion here is a KeyError at render time.""" + annotation = mmv1alpha1.Metric.model_fields["fromUnit"].annotation + literal = next(a for a in typing.get_args(annotation) if typing.get_origin(a) is typing.Literal) + assert set(typing.get_args(literal)) == set(collector._UNIT_FACTOR) + + +def test_a_percentage_is_divided_into_a_ratio() -> None: + """A component counting 0 to 100 under a name that says a ratio is 100x out. + + vLLM and SGLang both publish a fraction, so no built-in needs this, but + vLLM's is called kv_cache_usage_perc - the name is no guide, and an + engine that means it has to be able to say so. + """ + mapping = mmv1alpha1.MetricMapping.model_validate( + { + "spec": { + "metrics": [ + { + "from": "my_engine_cache_percent", + "to": "modelplane_kv_cache_utilization_ratio", + "fromUnit": "Percent", + } + ] } - ) - _, _, datapoint, _ = collector.statements([mapping]) - self.assertFalse([st for st in datapoint if st.startswith("delete_key")]) - - def test_a_value_rewrite_never_lands_in_the_metric_context(self) -> None: - """value_double is a datapoint path; the collector refuses to start on it here.""" - blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] - metric_block = next(b for b in blocks if b["context"] == "metric") - self.assertFalse([st for st in metric_block["statements"] if "value_double" in st]) - for block in blocks: - for st in block["statements"]: - self.assertNotIn("set(name,", st) - self.assertNotIn("set(value_double,", st) - - def test_sglang_carries_no_queue_time_or_preemption(self) -> None: - """SGLang publishes neither, so there is nothing to rename onto them. - - Checked against a running SGLang v0.4.9.post2: it has no per-request - queue-time metric and no retraction counters at all. The nearest - thing, sglang:avg_request_queue_latency, is a gauge of the mean over - the last batch - a different measurement from vLLM's per-request - histogram, and one name holding both makes a fleet quantile - meaningless. - """ - sglang = { - m.from_: m.to - for mapping in stacks.BUILTIN_MAPPINGS - for m in mapping.spec.metrics - if m.from_.startswith("sglang:") } - self.assertTrue(sglang, "the SGLang built-in went missing") - self.assertNotIn("modelplane_request_queue_seconds", sglang.values()) - self.assertNotIn("modelplane_requests_preempted_total", sglang.values()) - self.assertFalse([k for k in sglang if "retracted" in k or "queue_time" in k]) + ) + _, scale, _, _ = collector.statements([mapping]) + assert scale == ['scale_metric(0.01) where metric.name == "my_engine_cache_percent"'] + - def test_sglang_latency_histograms_are_not_renamed(self) -> None: - """Their buckets resolve to 100ms where vLLM's resolve to 1ms.""" - joined = " ".join(_metric_statements()) - self.assertNotIn("sglang:time_to_first_token_seconds", joined) - self.assertNotIn("sglang:inter_token_latency", joined) +def test_a_metric_name_cannot_end_the_comparison_early() -> None: + """A quote in `from` would rename whatever the rest of the line matched.""" + with pytest.raises(ValidationError, match="String should match pattern"): + mmv1alpha1.Metric.model_validate({"from": 'x" or true or name == "y', "to": "modelplane_x"}) + for mapping in stacks.BUILTIN_MAPPINGS: + for m in mapping.spec.metrics: + round_tripped = mmv1alpha1.Metric.model_validate({"from": m.from_, "to": m.to}) + assert round_tripped.from_ == m.from_ -class TestObjects(unittest.TestCase): - """The manifests this composes.""" +def test_a_label_value_cannot_end_the_string_it_sits_in() -> None: + """`from` is pattern-constrained; a label's value cannot be. - def _objects(self, secret: str | None = None) -> dict: - sinks = [_sink(secret=secret)] if secret else _SINKS - return { - k: m for k, m, _ in collector.objects("prod-us-east", list(stacks.BUILTIN_MAPPINGS), sinks, _EXTENSIONS) + A value and a `values` remap carry whatever vocabulary the component + already writes, so the schema has to take free text. A quote in one + would close the OTTL literal early and leave the remainder of the + value as OTTL - at best the collector refuses to start. + """ + mapping = mmv1alpha1.MetricMapping.model_validate( + { + "spec": { + "metrics": [ + { + "from": "my_engine_finish", + "to": "modelplane_requests_total", + "labels": [{"name": "reason", "from": "finish", "values": {'ab"c': 'x"y'}}], + } + ] + } } + ) + _, _, datapoint, _ = collector.statements([mapping]) + joined = " ".join(datapoint) + assert r'"ab\"c"' in joined + assert r'"x\"y"' in joined - def test_config_hash_is_stable_across_processes(self) -> None: - """hash() is seeded per process, so it would redeploy on every reconcile.""" - first = self._objects()["collector"]["spec"]["template"]["metadata"]["annotations"] - second = self._objects()["collector"]["spec"]["template"]["metadata"]["annotations"] - self.assertEqual(first, second) - self.assertRegex(first["modelplane.ai/config-hash"], r"^[0-9a-f]{16}$") - - def test_credentials_mount_as_a_file_and_an_environment_variable(self) -> None: - """A rotated token in an environment variable needs a restart to be read.""" - pod = self._objects(secret="telemetry-credentials")["collector"]["spec"]["template"]["spec"] - self.assertIn("credentials-primary", [v["name"] for v in pod["volumes"]]) - self.assertEqual(pod["containers"][0]["envFrom"], [{"secretRef": {"name": "telemetry-credentials"}}]) - - def test_each_sink_gets_its_own_credential_directory(self) -> None: - """Two sinks can both hold a key called token, and neither reads the other's.""" - sinks = [ - _sink(name="vendor", secret="vendor-token"), - _sink(name="prometheus", type_="prometheusremotewrite", secret="prom-token"), - ] - pod = { - k: m for k, m, _ in collector.objects("prod-us-east", list(stacks.BUILTIN_MAPPINGS), sinks, _EXTENSIONS) - }["collector"]["spec"]["template"]["spec"] - mounts = {m["name"]: m["mountPath"] for m in pod["containers"][0]["volumeMounts"]} - self.assertEqual(mounts["credentials-vendor"], "/etc/modelplane/telemetry/vendor") - self.assertEqual(mounts["credentials-prometheus"], "/etc/modelplane/telemetry/prometheus") - - def test_two_sinks_of_one_type_do_not_collide(self) -> None: - """The collector names a second instance of a component /.""" - rendered = collector.exporters([_sink(name="a"), _sink(name="b")]) - self.assertEqual(sorted(rendered), ["otlphttp/a", "otlphttp/b"]) - - def test_a_sink_that_addresses_its_destination_another_way(self) -> None: - """Kafka takes brokers, the debug exporter nothing; neither has an endpoint.""" - sinks = [ - tdv1alpha1.Sink.model_validate( - {"name": "bus", "type": "kafka", "config": {"brokers": ["kafka.acme.example:9092"]}} - ), - tdv1alpha1.Sink.model_validate({"name": "seen", "type": "debug"}), - ] - rendered = collector.exporters(sinks) - self.assertNotIn("endpoint", rendered["kafka/bus"]) - self.assertEqual(rendered["kafka/bus"]["brokers"], ["kafka.acme.example:9092"]) - self.assertEqual(rendered["debug/seen"], {}) - - def test_auth_composes_its_own_authenticator(self) -> None: - """The collector carries no credential on an exporter, only a reference.""" - sink = _sink(secret="telemetry-credentials") - self.assertEqual( - collector.authenticators([sink]), - {"bearertokenauth/primary": {"filename": "/etc/modelplane/telemetry/primary/token"}}, - ) - self.assertEqual( - collector.exporters([sink])["otlphttp/primary"]["auth"], - {"authenticator": "bearertokenauth/primary"}, - ) - def test_a_sinks_own_config_cannot_redirect_it(self) -> None: - """The endpoint is Modelplane's, and goes on after the operator's config.""" - sink = tdv1alpha1.Sink.model_validate( - { - "name": "primary", - "type": "otlphttp", - "endpoint": "https://otel.acme.example", - "config": {"endpoint": "https://elsewhere.example", "compression": "gzip"}, +def test_carrying_a_label_onto_itself_keeps_it() -> None: + """`from` equal to `name` is how a mapping remaps values in place. + + The delete that stops a carried label costing twice the cardinality + would otherwise take the label the statements before it just set, and + the series would lose the label entirely. + """ + mapping = mmv1alpha1.MetricMapping.model_validate( + { + "spec": { + "metrics": [ + { + "from": "my_engine_finish", + "to": "modelplane_requests_total", + "labels": [{"name": "reason", "from": "reason", "values": {"eos": "stop"}}], + } + ] } - ) - rendered = collector.exporters([sink])["otlphttp/primary"] - self.assertEqual(rendered["endpoint"], "https://otel.acme.example") - self.assertEqual(rendered["compression"], "gzip") - - def test_no_secret_mounts_nothing(self) -> None: - pod = self._objects()["collector"]["spec"]["template"]["spec"] - self.assertEqual([v["name"] for v in pod["volumes"]], ["config"]) - self.assertNotIn("envFrom", pod["containers"][0]) - - def test_rbac_is_read_only(self) -> None: - """Service discovery needs to list pods, and nothing needs to write.""" - rules = self._objects()["collector-clusterrole"]["rules"] - verbs = {v for r in rules for v in r["verbs"]} - self.assertEqual(verbs, {"get", "list", "watch"}) + } + ) + _, _, datapoint, _ = collector.statements([mapping]) + assert not [st for st in datapoint if st.startswith("delete_key")] + + +def test_a_value_rewrite_never_lands_in_the_metric_context() -> None: + """value_double is a datapoint path; the collector refuses to start on it here.""" + blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] + metric_block = next(b for b in blocks if b["context"] == "metric") + assert not [st for st in metric_block["statements"] if "value_double" in st] + for block in blocks: + for st in block["statements"]: + assert "set(name," not in st + assert "set(value_double," not in st + + +def test_sglang_carries_no_queue_time_or_preemption() -> None: + """SGLang publishes neither, so there is nothing to rename onto them. + + Checked against a running SGLang v0.4.9.post2: it has no per-request + queue-time metric and no retraction counters at all. The nearest + thing, sglang:avg_request_queue_latency, is a gauge of the mean over + the last batch - a different measurement from vLLM's per-request + histogram, and one name holding both makes a fleet quantile + meaningless. + """ + sglang = { + m.from_: m.to + for mapping in stacks.BUILTIN_MAPPINGS + for m in mapping.spec.metrics + if m.from_.startswith("sglang:") + } + assert sglang, "the SGLang built-in went missing" + assert "modelplane_request_queue_seconds" not in sglang.values() + assert "modelplane_requests_preempted_total" not in sglang.values() + assert not [k for k in sglang if "retracted" in k or "queue_time" in k] + + +def test_sglang_latency_histograms_are_not_renamed() -> None: + """Their buckets resolve to 100ms where vLLM's resolve to 1ms.""" + joined = " ".join(_metric_statements()) + assert "sglang:time_to_first_token_seconds" not in joined + assert "sglang:inter_token_latency" not in joined + + +def test_config_hash_is_stable_across_processes() -> None: + """hash() is seeded per process, so it would redeploy on every reconcile.""" + first = _objects()["collector"]["spec"]["template"]["metadata"]["annotations"] + second = _objects()["collector"]["spec"]["template"]["metadata"]["annotations"] + assert first == second + assert re.search(r"^[0-9a-f]{16}$", first["modelplane.ai/config-hash"]) + + +def test_credentials_mount_as_a_file_and_an_environment_variable() -> None: + """A rotated token in an environment variable needs a restart to be read.""" + pod = _objects(secret="telemetry-credentials")["collector"]["spec"]["template"]["spec"] + assert "credentials-primary" in [v["name"] for v in pod["volumes"]] + assert pod["containers"][0]["envFrom"] == [{"secretRef": {"name": "telemetry-credentials"}}] + + +def test_each_sink_gets_its_own_credential_directory() -> None: + """Two sinks can both hold a key called token, and neither reads the other's.""" + sinks = [ + _sink(name="vendor", secret="vendor-token"), + _sink(name="prometheus", type_="prometheusremotewrite", secret="prom-token"), + ] + pod = {k: m for k, m, _ in collector.objects("prod-us-east", list(stacks.BUILTIN_MAPPINGS), sinks, _EXTENSIONS)}[ + "collector" + ]["spec"]["template"]["spec"] + mounts = {m["name"]: m["mountPath"] for m in pod["containers"][0]["volumeMounts"]} + assert mounts["credentials-vendor"] == "/etc/modelplane/telemetry/vendor" + assert mounts["credentials-prometheus"] == "/etc/modelplane/telemetry/prometheus" + + +def test_two_sinks_of_one_type_do_not_collide() -> None: + """The collector names a second instance of a component /.""" + rendered = collector.exporters([_sink(name="a"), _sink(name="b")]) + assert sorted(rendered) == ["otlphttp/a", "otlphttp/b"] + + +def test_a_sink_that_addresses_its_destination_another_way() -> None: + """Kafka takes brokers, the debug exporter nothing; neither has an endpoint.""" + sinks = [ + tdv1alpha1.Sink.model_validate( + {"name": "bus", "type": "kafka", "config": {"brokers": ["kafka.acme.example:9092"]}} + ), + tdv1alpha1.Sink.model_validate({"name": "seen", "type": "debug"}), + ] + rendered = collector.exporters(sinks) + assert "endpoint" not in rendered["kafka/bus"] + assert rendered["kafka/bus"]["brokers"] == ["kafka.acme.example:9092"] + assert rendered["debug/seen"] == {} + + +def test_auth_composes_its_own_authenticator() -> None: + """The collector carries no credential on an exporter, only a reference.""" + sink = _sink(secret="telemetry-credentials") + assert collector.authenticators([sink]) == { + "bearertokenauth/primary": {"filename": "/etc/modelplane/telemetry/primary/token"} + } + assert collector.exporters([sink])["otlphttp/primary"]["auth"] == {"authenticator": "bearertokenauth/primary"} + + +def test_a_sinks_own_config_cannot_redirect_it() -> None: + """The endpoint is Modelplane's, and goes on after the operator's config.""" + sink = tdv1alpha1.Sink.model_validate( + { + "name": "primary", + "type": "otlphttp", + "endpoint": "https://otel.acme.example", + "config": {"endpoint": "https://elsewhere.example", "compression": "gzip"}, + } + ) + rendered = collector.exporters([sink])["otlphttp/primary"] + assert rendered["endpoint"] == "https://otel.acme.example" + assert rendered["compression"] == "gzip" + + +def test_no_secret_mounts_nothing() -> None: + pod = _objects()["collector"]["spec"]["template"]["spec"] + assert [v["name"] for v in pod["volumes"]] == ["config"] + assert "envFrom" not in pod["containers"][0] + + +def test_rbac_is_read_only() -> None: + """Service discovery needs to list pods, and nothing needs to write.""" + rules = _objects()["collector-clusterrole"]["rules"] + verbs = {v for r in rules for v in r["verbs"]} + assert verbs == {"get", "list", "watch"} diff --git a/functions/compose-serving-stack/tests/test_fn.py b/functions/compose-serving-stack/tests/test_fn.py index 70fcbef17..1529e3f2b 100644 --- a/functions/compose-serving-stack/tests/test_fn.py +++ b/functions/compose-serving-stack/tests/test_fn.py @@ -23,17 +23,19 @@ resource - for every cloud and stack, as frozen literals. """ +import asyncio import copy import dataclasses +import json import pathlib -import unittest +import pytest import yaml -from crossplane.function import logging, resource +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.servingstack import v1alpha1 from models.io.crossplane.m.helm.providerconfig import v1beta1 as helmpcv1beta1 @@ -45,11 +47,6 @@ from models.io.crossplane.protection.usage import v1beta1 as usagev1beta1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 - -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - # Precomputed child_name value for test-backend. _PC_NAME = "test-backend-cluster-63fde" @@ -93,26 +90,25 @@ def _crds(filename: str) -> list[dict]: ] -class TestClusterName(unittest.TestCase): - """The name every exported series is stamped with.""" +def _stack(labels: dict[str, str] | None) -> v1alpha1.ServingStack: + return v1alpha1.ServingStack( + metadata=metav1.ObjectMeta(name="local-serving-stack-d4206", labels=labels), + spec=v1alpha1.Spec( + cloud="Existing", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), + ), + ) - def _stack(self, labels: dict[str, str] | None) -> v1alpha1.ServingStack: - return v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="local-serving-stack-d4206", labels=labels), - spec=v1alpha1.Spec( - cloud="Existing", - secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], - gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), - ), - ) - def test_it_is_the_composite_an_operator_named(self) -> None: - """A ServingStack's own name is generated and carries a suffix.""" - self.assertEqual(fn._cluster_name(self._stack({"crossplane.io/composite": "local"})), "local") +def test_it_is_the_composite_an_operator_named() -> None: + """A ServingStack's own name is generated and carries a suffix.""" + assert fn._cluster_name(_stack({"crossplane.io/composite": "local"})) == "local" - def test_it_falls_back_to_the_stack(self) -> None: - """Better a generated name on the series than none at all.""" - self.assertEqual(fn._cluster_name(self._stack(None)), "local-serving-stack-d4206") + +def test_it_falls_back_to_the_stack() -> None: + """Better a generated name on the series than none at all.""" + assert fn._cluster_name(_stack(None)) == "local-serving-stack-d4206" def _request(cloud: str, stack: str, observed: dict | None = None) -> fnv1.RunFunctionRequest: @@ -882,429 +878,415 @@ class Case: want: fnv1.RunFunctionResponse -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - full = _provider_configs() | _EXISTING_DYNAMO_USAGES | _existing_dynamo_stack() - - # Second pass: PCs observed. depends_on gates first creation, so - # only the dependency-free wave renders; each dependent waits for - # its dependency's Ready before it is first created. - dep_gated = { - "envoy-gateway", # -> cert-manager - "ai-gateway", # -> ai-gateway-crds - "gateway-proxy", # -> gateway-namespace - "kai-queue-root", # -> kai-scheduler - "kai-queue", # -> kai-scheduler - "modelexpress-server", # -> modelexpress-crds - "gateway-selfsigned-issuer", # -> cert-manager, gateway-namespace - "trust-manager", # -> gateway-selfsigned-issuer - } - first_wave = {k: v for k, v in full.items() if k not in dep_gated} - - # Third pass: every rendered resource observed Ready (the gateway - # with its address assigned), so everything is marked ready and - # the address lands in the XR status. - rendered = [k for k in _existing_dynamo_stack() if k != "gateway"] - observed_ready = _observed_pcs() - for key in rendered: - observed_ready[key] = fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ) - observed_ready["gateway"] = fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "conditions": [{"type": "Ready", "status": "True"}], - "atProvider": { - "manifest": {"status": {"addresses": [{"type": "IPAddress", "value": "203.0.113.7"}]}}, - }, +def _compose_cases() -> list[Case]: + """The test_compose cases, built from the Existing/Dynamo stack's resources.""" + full = _provider_configs() | _EXISTING_DYNAMO_USAGES | _existing_dynamo_stack() + + # Second pass: PCs observed. depends_on gates first creation, so + # only the dependency-free wave renders; each dependent waits for + # its dependency's Ready before it is first created. + dep_gated = { + "envoy-gateway", # -> cert-manager + "ai-gateway", # -> ai-gateway-crds + "gateway-proxy", # -> gateway-namespace + "kai-queue-root", # -> kai-scheduler + "kai-queue", # -> kai-scheduler + "modelexpress-server", # -> modelexpress-crds + "gateway-selfsigned-issuer", # -> cert-manager, gateway-namespace + "trust-manager", # -> gateway-selfsigned-issuer + } + first_wave = {k: v for k, v in full.items() if k not in dep_gated} + + # Third pass: every rendered resource observed Ready (the gateway + # with its address assigned), so everything is marked ready and + # the address lands in the XR status. + rendered = [k for k in _existing_dynamo_stack() if k != "gateway"] + observed_ready = _observed_pcs() + for key in rendered: + observed_ready[key] = fnv1.Resource( + resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) + ) + observed_ready["gateway"] = fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "conditions": [{"type": "Ready", "status": "True"}], + "atProvider": { + "manifest": {"status": {"addresses": [{"type": "IPAddress", "value": "203.0.113.7"}]}}, }, - } - ) + }, + } ) - # Every component observed Ready; PCs and Usages are ready on arrival. - all_ready = copy.deepcopy(full) - for res in all_ready.values(): - res.ready = fnv1.READY_TRUE - - cases = [ - Case( - name="first pass composes only the provider configs and usages", - req=_request("Existing", "Dynamo"), - # Everything targeting the remote cluster is gated on the - # ProviderConfigs having been observed; Usages reference - # nothing remote and compose immediately. The unready - # ProviderConfigs keep the composite unready until the - # stack actually renders. - want=_response(_provider_configs(ready=False) | _EXISTING_DYNAMO_USAGES), - ), - Case( - name="second pass renders the dependency-free wave", - req=_request("Existing", "Dynamo", observed=_observed_pcs()), - want=_response(first_wave), - ), - Case( - name="all dependencies ready renders the whole stack, marks it ready, and writes the gateway address", - req=_request("Existing", "Dynamo", observed=observed_ready), - want=_response(all_ready, status={"gateway": {"address": "203.0.113.7"}}), - ), - ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) + ) + # Every component observed Ready; PCs and Usages are ready on arrival. + all_ready = copy.deepcopy(full) + for res in all_ready.values(): + res.ready = fnv1.READY_TRUE - async def test_identity_secret_type_flows_to_provider_configs(self) -> None: - """A non-GCP identity secret's type is stamped verbatim on both - ProviderConfigs rather than being forced to GoogleApplicationCredentials, - and its own namespace wins over the XR's.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Nebius", - secrets=[ - v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), - v1alpha1.Secret( - type="NebiusServiceAccountCredentials", - name="nebius-secret", - key="credentials.json", - namespace="other-ns", - ), - ], - gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), - ), - ).model_dump(exclude_none=True, mode="json") - ), + return [ + Case( + name="first pass composes only the provider configs and usages", + req=_request("Existing", "Dynamo"), + # Everything targeting the remote cluster is gated on the + # ProviderConfigs having been observed; Usages reference + # nothing remote and compose immediately. The unready + # ProviderConfigs keep the composite unready until the + # stack actually renders. + want=_response(_provider_configs(ready=False) | _EXISTING_DYNAMO_USAGES), + ), + Case( + name="second pass renders the dependency-free wave", + req=_request("Existing", "Dynamo", observed=_observed_pcs()), + want=_response(first_wave), + ), + Case( + name="all dependencies ready renders the whole stack, marks it ready, and writes the gateway address", + req=_request("Existing", "Dynamo", observed=observed_ready), + want=_response(all_ready, status={"gateway": {"address": "203.0.113.7"}}), + ), + ] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes the Existing/Dynamo stack across the reconcile passes.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) + + +def test_identity_secret_type_flows_to_provider_configs() -> None: + """A non-GCP identity secret's type and namespace reach both ProviderConfigs verbatim.""" + # The type is stamped as is rather than being forced to + # GoogleApplicationCredentials, and the secret's own namespace wins over + # the XR's. + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ServingStack( + metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), + spec=v1alpha1.Spec( + cloud="Nebius", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret( + type="NebiusServiceAccountCredentials", + name="nebius-secret", + key="credentials.json", + namespace="other-ns", + ), + ], + gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), + ), + ).model_dump(exclude_none=True, mode="json") ), ), + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + pc = resource.struct_to_dict(got.desired.resources["provider-config-kubernetes"].resource) + assert pc["spec"]["identity"]["type"] == "NebiusServiceAccountCredentials" + assert pc["spec"]["identity"]["secretRef"]["namespace"] == "other-ns" + helm_pc = resource.struct_to_dict(got.desired.resources["provider-config-helm"].resource) + assert helm_pc["spec"]["identity"]["type"] == "NebiusServiceAccountCredentials" + + +def test_gpu_pool_nvlink_disable_flows_to_components() -> None: + """A Civo ServingStack whose spec.gpu flags a pool for NVLink + disable composes the gpu-operator release in NVIDIADriver-CRD + mode, the kernel module ConfigMap, and a per-pool NVIDIADriver + selecting that pool's nodes - and only that pool's.""" + # The install gate composes a component once its dependencies are + # observed Ready; observe the chain up to the per-pool driver. + observed = _observed_pcs() + for key in ( + "cert-manager", + "node-feature-discovery", + "gpu-operator", + "nvlink-disable-config-gpu-operator", + "nvlink-disable-config-nvidia-kernel-config", + ): + observed[key] = fnv1.Resource( + resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) ) - got = await self.runner.RunFunction(req, None) - pc = resource.struct_to_dict(got.desired.resources["provider-config-kubernetes"].resource) - self.assertEqual("NebiusServiceAccountCredentials", pc["spec"]["identity"]["type"]) - self.assertEqual("other-ns", pc["spec"]["identity"]["secretRef"]["namespace"]) - helm_pc = resource.struct_to_dict(got.desired.resources["provider-config-helm"].resource) - self.assertEqual("NebiusServiceAccountCredentials", helm_pc["spec"]["identity"]["type"]) - - async def test_gpu_pool_nvlink_disable_flows_to_components(self) -> None: - """A Civo ServingStack whose spec.gpu flags a pool for NVLink - disable composes the gpu-operator release in NVIDIADriver-CRD - mode, the kernel module ConfigMap, and a per-pool NVIDIADriver - selecting that pool's nodes - and only that pool's.""" - # The install gate composes a component once its dependencies are - # observed Ready; observe the chain up to the per-pool driver. - observed = _observed_pcs() - for key in ( - "cert-manager", - "node-feature-discovery", - "gpu-operator", - "nvlink-disable-config-gpu-operator", - "nvlink-disable-config-nvidia-kernel-config", - ): - observed[key] = fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ) - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Civo", - secrets=[ - v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), - ], - gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), - gpu=v1alpha1.Gpu( - pools=[v1alpha1.Pool(name="h100-pool", disableNvLink=True)], - ), + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ServingStack( + metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), + spec=v1alpha1.Spec( + cloud="Civo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + ], + gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), + gpu=v1alpha1.Gpu( + pools=[v1alpha1.Pool(name="h100-pool", disableNvLink=True)], ), - ).model_dump(exclude_none=True, mode="json") - ), + ), + ).model_dump(exclude_none=True, mode="json") ), - resources=observed, ), - ) - got = await self.runner.RunFunction(req, None) + resources=observed, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - release = resource.struct_to_dict(got.desired.resources["gpu-operator"].resource) - self.assertEqual( - {"enabled": True, "deployDefaultCR": True}, - release["spec"]["forProvider"]["values"]["driver"]["nvidiaDriverCRD"], - ) + release = resource.struct_to_dict(got.desired.resources["gpu-operator"].resource) + assert release["spec"]["forProvider"]["values"]["driver"]["nvidiaDriverCRD"] == { + "enabled": True, + "deployDefaultCR": True, + } - config = resource.struct_to_dict(got.desired.resources["nvlink-disable-config-nvidia-kernel-config"].resource) - self.assertEqual( - {"nvidia.conf": "options nvidia NVreg_NvLinkDisable=1"}, - config["spec"]["forProvider"]["manifest"]["data"], - ) + config = resource.struct_to_dict(got.desired.resources["nvlink-disable-config-nvidia-kernel-config"].resource) + assert config["spec"]["forProvider"]["manifest"]["data"] == {"nvidia.conf": "options nvidia NVreg_NvLinkDisable=1"} - driver = resource.struct_to_dict(got.desired.resources["nvlink-disabled-driver-h100-pool"].resource) - manifest = driver["spec"]["forProvider"]["manifest"] - self.assertEqual("NVIDIADriver", manifest["kind"]) - self.assertEqual({"modelplane.ai/pool": "h100-pool"}, manifest["spec"]["nodeSelector"]) - self.assertEqual({"name": "nvidia-kernel-config"}, manifest["spec"]["kernelModuleConfig"]) - - # The derived Usages hold the operator release and the ConfigMap - # until the per-pool driver is gone. - self.assertIn("usage-gpu-operator-by-nvlink-disabled-driver-h100-pool", got.desired.resources) - - async def test_gpu_pool_without_nvlink_disable_changes_nothing(self) -> None: - """A Civo ServingStack whose spec.gpu flags no pool composes the - stock component list: ClusterPolicy-managed driver, no NVIDIADriver - or kernel module ConfigMap objects.""" - observed = _observed_pcs() - observed["gpu-operator"] = fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ) - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Civo", - secrets=[ - v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), - ], - gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), - gpu=v1alpha1.Gpu( - pools=[v1alpha1.Pool(name="l40s-pool", disableNvLink=False)], - ), + driver = resource.struct_to_dict(got.desired.resources["nvlink-disabled-driver-h100-pool"].resource) + manifest = driver["spec"]["forProvider"]["manifest"] + assert manifest["kind"] == "NVIDIADriver" + assert manifest["spec"]["nodeSelector"] == {"modelplane.ai/pool": "h100-pool"} + assert manifest["spec"]["kernelModuleConfig"] == {"name": "nvidia-kernel-config"} + + # The derived Usages hold the operator release and the ConfigMap + # until the per-pool driver is gone. + assert "usage-gpu-operator-by-nvlink-disabled-driver-h100-pool" in got.desired.resources + + +def test_gpu_pool_without_nvlink_disable_changes_nothing() -> None: + """A Civo ServingStack whose spec.gpu flags no pool composes the + stock component list: ClusterPolicy-managed driver, no NVIDIADriver + or kernel module ConfigMap objects.""" + observed = _observed_pcs() + observed["gpu-operator"] = fnv1.Resource( + resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) + ) + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ServingStack( + metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), + spec=v1alpha1.Spec( + cloud="Civo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + ], + gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), + gpu=v1alpha1.Gpu( + pools=[v1alpha1.Pool(name="l40s-pool", disableNvLink=False)], ), - ).model_dump(exclude_none=True, mode="json") - ), + ), + ).model_dump(exclude_none=True, mode="json") ), - resources=observed, ), - ) - got = await self.runner.RunFunction(req, None) - - release = resource.struct_to_dict(got.desired.resources["gpu-operator"].resource) - self.assertNotIn("nvidiaDriverCRD", release["spec"]["forProvider"]["values"]["driver"]) - for key in got.desired.resources: - self.assertNotIn("nvlink", key) - - async def test_cluster_gateway_composes_mtls_with_ca(self) -> None: - """A cluster with an InferenceGateway CA serves mTLS: it issues its own - PKI, republishes the CA without its key, demands a client certificate on - its HTTPS listener, and publishes the CA in status. - - The hostname is a full Service FQDN, so the CA certificate's commonName - overflows the 64-byte X.509 limit and is truncated. - """ - hostname = "gateway-test-backend-12345.modelplane-system.svc.cluster.local" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Existing", - stack="Standard", - secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], - gateway=v1alpha1.Gateway( - hostname=hostname, - # Deliberately out of name order, to prove the - # bundle sorts before concatenating. - clientCAs=[ - v1alpha1.ClientCA(name="fleet-b", certificate="BBB"), - v1alpha1.ClientCA(name="fleet-a", certificate="AAA"), - ], - ), + resources=observed, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + release = resource.struct_to_dict(got.desired.resources["gpu-operator"].resource) + assert "nvidiaDriverCRD" not in release["spec"]["forProvider"]["values"]["driver"] + for key in got.desired.resources: + assert "nvlink" not in key + + +def test_cluster_gateway_composes_mtls_with_ca() -> None: + """A cluster with an InferenceGateway CA composes its own PKI and serves mTLS.""" + # It issues its own PKI, republishes the CA without its key, demands a + # client certificate on its HTTPS listener, and publishes the CA in status. + # + # The hostname is a full Service FQDN, so the CA certificate's commonName + # overflows the 64-byte X.509 limit and is truncated. + hostname = "gateway-test-backend-12345.modelplane-system.svc.cluster.local" + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ServingStack( + metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), + spec=v1alpha1.Spec( + cloud="Existing", + stack="Standard", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway( + hostname=hostname, + # Deliberately out of name order, to prove the + # bundle sorts before concatenating. + clientCAs=[ + v1alpha1.ClientCA(name="fleet-b", certificate="BBB"), + v1alpha1.ClientCA(name="fleet-a", certificate="AAA"), + ], ), - ).model_dump(exclude_none=True, mode="json") - ), + ), + ).model_dump(exclude_none=True, mode="json") ), - # PCs observed, the self-signed Issuer Ready (so trust-manager and - # the CA chain proceed), and the CA ConfigMap trust-manager syncs - # carrying the certificate back for status. - resources=_observed_pcs() - | { - "gateway-selfsigned-issuer": fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"conditions": [{"type": "Ready", "status": "True"}]}} - ) - ), - "gateway-ca-configmap": fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}}} - ) - ), - }, ), - ) - got = await self.runner.RunFunction(req, None) + # PCs observed, the self-signed Issuer Ready (so trust-manager and + # the CA chain proceed), and the CA ConfigMap trust-manager syncs + # carrying the certificate back for status. + resources=_observed_pcs() + | { + "gateway-selfsigned-issuer": fnv1.Resource( + resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) + ), + "gateway-ca-configmap": fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}}} + ) + ), + }, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + def manifest(key: str) -> dict: + return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] + + ca_cert = manifest("gateway-ca-certificate") + assert ca_cert["spec"]["commonName"] == "modelplane cluster CA gateway-test-backend-12345.modelplane-syst" + assert len(ca_cert["spec"]["commonName"]) <= 64 + assert ca_cert["spec"]["isCA"] + assert ca_cert["spec"]["issuerRef"]["name"] == "modelplane-selfsigned" + + serving = manifest("gateway-serving-certificate") + assert serving["spec"]["dnsNames"] == [hostname] + assert serving["spec"]["issuerRef"]["name"] == "modelplane-cluster-ca" + + bundle = manifest("gateway-ca-bundle") + assert bundle["apiVersion"] == "trust.cert-manager.io/v1alpha1" + assert bundle["spec"]["sources"] == [{"secret": {"name": "modelplane-cluster-ca", "key": "ca.crt"}}] + + # Observed only, never managed: trust-manager owns the ConfigMap. + ca_cm = got.desired.resources["gateway-ca-configmap"] + assert resource.struct_to_dict(ca_cm.resource)["spec"]["managementPolicies"] == ["Observe"] + + # Every InferenceGateway's CA, sorted by name and concatenated. + client_bundle = manifest("gateway-client-ca-bundle") + assert client_bundle["data"]["ca.crt"] == "AAA\nBBB\n" + + client_auth = manifest("gateway-client-auth") + assert client_auth["kind"] == "ClientTrafficPolicy" + assert client_auth["spec"]["targetRefs"][0]["sectionName"] == "https" + assert ( + client_auth["spec"]["tls"]["clientValidation"]["caCertificateRefs"][0]["name"] + == "modelplane-inference-gateway-cas" + ) - def manifest(key: str) -> dict: - return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] + # One HTTPS listener, terminating TLS with the serving certificate. + gateway = manifest("gateway") + assert gateway["spec"]["listeners"] == [ + { + "name": "https", + "protocol": "HTTPS", + "port": 443, + "hostname": hostname, + "tls": {"mode": "Terminate", "certificateRefs": [{"name": "cluster-gateway-serving"}]}, + "allowedRoutes": { + "namespaces": { + "from": "Selector", + "selector": {"matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}]}, + } + }, + } + ] - ca_cert = manifest("gateway-ca-certificate") - self.assertEqual( - "modelplane cluster CA gateway-test-backend-12345.modelplane-syst", ca_cert["spec"]["commonName"] - ) - self.assertLessEqual(len(ca_cert["spec"]["commonName"]), 64) - self.assertTrue(ca_cert["spec"]["isCA"]) - self.assertEqual("modelplane-selfsigned", ca_cert["spec"]["issuerRef"]["name"]) - - serving = manifest("gateway-serving-certificate") - self.assertEqual([hostname], serving["spec"]["dnsNames"]) - self.assertEqual("modelplane-cluster-ca", serving["spec"]["issuerRef"]["name"]) - - bundle = manifest("gateway-ca-bundle") - self.assertEqual("trust.cert-manager.io/v1alpha1", bundle["apiVersion"]) - self.assertEqual([{"secret": {"name": "modelplane-cluster-ca", "key": "ca.crt"}}], bundle["spec"]["sources"]) - - # Observed only, never managed: trust-manager owns the ConfigMap. - ca_cm = got.desired.resources["gateway-ca-configmap"] - self.assertEqual(["Observe"], resource.struct_to_dict(ca_cm.resource)["spec"]["managementPolicies"]) - - # Every InferenceGateway's CA, sorted by name and concatenated. - client_bundle = manifest("gateway-client-ca-bundle") - self.assertEqual("AAA\nBBB\n", client_bundle["data"]["ca.crt"]) - - client_auth = manifest("gateway-client-auth") - self.assertEqual("ClientTrafficPolicy", client_auth["kind"]) - self.assertEqual("https", client_auth["spec"]["targetRefs"][0]["sectionName"]) - self.assertEqual( - "modelplane-inference-gateway-cas", - client_auth["spec"]["tls"]["clientValidation"]["caCertificateRefs"][0]["name"], - ) + status = resource.struct_to_dict(got.desired.composite.resource)["status"] + assert status["gateway"]["caCertificate"] == "CLUSTERCA" - # One HTTPS listener, terminating TLS with the serving certificate. - gateway = manifest("gateway") - self.assertEqual( - [ + # Every PKI resource must be tracked for readiness: + # compose_gateway_pki marks only the keys it returns, so one composed + # but not returned would silently hold the cluster un-Ready. Observe + # each Ready and assert it's marked ready, which fails if the key was + # dropped from the rendered list. (The self-signed Issuer and + # trust-manager are stack components, covered by the golden test.) + pki_keys = [ + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + ] + for key in pki_keys: + req.observed.resources[key].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) + ) + ) + # Preserve the CA ConfigMap's data alongside its Ready condition. + req.observed.resources["gateway-ca-configmap"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( { - "name": "https", - "protocol": "HTTPS", - "port": 443, - "hostname": hostname, - "tls": {"mode": "Terminate", "certificateRefs": [{"name": "cluster-gateway-serving"}]}, - "allowedRoutes": { - "namespaces": { - "from": "Selector", - "selector": { - "matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}] - }, - } - }, + "status": { + "conditions": [{"type": "Ready", "status": "True"}], + "atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}, + } } - ], - gateway["spec"]["listeners"], + ) ) + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + for key in pki_keys: + assert got.desired.resources[key].ready == fnv1.READY_TRUE, f"{key} not marked ready" - status = resource.struct_to_dict(got.desired.composite.resource)["status"] - self.assertEqual("CLUSTERCA", status["gateway"]["caCertificate"]) - - # Every PKI resource must be tracked for readiness: - # compose_gateway_pki marks only the keys it returns, so one composed - # but not returned would silently hold the cluster un-Ready. Observe - # each Ready and assert it's marked ready, which fails if the key was - # dropped from the rendered list. (The self-signed Issuer and - # trust-manager are stack components, covered by the golden test.) - pki_keys = [ - "gateway-ca-certificate", - "gateway-ca-issuer", - "gateway-serving-certificate", - "gateway-ca-bundle", - "gateway-ca-configmap", - "gateway-client-ca-bundle", - "gateway-client-auth", - ] - for key in pki_keys: - req.observed.resources[key].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ) - ) - # Preserve the CA ConfigMap's data alongside its Ready condition. - req.observed.resources["gateway-ca-configmap"].CopyFrom( - fnv1.Resource( + +def test_cluster_gateway_without_ca_serves_nothing() -> None: + """A cluster with no InferenceGateway CA withholds its Gateway, and warns.""" + # Withholding the Gateway entirely, rather than serving the engines + # unauthenticated. + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( - { - "status": { - "conditions": [{"type": "Ready", "status": "True"}], - "atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}, - } - } - ) - ) - ) - got = await self.runner.RunFunction(req, None) - for key in pki_keys: - self.assertEqual(fnv1.READY_TRUE, got.desired.resources[key].ready, f"{key} not marked ready") - - async def test_cluster_gateway_without_ca_serves_nothing(self) -> None: - """A cluster with no InferenceGateway CA withholds the Gateway entirely - rather than serving the engines unauthenticated, and warns.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Existing", - stack="Standard", - secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], - gateway=v1alpha1.Gateway(hostname="gw.clusters.example.com"), - ), - ).model_dump(exclude_none=True, mode="json") - ), + v1alpha1.ServingStack( + metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), + spec=v1alpha1.Spec( + cloud="Existing", + stack="Standard", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway(hostname="gw.clusters.example.com"), + ), + ).model_dump(exclude_none=True, mode="json") ), - resources=_observed_pcs(), + ), + resources=_observed_pcs(), + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + # The GatewayClass and the cluster's own PKI are composed, so the CA is + # ready to publish when the first InferenceGateway's CA arrives. The + # Gateway, the client CA bundle and the policy demanding a client + # certificate aren't, and nor is the Usage protecting the Gateway. + gateway_keys = {k for k in got.desired.resources if k.startswith(("gateway", "usage-gateway"))} + assert gateway_keys == { + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-namespace", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + } + assert list(got.results) == [ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message=( + "Gateway gw.clusters.example.com not served: no InferenceGateway has published a client " + "CA for this cluster to trust, and serving without one would accept unauthenticated callers" ), ) - got = await self.runner.RunFunction(req, None) - # The GatewayClass and the cluster's own PKI are composed, so the CA is - # ready to publish when the first InferenceGateway's CA arrives. The - # Gateway, the client CA bundle and the policy demanding a client - # certificate aren't, and nor is the Usage protecting the Gateway. - gateway_keys = {k for k in got.desired.resources if k.startswith(("gateway", "usage-gateway"))} - self.assertEqual( - { - "gateway-class", - "gateway-ca-certificate", - "gateway-ca-issuer", - "gateway-serving-certificate", - "gateway-ca-bundle", - "gateway-ca-configmap", - "gateway-namespace", - "usage-gateway-namespace-by-gateway-proxy", - "usage-gateway-namespace-by-gateway-selfsigned-issuer", - "usage-gateway-selfsigned-issuer-by-trust-manager", - }, - gateway_keys, - ) - self.assertEqual( - [ - fnv1.Result( - severity=fnv1.SEVERITY_WARNING, - message=( - "Gateway gw.clusters.example.com not served: no InferenceGateway has published a client " - "CA for this cluster to trust, and serving without one would accept unauthenticated callers" - ), - ) - ], - list(got.results), - ) + ] # The composed-resource key a component renders under is its identity: @@ -1465,242 +1447,223 @@ async def test_cluster_gateway_without_ca_serves_nothing(self) -> None: } -class TestKeyInventory(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_composed_resource_keys(self) -> None: - for cloud, cloud_keys in _INVENTORY.items(): - for stack, stack_keys in (("Standard", _STANDARD), ("Dynamo", _DYNAMO)): - with self.subTest(cloud=cloud, stack=stack): - expected = _ALWAYS | _COMMON | cloud_keys | stack_keys - # Observe every expected key Ready so the depends_on - # install gate opens and the full stack renders; a - # key the function doesn't render still fails the - # comparison. - observed = _observed_pcs() - for key in expected: - observed[key] = fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"conditions": [{"type": "Ready", "status": "True"}]}} - ) - ) - got = await self.runner.RunFunction(_request(cloud, stack, observed=observed), None) - self.assertEqual(expected, set(got.desired.resources.keys())) - - -class TestCollectorReadiness(unittest.IsolatedAsyncioTestCase): - """The collector is composed, but the stack never waits on it.""" - - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - @staticmethod - def _with_destination(req: fnv1.RunFunctionRequest, *, secret: str | None = None) -> fnv1.RunFunctionRequest: - sink: dict = {"name": "primary", "type": "otlphttp", "endpoint": "https://otel.acme.example"} - if secret: - sink |= {"secretRef": {"name": secret}, "auth": {"bearerTokenKey": "token"}} +@pytest.mark.parametrize( + ("stack", "stack_keys"), [("Standard", _STANDARD), ("Dynamo", _DYNAMO)], ids=["Standard", "Dynamo"] +) +@pytest.mark.parametrize(("cloud", "cloud_keys"), list(_INVENTORY.items()), ids=list(_INVENTORY)) +def test_composed_resource_keys(cloud: str, cloud_keys: frozenset[str], stack: str, stack_keys: frozenset[str]) -> None: + """Every cloud and stack composes exactly its inventoried resource keys.""" + expected = _ALWAYS | _COMMON | cloud_keys | stack_keys + # Observe every expected key Ready so the depends_on + # install gate opens and the full stack renders; a + # key the function doesn't render still fails the + # comparison. + observed = _observed_pcs() + for key in expected: + observed[key] = fnv1.Resource( + resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(_request(cloud, stack, observed=observed), None)) + assert set(got.desired.resources.keys()) == expected + + +def _with_destination(req: fnv1.RunFunctionRequest, *, secret: str | None = None) -> fnv1.RunFunctionRequest: + sink: dict = {"name": "primary", "type": "otlphttp", "endpoint": "https://otel.acme.example"} + if secret: + sink |= {"secretRef": {"name": secret}, "auth": {"bearerTokenKey": "token"}} + req.required_resources["destinations"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "TelemetryDestination", + "metadata": {"name": "acme"}, + "spec": {"sinks": [sink]}, + } + ) + ) + ) + req.required_resources["mappings"].items.extend([]) + if secret: + req.required_resources["collector-secret-primary"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": secret, "namespace": "modelplane-system"}, + "data": {"token": "c2hoaGg="}, + } + ) + ) + ) + return req + + +def test_the_collector_does_not_gate_the_stack() -> None: + """A collector nothing has observed yet is still Ready. + + Everything else the stack composes is Ready only once its observed + Ready condition says so, because the fleet cannot serve without it. + The collector only watches, so gating on it would put placing a + replica behind exporting a metric: one destination pointing at an + endpoint that has gone away would take every InferenceCluster in the + fleet out of Ready and stop the scheduler. + """ + req = _with_destination(_request("GKE", "Standard", observed=_observed_pcs())) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + collector_keys = [k for k in got.desired.resources if k == "collector" or k.startswith("collector-")] + # The Deployment, which is the one with a readiness CEL of its own and + # so the one that would have gated the stack. + assert "collector" in collector_keys + for key in collector_keys: + assert key not in req.observed.resources, key + assert got.desired.resources[key].ready == fnv1.READY_TRUE, key + + +def test_the_credential_reaches_the_cluster_that_mounts_it() -> None: + """The operator writes one Secret; the collector mounts it elsewhere. + + A TelemetryDestination is cluster-scoped on the control plane and the + collector runs on every workload cluster in the fleet. Resolving the + Secret and stopping there leaves the Deployment mounting a name + nothing out there creates, so the pod never starts and the fleet + exports nothing - the failure every destination with a credential + would hit, which is every destination that reaches a real backend. + """ + req = _with_destination(_request("GKE", "Standard", observed=_observed_pcs()), secret="telemetry-credentials") + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert "collector-secret-primary" in got.desired.resources + composed = resource.struct_to_dict(got.desired.resources["collector-secret-primary"].resource) + manifest = composed["spec"]["forProvider"]["manifest"] + assert manifest["kind"] == "Secret" + assert manifest["metadata"]["name"] == "telemetry-credentials" + assert manifest["metadata"]["namespace"] == "modelplane-system" + # Copied verbatim: re-encoding would corrupt a credential that is not + # text, and the mount reads the same key the sink's auth names. + assert manifest["data"] == {"token": "c2hoaGg="} + + +def test_a_destination_asks_for_its_credential_in_one_namespace() -> None: + """Unqualified, the requirement matches a Secret of that name anywhere.""" + req = _with_destination(_request("GKE", "Standard", observed=_observed_pcs()), secret="telemetry-credentials") + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + selector = got.requirements.resources["collector-secret-primary"] + assert selector.match_name == "telemetry-credentials" + assert selector.namespace == "modelplane-system" + + +def test_every_destination_contributes_its_sinks() -> None: + """A second backend is a second object, not an edit to a singleton. + + Picking one destination and warning about the rest means a team + adding an export has to edit an object another team owns, and gets + silence if they create their own instead. + """ + req = _request("GKE", "Standard", observed=_observed_pcs()) + for name, sink in ( + ("acme", {"name": "vendor", "type": "otlphttp", "endpoint": "https://otel.vendor.example"}), + ("zeta", {"name": "prom", "type": "prometheus_remote_write", "endpoint": "https://p.example/w"}), + ): req.required_resources["destinations"].items.append( fnv1.Resource( resource=resource.dict_to_struct( { "apiVersion": "modelplane.ai/v1alpha1", "kind": "TelemetryDestination", - "metadata": {"name": "acme"}, + "metadata": {"name": name}, "spec": {"sinks": [sink]}, } ) ) ) - req.required_resources["mappings"].items.extend([]) - if secret: - req.required_resources["collector-secret-primary"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": secret, "namespace": "modelplane-system"}, - "data": {"token": "c2hoaGg="}, - } - ) - ) - ) - return req - - async def test_the_collector_does_not_gate_the_stack(self) -> None: - """A collector nothing has observed yet is still Ready. - - Everything else the stack composes is Ready only once its observed - Ready condition says so, because the fleet cannot serve without it. - The collector only watches, so gating on it would put placing a - replica behind exporting a metric: one destination pointing at an - endpoint that has gone away would take every InferenceCluster in the - fleet out of Ready and stop the scheduler. - """ - req = self._with_destination(_request("GKE", "Standard", observed=_observed_pcs())) - got = await self.runner.RunFunction(req, None) - collector_keys = [k for k in got.desired.resources if k == "collector" or k.startswith("collector-")] - # The Deployment, which is the one with a readiness CEL of its own and - # so the one that would have gated the stack. - self.assertIn("collector", collector_keys) - for key in collector_keys: - with self.subTest(key=key): - self.assertNotIn(key, req.observed.resources) - self.assertEqual(got.desired.resources[key].ready, fnv1.READY_TRUE) - - async def test_the_credential_reaches_the_cluster_that_mounts_it(self) -> None: - """The operator writes one Secret; the collector mounts it elsewhere. - - A TelemetryDestination is cluster-scoped on the control plane and the - collector runs on every workload cluster in the fleet. Resolving the - Secret and stopping there leaves the Deployment mounting a name - nothing out there creates, so the pod never starts and the fleet - exports nothing - the failure every destination with a credential - would hit, which is every destination that reaches a real backend. - """ - req = self._with_destination( - _request("GKE", "Standard", observed=_observed_pcs()), secret="telemetry-credentials" - ) - got = await self.runner.RunFunction(req, None) - self.assertIn("collector-secret-primary", got.desired.resources) - composed = resource.struct_to_dict(got.desired.resources["collector-secret-primary"].resource) - manifest = composed["spec"]["forProvider"]["manifest"] - self.assertEqual(manifest["kind"], "Secret") - self.assertEqual(manifest["metadata"]["name"], "telemetry-credentials") - self.assertEqual(manifest["metadata"]["namespace"], "modelplane-system") - # Copied verbatim: re-encoding would corrupt a credential that is not - # text, and the mount reads the same key the sink's auth names. - self.assertEqual(manifest["data"], {"token": "c2hoaGg="}) - - async def test_a_destination_asks_for_its_credential_in_one_namespace(self) -> None: - """Unqualified, the requirement matches a Secret of that name anywhere.""" - req = self._with_destination( - _request("GKE", "Standard", observed=_observed_pcs()), secret="telemetry-credentials" - ) - got = await self.runner.RunFunction(req, None) - selector = got.requirements.resources["collector-secret-primary"] - self.assertEqual(selector.match_name, "telemetry-credentials") - self.assertEqual(selector.namespace, "modelplane-system") - - async def test_every_destination_contributes_its_sinks(self) -> None: - """A second backend is a second object, not an edit to a singleton. - - Picking one destination and warning about the rest means a team - adding an export has to edit an object another team owns, and gets - silence if they create their own instead. - """ - req = _request("GKE", "Standard", observed=_observed_pcs()) - for name, sink in ( - ("acme", {"name": "vendor", "type": "otlphttp", "endpoint": "https://otel.vendor.example"}), - ("zeta", {"name": "prom", "type": "prometheus_remote_write", "endpoint": "https://p.example/w"}), - ): - req.required_resources["destinations"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "TelemetryDestination", - "metadata": {"name": name}, - "spec": {"sinks": [sink]}, - } - ) - ) - ) - req.required_resources["mappings"].items.extend([]) - got = await self.runner.RunFunction(req, None) - config = yaml.safe_load( - resource.struct_to_dict(got.desired.resources["collector-config"].resource)["spec"]["forProvider"][ - "manifest" - ]["data"]["collector.yaml"] - ) - self.assertEqual( - sorted(config["exporters"]), - ["otlphttp/vendor", "prometheus_remote_write/prom"], - ) - self.assertEqual( - sorted(config["service"]["pipelines"]["metrics"]["exporters"]), - ["otlphttp/vendor", "prometheus_remote_write/prom"], - ) + req.required_resources["mappings"].items.extend([]) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + config = yaml.safe_load( + resource.struct_to_dict(got.desired.resources["collector-config"].resource)["spec"]["forProvider"]["manifest"][ + "data" + ]["collector.yaml"] + ) + assert sorted(config["exporters"]) == ["otlphttp/vendor", "prometheus_remote_write/prom"] + assert sorted(config["service"]["pipelines"]["metrics"]["exporters"]) == [ + "otlphttp/vendor", + "prometheus_remote_write/prom", + ] - async def test_two_destinations_cannot_name_one_exporter(self) -> None: - """A sink names the collector's exporter instance. - - Two of them under one name is one exporter with two meanings. The - destination sorting first keeps it and the other is dropped with a - warning, rather than failing the whole fleet's telemetry over a name. - """ - req = _request("GKE", "Standard", observed=_observed_pcs()) - for name, endpoint in (("acme", "https://a.example"), ("zeta", "https://z.example")): - req.required_resources["destinations"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "TelemetryDestination", - "metadata": {"name": name}, - "spec": {"sinks": [{"name": "primary", "type": "otlphttp", "endpoint": endpoint}]}, - } - ) - ) - ) - req.required_resources["mappings"].items.extend([]) - got = await self.runner.RunFunction(req, None) - config = yaml.safe_load( - resource.struct_to_dict(got.desired.resources["collector-config"].resource)["spec"]["forProvider"][ - "manifest" - ]["data"]["collector.yaml"] - ) - self.assertEqual(list(config["exporters"]), ["otlphttp/primary"]) - self.assertEqual(config["exporters"]["otlphttp/primary"]["endpoint"], "https://a.example") - self.assertTrue([r for r in got.results if "zeta" in r.message]) - - async def test_a_stale_mapping_does_not_break_the_stack(self) -> None: - """A CRD validates on write, not on what it already stored. - - A MetricMapping written against an older schema comes back on read - exactly as it was stored, so a value the enum no longer carries - reaches the parser. Parsing it raises, and raising fails the whole - pipeline step - so the serving stack composes nothing and the fleet - stops placing replicas, because one telemetry object is out of date. - Seen on a real cluster, where a mapping predating a required field - did it. - """ - req = self._with_destination(_request("GKE", "Standard", observed=_observed_pcs())) - req.required_resources["mappings"].items.append( + +def test_two_destinations_cannot_name_one_exporter() -> None: + """A sink names the collector's exporter instance. + + Two of them under one name is one exporter with two meanings. The + destination sorting first keeps it and the other is dropped with a + warning, rather than failing the whole fleet's telemetry over a name. + """ + req = _request("GKE", "Standard", observed=_observed_pcs()) + for name, endpoint in (("acme", "https://a.example"), ("zeta", "https://z.example")): + req.required_resources["destinations"].items.append( fnv1.Resource( resource=resource.dict_to_struct( { "apiVersion": "modelplane.ai/v1alpha1", - "kind": "MetricMapping", - "metadata": {"name": "stale"}, - # A unit the enum no longer carries. - "spec": { - "metrics": [ - { - "from": "old_engine_transfer", - "to": "modelplane_request_kv_transfer_seconds", - "fromUnit": "Centiseconds", - } - ] - }, + "kind": "TelemetryDestination", + "metadata": {"name": name}, + "spec": {"sinks": [{"name": "primary", "type": "otlphttp", "endpoint": endpoint}]}, } ) ) ) - got = await self.runner.RunFunction(req, None) - # The stack still composes, and says what it dropped. - self.assertIn("collector", got.desired.resources) - self.assertTrue([r for r in got.results if "stale" in r.message]) - config = yaml.safe_load( - resource.struct_to_dict(got.desired.resources["collector-config"].resource)["spec"]["forProvider"][ - "manifest" - ]["data"]["collector.yaml"] + req.required_resources["mappings"].items.extend([]) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + config = yaml.safe_load( + resource.struct_to_dict(got.desired.resources["collector-config"].resource)["spec"]["forProvider"]["manifest"][ + "data" + ]["collector.yaml"] + ) + assert list(config["exporters"]) == ["otlphttp/primary"] + assert config["exporters"]["otlphttp/primary"]["endpoint"] == "https://a.example" + assert [r for r in got.results if "zeta" in r.message] + + +def test_a_stale_mapping_does_not_break_the_stack() -> None: + """A CRD validates on write, not on what it already stored. + + A MetricMapping written against an older schema comes back on read + exactly as it was stored, so a value the enum no longer carries + reaches the parser. Parsing it raises, and raising fails the whole + pipeline step - so the serving stack composes nothing and the fleet + stops placing replicas, because one telemetry object is out of date. + Seen on a real cluster, where a mapping predating a required field + did it. + """ + req = _with_destination(_request("GKE", "Standard", observed=_observed_pcs())) + req.required_resources["mappings"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "MetricMapping", + "metadata": {"name": "stale"}, + # A unit the enum no longer carries. + "spec": { + "metrics": [ + { + "from": "old_engine_transfer", + "to": "modelplane_request_kv_transfer_seconds", + "fromUnit": "Centiseconds", + } + ] + }, + } + ) ) - self.assertNotIn("old_engine_waiting", yaml.safe_dump(config)) + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + # The stack still composes, and says what it dropped. + assert "collector" in got.desired.resources + assert [r for r in got.results if "stale" in r.message] + config = yaml.safe_load( + resource.struct_to_dict(got.desired.resources["collector-config"].resource)["spec"]["forProvider"]["manifest"][ + "data" + ]["collector.yaml"] + ) + assert "old_engine_waiting" not in yaml.safe_dump(config) diff --git a/functions/compose-serving-stack/tests/test_stacks.py b/functions/compose-serving-stack/tests/test_stacks.py index f17464637..592a6ec28 100644 --- a/functions/compose-serving-stack/tests/test_stacks.py +++ b/functions/compose-serving-stack/tests/test_stacks.py @@ -20,185 +20,187 @@ generated ones, once mapped - without involving fn.py. """ -import unittest - +import pytest from function import stacks from function.stacks.clouds import civo -class TestComponents(unittest.TestCase): - def test_every_cloud_and_stack_joins(self) -> None: - for cloud in stacks.clouds(): - for stack in stacks.stacks(): - with self.subTest(cloud=cloud, stack=stack): - got = stacks.join(cloud, stack) - self.assertTrue(got, "a joined stack can't be empty") - - def test_charts_have_reserved_release_names(self) -> None: - for cloud in stacks.clouds(): - for stack in stacks.stacks(): - for c in stacks.join(cloud, stack): - if isinstance(c, stacks.Chart): - with self.subTest(cloud=cloud, stack=stack, key=c.key): - self.assertEqual( - f"mp-{c.chart}", - c.release, - "release names are mp-: stable across upgrades, reserved to Modelplane", - ) - - def test_manifests_are_populated(self) -> None: - for cloud in stacks.clouds(): - for stack in stacks.stacks(): - for c in stacks.join(cloud, stack): - if isinstance(c, stacks.Manifests): - with self.subTest(cloud=cloud, stack=stack, key=c.key): - self.assertTrue(c.manifests, "a Manifests entry can't be empty") - - def test_multi_doc_manifests_derive_per_doc_keys(self) -> None: - for cloud in stacks.clouds(): - for stack in stacks.stacks(): - for c in stacks.join(cloud, stack): - keys = stacks.components.doc_keys(c) - if isinstance(c, stacks.Chart) or len(c.manifests) == 1: - self.assertEqual([c.key], keys) - continue - with self.subTest(cloud=cloud, stack=stack, key=c.key): - self.assertEqual( - [f"{c.key}-{doc['metadata']['name']}" for doc in c.manifests], - keys, - "a multi-doc bundle renders one Object per doc, keyed -", - ) - - def test_ready_entries_are_single_doc(self) -> None: - # A readiness CEL query applies to every doc in an entry, so an - # entry carrying one keeps to a single manifest - a Service or - # ServiceAccount has no status conditions to satisfy it. - for cloud in stacks.clouds(): - for stack in stacks.stacks(): - for c in stacks.join(cloud, stack): - if isinstance(c, stacks.Manifests) and c.ready is not None: - with self.subTest(cloud=cloud, stack=stack, key=c.key): - self.assertEqual(1, len(c.manifests)) - - def test_depended_on_charts_wait(self) -> None: - # A chart another component depends on renders with helm --wait, - # so its Ready means healthy and the install gate orders - # dependents on health rather than deploy. Without this, the - # gate would open the moment Helm accepted the manifests. - for cloud in stacks.clouds(): - for stack in stacks.stacks(): - joined = stacks.join(cloud, stack) - depended_on = {dep for c in joined for dep in c.depends_on} - for c in joined: - if isinstance(c, stacks.Chart) and c.key in depended_on: - with self.subTest(cloud=cloud, stack=stack, key=c.key): - self.assertTrue(c.wait, "a depended-on chart must set wait") - - def test_no_wildcard_tolerations(self) -> None: - # A keyless toleration tolerates every taint, so the pod lands - # on tainted GPU nodes: control-plane charts squat on - # accelerated capacity and their eviction stalls autoscaler - # scale-down. aicr's bundler stamps exactly that wildcard on - # every pod it renders; the generator scopes each one - # (TOLERATIONS in generate.py). This pins that no keyless - # toleration survives in any joined stack, chart values and - # manifests alike. - def check(node: object, where: str) -> None: - if isinstance(node, dict): - for key, val in node.items(): - if key == "tolerations" and isinstance(val, list): - for toleration in val: - self.assertTrue( - isinstance(toleration, dict) and "key" in toleration, - f"keyless (wildcard) toleration in {where}", - ) - else: - check(val, where) - elif isinstance(node, list): - for item in node: - check(item, where) - - for cloud in stacks.clouds(): - for stack in stacks.stacks(): - for c in stacks.join(cloud, stack): - with self.subTest(cloud=cloud, stack=stack, key=c.key): - check(c.values if isinstance(c, stacks.Chart) else c.manifests, c.key) - - def test_unknown_cloud_and_stack_fail_closed(self) -> None: - # The Literal types reject these at type-checking time; this - # exercises the runtime guard behind them, which catches the API - # and the stacks package disagreeing on a value. - with self.assertRaises(ValueError): - stacks.join("Mars", "Standard") # ty: ignore[invalid-argument-type] - with self.assertRaises(ValueError): - stacks.join("Nebius", "Turbo") # ty: ignore[invalid-argument-type] - - -class TestWithNvLinkDisabled(unittest.TestCase): - """with_nvlink_disabled scopes NVLink disable to the named pools.""" - - def test_gpu_operator_switches_to_nvidia_driver_crd(self) -> None: - # The chart's default NVIDIADriver (deployDefaultCR) keeps driving - # pools the transform doesn't name, so flipping modes changes - # nothing for them. - got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["h100-pool"]) - op = next(c for c in got if isinstance(c, stacks.Chart) and c.key == "gpu-operator") - assert op.values is not None - self.assertEqual({"enabled": True, "deployDefaultCR": True}, op.values["driver"]["nvidiaDriverCRD"]) - - def test_each_pool_gets_its_own_driver(self) -> None: - got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["pool-a", "pool-b"]) - drivers = [c for c in got if isinstance(c, stacks.Manifests) and c.key.startswith("nvlink-disabled-driver-")] - self.assertEqual(["nvlink-disabled-driver-pool-a", "nvlink-disabled-driver-pool-b"], [c.key for c in drivers]) - for c, pool in zip(drivers, ["pool-a", "pool-b"], strict=True): - with self.subTest(pool=pool): - # Ready entries keep to a single doc, and gate on the - # operator-populated state so the DRA driver's install - # gate orders on driver health. - self.assertEqual(1, len(c.manifests)) - self.assertIsNotNone(c.ready) - self.assertEqual(["gpu-operator", "nvlink-disable-config"], c.depends_on) - spec = c.manifests[0]["spec"] - self.assertEqual({"modelplane.ai/pool": pool}, spec["nodeSelector"]) - self.assertEqual({"name": "nvidia-kernel-config"}, spec["kernelModuleConfig"]) - - def test_driver_pin_mirrors_the_chart(self) -> None: - # One review moves both: the per-pool NVIDIADriver must install - # the same driver the chart's default CR does. - got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["h100-pool"]) - op = next(c for c in got if isinstance(c, stacks.Chart) and c.key == "gpu-operator") - driver = next(c for c in got if isinstance(c, stacks.Manifests) and c.key == "nvlink-disabled-driver-h100-pool") - assert op.values is not None - spec = driver.manifests[0]["spec"] - self.assertEqual(op.values["driver"]["version"], spec["version"]) - self.assertEqual(op.values["driver"]["useOpenKernelModules"], spec["useOpenKernelModules"]) - - def test_configmap_carries_the_module_option(self) -> None: - got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["h100-pool"]) - config = next(c for c in got if isinstance(c, stacks.Manifests) and c.key == "nvlink-disable-config") - kinds = [doc["kind"] for doc in config.manifests] - self.assertEqual(["Namespace", "ConfigMap"], kinds) - self.assertEqual( - {"nvidia.conf": "options nvidia NVreg_NvLinkDisable=1"}, - config.manifests[1]["data"], +@pytest.mark.parametrize("stack", stacks.stacks()) +@pytest.mark.parametrize("cloud", stacks.clouds()) +def test_every_cloud_and_stack_joins(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """Every cloud and stack pair joins into a non-empty stack.""" + got = stacks.join(cloud, stack) + assert got, "a joined stack can't be empty" + + +@pytest.mark.parametrize("stack", stacks.stacks()) +@pytest.mark.parametrize("cloud", stacks.clouds()) +def test_charts_have_reserved_release_names(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """Every Chart's Helm release is named mp-.""" + for c in stacks.join(cloud, stack): + if isinstance(c, stacks.Chart): + assert c.release == f"mp-{c.chart}", ( + f"{c.key}: release names are mp-: stable across upgrades, reserved to Modelplane" + ) + + +@pytest.mark.parametrize("stack", stacks.stacks()) +@pytest.mark.parametrize("cloud", stacks.clouds()) +def test_manifests_are_populated(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """Every Manifests entry carries at least one manifest.""" + for c in stacks.join(cloud, stack): + if isinstance(c, stacks.Manifests): + assert c.manifests, f"{c.key}: a Manifests entry can't be empty" + + +@pytest.mark.parametrize("stack", stacks.stacks()) +@pytest.mark.parametrize("cloud", stacks.clouds()) +def test_multi_doc_manifests_derive_per_doc_keys(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """A multi-doc Manifests entry renders a key per doc, and anything else renders its own key.""" + for c in stacks.join(cloud, stack): + keys = stacks.components.doc_keys(c) + if isinstance(c, stacks.Chart) or len(c.manifests) == 1: + assert keys == [c.key] + continue + assert keys == [f"{c.key}-{doc['metadata']['name']}" for doc in c.manifests], ( + f"{c.key}: a multi-doc bundle renders one Object per doc, keyed -" ) - def test_dra_driver_gates_on_pool_drivers(self) -> None: - got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["pool-a", "pool-b"]) - dra = next(c for c in got if isinstance(c, stacks.Chart) and c.key == "nvidia-dra-driver-gpu") - self.assertEqual( - ["gpu-operator", "nvlink-disabled-driver-pool-a", "nvlink-disabled-driver-pool-b"], - dra.depends_on, - ) - def test_join_is_not_mutated(self) -> None: - # The transform must copy: the joined lists share the module-level - # component objects, and mutating them would leak NVLink disable - # into every later request. - civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["h100-pool"]) - joined = stacks.join("Civo", "Standard") - op = next(c for c in joined if isinstance(c, stacks.Chart) and c.key == "gpu-operator") - dra = next(c for c in joined if isinstance(c, stacks.Chart) and c.key == "nvidia-dra-driver-gpu") - assert op.values is not None - self.assertNotIn("nvidiaDriverCRD", op.values["driver"]) - self.assertEqual(["gpu-operator"], dra.depends_on) +@pytest.mark.parametrize("stack", stacks.stacks()) +@pytest.mark.parametrize("cloud", stacks.clouds()) +def test_ready_entries_are_single_doc(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """A Manifests entry with a readiness query carries a single manifest.""" + # A readiness CEL query applies to every doc in an entry, so an + # entry carrying one keeps to a single manifest - a Service or + # ServiceAccount has no status conditions to satisfy it. + for c in stacks.join(cloud, stack): + if isinstance(c, stacks.Manifests) and c.ready is not None: + assert len(c.manifests) == 1, c.key + + +@pytest.mark.parametrize("stack", stacks.stacks()) +@pytest.mark.parametrize("cloud", stacks.clouds()) +def test_depended_on_charts_wait(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """Every Chart another component depends on sets wait.""" + # A chart another component depends on renders with helm --wait, + # so its Ready means healthy and the install gate orders + # dependents on health rather than deploy. Without this, the + # gate would open the moment Helm accepted the manifests. + joined = stacks.join(cloud, stack) + depended_on = {dep for c in joined for dep in c.depends_on} + for c in joined: + if isinstance(c, stacks.Chart) and c.key in depended_on: + assert c.wait, f"{c.key}: a depended-on chart must set wait" + + +@pytest.mark.parametrize("stack", stacks.stacks()) +@pytest.mark.parametrize("cloud", stacks.clouds()) +def test_no_wildcard_tolerations(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """No component of a joined stack carries a keyless toleration.""" + + # A keyless toleration tolerates every taint, so the pod lands + # on tainted GPU nodes: control-plane charts squat on + # accelerated capacity and their eviction stalls autoscaler + # scale-down. aicr's bundler stamps exactly that wildcard on + # every pod it renders; the generator scopes each one + # (TOLERATIONS in generate.py). This pins that no keyless + # toleration survives in any joined stack, chart values and + # manifests alike. + def check(node: object, where: str) -> None: + if isinstance(node, dict): + for key, val in node.items(): + if key == "tolerations" and isinstance(val, list): + for toleration in val: + assert isinstance(toleration, dict), f"keyless (wildcard) toleration in {where}" + assert "key" in toleration, f"keyless (wildcard) toleration in {where}" + else: + check(val, where) + elif isinstance(node, list): + for item in node: + check(item, where) + + for c in stacks.join(cloud, stack): + check(c.values if isinstance(c, stacks.Chart) else c.manifests, c.key) + + +def test_unknown_cloud_and_stack_fail_closed() -> None: + """join rejects an unknown cloud or stack.""" + # The Literal types reject these at type-checking time; this + # exercises the runtime guard behind them, which catches the API + # and the stacks package disagreeing on a value. + with pytest.raises(ValueError, match="unknown cloud 'Mars'"): + stacks.join("Mars", "Standard") # ty: ignore[invalid-argument-type] + with pytest.raises(ValueError, match="unknown stack 'Turbo'"): + stacks.join("Nebius", "Turbo") # ty: ignore[invalid-argument-type] + + +def test_gpu_operator_switches_to_nvidia_driver_crd() -> None: + """with_nvlink_disabled switches the gpu-operator chart to NVIDIADriver-CRD mode.""" + # The chart's default NVIDIADriver (deployDefaultCR) keeps driving + # pools the transform doesn't name, so flipping modes changes + # nothing for them. + got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["h100-pool"]) + op = next(c for c in got if isinstance(c, stacks.Chart) and c.key == "gpu-operator") + assert op.values is not None + assert op.values["driver"]["nvidiaDriverCRD"] == {"enabled": True, "deployDefaultCR": True} + + +def test_each_pool_gets_its_own_driver() -> None: + """with_nvlink_disabled adds one NVIDIADriver per named pool, selecting that pool's nodes.""" + got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["pool-a", "pool-b"]) + drivers = [c for c in got if isinstance(c, stacks.Manifests) and c.key.startswith("nvlink-disabled-driver-")] + assert [c.key for c in drivers] == ["nvlink-disabled-driver-pool-a", "nvlink-disabled-driver-pool-b"] + for c, pool in zip(drivers, ["pool-a", "pool-b"], strict=True): + # Ready entries keep to a single doc, and gate on the + # operator-populated state so the DRA driver's install + # gate orders on driver health. + assert len(c.manifests) == 1, pool + assert c.ready is not None, pool + assert c.depends_on == ["gpu-operator", "nvlink-disable-config"], pool + spec = c.manifests[0]["spec"] + assert spec["nodeSelector"] == {"modelplane.ai/pool": pool}, pool + assert spec["kernelModuleConfig"] == {"name": "nvidia-kernel-config"}, pool + + +def test_driver_pin_mirrors_the_chart() -> None: + """A per-pool NVIDIADriver installs the same driver as the chart's default CR.""" + # One review moves both: the per-pool NVIDIADriver must install + # the same driver the chart's default CR does. + got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["h100-pool"]) + op = next(c for c in got if isinstance(c, stacks.Chart) and c.key == "gpu-operator") + driver = next(c for c in got if isinstance(c, stacks.Manifests) and c.key == "nvlink-disabled-driver-h100-pool") + assert op.values is not None + spec = driver.manifests[0]["spec"] + assert spec["version"] == op.values["driver"]["version"] + assert spec["useOpenKernelModules"] == op.values["driver"]["useOpenKernelModules"] + + +def test_configmap_carries_the_module_option() -> None: + """with_nvlink_disabled adds the kernel module ConfigMap, in its own Namespace.""" + got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["h100-pool"]) + config = next(c for c in got if isinstance(c, stacks.Manifests) and c.key == "nvlink-disable-config") + kinds = [doc["kind"] for doc in config.manifests] + assert kinds == ["Namespace", "ConfigMap"] + assert config.manifests[1]["data"] == {"nvidia.conf": "options nvidia NVreg_NvLinkDisable=1"} + + +def test_dra_driver_gates_on_pool_drivers() -> None: + """The DRA driver depends on every per-pool NVIDIADriver.""" + got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["pool-a", "pool-b"]) + dra = next(c for c in got if isinstance(c, stacks.Chart) and c.key == "nvidia-dra-driver-gpu") + assert dra.depends_on == ["gpu-operator", "nvlink-disabled-driver-pool-a", "nvlink-disabled-driver-pool-b"] + + +def test_join_is_not_mutated() -> None: + """with_nvlink_disabled leaves the joined stack it was given unchanged.""" + # The transform must copy: the joined lists share the module-level + # component objects, and mutating them would leak NVLink disable + # into every later request. + civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["h100-pool"]) + joined = stacks.join("Civo", "Standard") + op = next(c for c in joined if isinstance(c, stacks.Chart) and c.key == "gpu-operator") + dra = next(c for c in joined if isinstance(c, stacks.Chart) and c.key == "nvidia-dra-driver-gpu") + assert op.values is not None + assert "nvidiaDriverCRD" not in op.values["driver"] + assert dra.depends_on == ["gpu-operator"] diff --git a/functions/compose-telemetry-destination/tests/__init__.py b/functions/compose-telemetry-destination/tests/__init__.py deleted file mode 100644 index ebf4b2ad4..000000000 --- a/functions/compose-telemetry-destination/tests/__init__.py +++ /dev/null @@ -1,13 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. diff --git a/functions/compose-telemetry-destination/tests/test_fn.py b/functions/compose-telemetry-destination/tests/test_fn.py index 02367c924..14d1d4cc6 100644 --- a/functions/compose-telemetry-destination/tests/test_fn.py +++ b/functions/compose-telemetry-destination/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-telemetry-destination function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb @@ -34,222 +36,211 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function reports whether a destination can actually be sent through.""" - - def sink(name: str = "primary", type_: str = "otlphttp", secret: str | None = None) -> dict: - """A sink wiring its own authenticator, which is the case worth validating.""" - return { - "name": name, - "type": type_, - "endpoint": "https://otel.acme.example", - "config": {"auth": {"authenticator": "oauth2client/acme"}}, - **({"secretRef": {"name": secret}} if secret else {}), - } - - sinks = [sink()] - extensions = {"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}} - - def xr(spec: dict) -> dict: - return { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "TelemetryDestination", - "metadata": {"name": "default"}, - "spec": spec, - } - - def req(spec: dict, secrets: list | None = None) -> fnv1.RunFunctionRequest: - r = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(xr(spec)))), - ) - if secrets is not None: - r.required_resources["secret-primary"].items.extend([fnv1.Resource(resource=s) for s in secrets]) - return r - - def want( - ready: fnv1.Ready, status: dict | None, cond: fnv1.Condition, secret: str | None = None - ) -> fnv1.RunFunctionResponse: - composite = fnv1.Resource(ready=ready) - if status is not None: - composite.resource.CopyFrom(resource.dict_to_struct(status)) - rsp = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=composite), - conditions=[cond], - context=structpb.Struct(), - ) - if secret is not None: - rsp.requirements.resources["secret-primary"].api_version = "v1" - rsp.requirements.resources["secret-primary"].kind = "Secret" - rsp.requirements.resources["secret-primary"].match_name = secret - # Qualified: unqualified it would resolve a Secret of that - # name in any namespace, and accept the wrong credential. - rsp.requirements.resources["secret-primary"].namespace = "modelplane-system" - return rsp - - cases = [ - Case( - name="ready, naming the sinks it sends through", - req=req({"sinks": sinks, "extensions": extensions}), - want=want( - fnv1.READY_TRUE, - {"status": {}}, - fnv1.Condition( - type="Accepted", - status=fnv1.STATUS_CONDITION_TRUE, - reason="Available", - message="Exporting through otlphttp/primary", - ), +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +def _compose_cases() -> list[Case]: + def sink(name: str = "primary", type_: str = "otlphttp", secret: str | None = None) -> dict: + """A sink wiring its own authenticator, which is the case worth validating.""" + return { + "name": name, + "type": type_, + "endpoint": "https://otel.acme.example", + "config": {"auth": {"authenticator": "oauth2client/acme"}}, + **({"secretRef": {"name": secret}} if secret else {}), + } + + sinks = [sink()] + extensions = {"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}} + + def xr(spec: dict) -> dict: + return { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "TelemetryDestination", + "metadata": {"name": "default"}, + "spec": spec, + } + + def req(spec: dict, secrets: list | None = None) -> fnv1.RunFunctionRequest: + r = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(xr(spec)))), + ) + if secrets is not None: + r.required_resources["secret-primary"].items.extend([fnv1.Resource(resource=s) for s in secrets]) + return r + + def want( + ready: fnv1.Ready, status: dict | None, cond: fnv1.Condition, secret: str | None = None + ) -> fnv1.RunFunctionResponse: + composite = fnv1.Resource(ready=ready) + if status is not None: + composite.resource.CopyFrom(resource.dict_to_struct(status)) + rsp = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=composite), + conditions=[cond], + context=structpb.Struct(), + ) + if secret is not None: + rsp.requirements.resources["secret-primary"].api_version = "v1" + rsp.requirements.resources["secret-primary"].kind = "Secret" + rsp.requirements.resources["secret-primary"].match_name = secret + # Qualified: unqualified it would resolve a Secret of that + # name in any namespace, and accept the wrong credential. + rsp.requirements.resources["secret-primary"].namespace = "modelplane-system" + return rsp + + return [ + Case( + name="ready, naming the sinks it sends through", + req=req({"sinks": sinks, "extensions": extensions}), + want=want( + fnv1.READY_TRUE, + {"status": {}}, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_TRUE, + reason="Available", + message="Exporting through otlphttp/primary", ), ), - Case( - name="ready with an exporter that references no authenticator at all", - req=req( - { - "sinks": [ - { - "name": "prom", - "type": "prometheusremotewrite", - "endpoint": "https://prom.acme.example/api/v1/write", - } - ] - } - ), - want=want( - fnv1.READY_TRUE, - {"status": {}}, - fnv1.Condition( - type="Accepted", - status=fnv1.STATUS_CONDITION_TRUE, - reason="Available", - message="Exporting through prometheusremotewrite/prom", - ), - ), + ), + Case( + name="ready with an exporter that references no authenticator at all", + req=req( + { + "sinks": [ + { + "name": "prom", + "type": "prometheusremotewrite", + "endpoint": "https://prom.acme.example/api/v1/write", + } + ] + } ), - Case( - name="ready with no extensions, because Modelplane composes the authenticator", - req=req( - { - "sinks": [ - { - "name": "primary", - "type": "otlphttp", - "endpoint": "https://otel.acme.example", - "secretRef": {"name": "telemetry-credentials"}, - "auth": {"bearerTokenKey": "token"}, - } - ] - }, - secrets=[ - resource.dict_to_struct( - {"apiVersion": "v1", "kind": "Secret", "metadata": {"name": "telemetry-credentials"}} - ) - ], - ), - want=want( - fnv1.READY_TRUE, - {"status": {}}, - fnv1.Condition( - type="Accepted", - status=fnv1.STATUS_CONDITION_TRUE, - reason="Available", - message="Exporting through otlphttp/primary", - ), - secret="telemetry-credentials", + want=want( + fnv1.READY_TRUE, + {"status": {}}, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_TRUE, + reason="Available", + message="Exporting through prometheusremotewrite/prom", ), ), - Case( - name="ready once the credential Secret exists", - req=req( - {"sinks": [sink(secret="telemetry-credentials")], "extensions": extensions}, - secrets=[ - resource.dict_to_struct( - {"apiVersion": "v1", "kind": "Secret", "metadata": {"name": "telemetry-credentials"}} - ) - ], - ), - want=want( - fnv1.READY_TRUE, - {"status": {}}, - fnv1.Condition( - type="Accepted", - status=fnv1.STATUS_CONDITION_TRUE, - reason="Available", - message="Exporting through otlphttp/primary", - ), - secret="telemetry-credentials", - ), + ), + Case( + name="ready with no extensions, because Modelplane composes the authenticator", + req=req( + { + "sinks": [ + { + "name": "primary", + "type": "otlphttp", + "endpoint": "https://otel.acme.example", + "secretRef": {"name": "telemetry-credentials"}, + "auth": {"bearerTokenKey": "token"}, + } + ] + }, + secrets=[ + resource.dict_to_struct( + {"apiVersion": "v1", "kind": "Secret", "metadata": {"name": "telemetry-credentials"}} + ) + ], ), - Case( - name="waits for the credential Secret to resolve", - req=req( - {"sinks": [sink(secret="telemetry-credentials")], "extensions": extensions}, + want=want( + fnv1.READY_TRUE, + {"status": {}}, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_TRUE, + reason="Available", + message="Exporting through otlphttp/primary", ), - want=want( - fnv1.READY_FALSE, - None, - fnv1.Condition( - type="Accepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForSecret", - message="Waiting for the credential Secret to resolve", - ), - secret="telemetry-credentials", + secret="telemetry-credentials", + ), + ), + Case( + name="ready once the credential Secret exists", + req=req( + {"sinks": [sink(secret="telemetry-credentials")], "extensions": extensions}, + secrets=[ + resource.dict_to_struct( + {"apiVersion": "v1", "kind": "Secret", "metadata": {"name": "telemetry-credentials"}} + ) + ], + ), + want=want( + fnv1.READY_TRUE, + {"status": {}}, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_TRUE, + reason="Available", + message="Exporting through otlphttp/primary", ), + secret="telemetry-credentials", ), - Case( - name="not ready when a sink names an authenticator nothing defines", - req=req({"sinks": sinks}), - want=want( - fnv1.READY_FALSE, - None, - fnv1.Condition( - type="Accepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="UnknownAuthenticator", - message="No extension defines oauth2client/acme, so the collector would refuse to start", - ), + ), + Case( + name="waits for the credential Secret to resolve", + req=req( + {"sinks": [sink(secret="telemetry-credentials")], "extensions": extensions}, + ), + want=want( + fnv1.READY_FALSE, + None, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForSecret", + message="Waiting for the credential Secret to resolve", ), + secret="telemetry-credentials", ), - Case( - name="not ready when the credential Secret is missing", - req=req( - {"sinks": [sink(secret="telemetry-credentials")], "extensions": extensions}, - secrets=[], + ), + Case( + name="not ready when a sink names an authenticator nothing defines", + req=req({"sinks": sinks}), + want=want( + fnv1.READY_FALSE, + None, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="UnknownAuthenticator", + message="No extension defines oauth2client/acme, so the collector would refuse to start", ), - want=want( - fnv1.READY_FALSE, - None, - fnv1.Condition( - type="Accepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="SecretNotFound", - message=( - "Secret telemetry-credentials does not exist, " - "so sink primary has no credential to send with" - ), + ), + ), + Case( + name="not ready when the credential Secret is missing", + req=req( + {"sinks": [sink(secret="telemetry-credentials")], "extensions": extensions}, + secrets=[], + ), + want=want( + fnv1.READY_FALSE, + None, + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="SecretNotFound", + message=( + "Secret telemetry-credentials does not exist, so sink primary has no credential to send with" ), - secret="telemetry-credentials", ), + secret="telemetry-credentials", ), - ] - - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) + ), + ] + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """The function reports whether a destination can actually be sent through.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-usages/tests/__init__.py b/functions/compose-usages/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-usages/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-usages/tests/test_fn.py b/functions/compose-usages/tests/test_fn.py index 531ce2c7c..912f59a70 100644 --- a/functions/compose-usages/tests/test_fn.py +++ b/functions/compose-usages/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-usages function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb _NAMESPACE = "test-ns" @@ -131,132 +133,123 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - cases = [ - Case( - name="labels each consumer and composes a Usage per ProviderConfig reference", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=_composite())), - desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), - resources={ - "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), - "gateway-namespace": fnv1.Resource(resource=resource.dict_to_struct(_OBJECT)), - "prometheus": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE_WITH_LABEL)), - "config-map": fnv1.Resource(resource=resource.dict_to_struct(_OBJECT_NO_PC)), - "provider-config-helm": fnv1.Resource(resource=resource.dict_to_struct(_PROVIDER_CONFIG)), - }, +COMPOSE_CASES = [ + Case( + name="labels each consumer and composes a Usage per ProviderConfig reference", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=_composite())), + desired=fnv1.State( + composite=fnv1.Resource(resource=_composite()), + resources={ + "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), + "gateway-namespace": fnv1.Resource(resource=resource.dict_to_struct(_OBJECT)), + "prometheus": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE_WITH_LABEL)), + "config-map": fnv1.Resource(resource=resource.dict_to_struct(_OBJECT_NO_PC)), + "provider-config-helm": fnv1.Resource(resource=resource.dict_to_struct(_PROVIDER_CONFIG)), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=_composite()), + resources={ + "cert-manager": fnv1.Resource( + resource=resource.dict_to_struct(_labelled(_RELEASE, "cert-manager")), ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), - resources={ - "cert-manager": fnv1.Resource( - resource=resource.dict_to_struct(_labelled(_RELEASE, "cert-manager")), - ), - "gateway-namespace": fnv1.Resource( - resource=resource.dict_to_struct(_labelled(_OBJECT, "gateway-namespace")), - ), - # Existing labels are preserved when the consumer label is stamped. - "prometheus": fnv1.Resource( - resource=resource.dict_to_struct(_labelled(_RELEASE_WITH_LABEL, "prometheus")), - ), - # An Object with no providerConfigRef is left untouched, no Usage. - "config-map": fnv1.Resource( - resource=resource.dict_to_struct(_OBJECT_NO_PC), - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_PROVIDER_CONFIG), - ), - "usage-pc-cert-manager": fnv1.Resource( - resource=resource.dict_to_struct( - _usage("helm.m.crossplane.io/v1beta1", "Release", "cert-manager") - ), - ready=fnv1.READY_TRUE, - ), - "usage-pc-gateway-namespace": fnv1.Resource( - resource=resource.dict_to_struct( - _usage("kubernetes.m.crossplane.io/v1alpha1", "Object", "gateway-namespace") - ), - ready=fnv1.READY_TRUE, - ), - "usage-pc-prometheus": fnv1.Resource( - resource=resource.dict_to_struct( - _usage("helm.m.crossplane.io/v1beta1", "Release", "prometheus") - ), - ready=fnv1.READY_TRUE, - ), - }, + "gateway-namespace": fnv1.Resource( + resource=resource.dict_to_struct(_labelled(_OBJECT, "gateway-namespace")), ), - context=structpb.Struct(), - ), - ), - Case( - name="no Usages when the composite has no namespace", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=_composite(namespace=None))), - desired=fnv1.State( - composite=fnv1.Resource(resource=_composite(namespace=None)), - resources={ - "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), - }, + # Existing labels are preserved when the consumer label is stamped. + "prometheus": fnv1.Resource( + resource=resource.dict_to_struct(_labelled(_RELEASE_WITH_LABEL, "prometheus")), ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=_composite(namespace=None)), - resources={ - "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), - }, + # An Object with no providerConfigRef is left untouched, no Usage. + "config-map": fnv1.Resource( + resource=resource.dict_to_struct(_OBJECT_NO_PC), ), - context=structpb.Struct(), - ), - ), - Case( - name="no consumers means no Usages", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=_composite())), - desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), - resources={ - "provider-config-helm": fnv1.Resource(resource=resource.dict_to_struct(_PROVIDER_CONFIG)), - }, + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_PROVIDER_CONFIG), + ), + "usage-pc-cert-manager": fnv1.Resource( + resource=resource.dict_to_struct( + _usage("helm.m.crossplane.io/v1beta1", "Release", "cert-manager") + ), + ready=fnv1.READY_TRUE, ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), - resources={ - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_PROVIDER_CONFIG), - ), - }, + "usage-pc-gateway-namespace": fnv1.Resource( + resource=resource.dict_to_struct( + _usage("kubernetes.m.crossplane.io/v1alpha1", "Object", "gateway-namespace") + ), + ready=fnv1.READY_TRUE, ), - context=structpb.Struct(), - ), + "usage-pc-prometheus": fnv1.Resource( + resource=resource.dict_to_struct( + _usage("helm.m.crossplane.io/v1beta1", "Release", "prometheus") + ), + ready=fnv1.READY_TRUE, + ), + }, ), - ] + context=structpb.Struct(), + ), + ), + Case( + name="no Usages when the composite has no namespace", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=_composite(namespace=None))), + desired=fnv1.State( + composite=fnv1.Resource(resource=_composite(namespace=None)), + resources={ + "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=_composite(namespace=None)), + resources={ + "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), + }, + ), + context=structpb.Struct(), + ), + ), + Case( + name="no consumers means no Usages", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=_composite())), + desired=fnv1.State( + composite=fnv1.Resource(resource=_composite()), + resources={ + "provider-config-helm": fnv1.Resource(resource=resource.dict_to_struct(_PROVIDER_CONFIG)), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=_composite()), + resources={ + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_PROVIDER_CONFIG), + ), + }, + ), + context=structpb.Struct(), + ), + ), +] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction labels consumers and composes their Usages.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-vultr-cluster/tests/test_fn.py b/functions/compose-vultr-cluster/tests/test_fn.py index 436a356c0..608ed76f7 100644 --- a/functions/compose-vultr-cluster/tests/test_fn.py +++ b/functions/compose-vultr-cluster/tests/test_fn.py @@ -14,15 +14,17 @@ """Tests for the compose-vultr-cluster function.""" +import asyncio import dataclasses -import unittest +import json from typing import Any -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.vultrcluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -37,10 +39,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - # Name of the cluster's connection secret. Derived like the function derives # it - the hash suffix depends only on the parent and child names. _KUBECONFIG_SECRET_NAME = resource.child_name("test-cluster", "kubeconfig") @@ -277,277 +275,269 @@ def _observed_unready(desired: dict) -> fnv1.Resource: ) -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function composes VKE cluster infrastructure.""" - cases = [ - Case( - name="cluster composed first; node pools withheld until cluster Ready", - req=_req([_GPU_POOL]), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - }, +COMPOSE_CASES = [ + Case( + name="cluster composed first; node pools withheld until cluster Ready", + req=_req([_GPU_POOL]), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), ), - context=structpb.Struct(), - ), + }, ), - Case( - name="node pools and GPU observer composed once cluster is Ready; autoscaling from maxNodeCount", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), - ), - }, + context=structpb.Struct(), + ), + ), + Case( + name="node pools and GPU observer composed once cluster is Ready; autoscaling from maxNodeCount", + req=_req( + [_GPU_POOL], + observed_resources={ + "cluster": _observed_ready(_cluster()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, ), - context=structpb.Struct(), - ), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config()), + ready=fnv1.READY_TRUE, + ), + "gpu-observer": fnv1.Resource( + resource=resource.dict_to_struct(_gpu_observer()), + ), + }, ), - Case( - name="dependents kept when the cluster Ready condition transiently regresses", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_unready(_cluster()), - "provider-config-kubernetes": _observed_ready(_provider_config()), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), - ), - }, + context=structpb.Struct(), + ), + ), + Case( + name="dependents kept when the cluster Ready condition transiently regresses", + req=_req( + [_GPU_POOL], + observed_resources={ + "cluster": _observed_unready(_cluster()), + "provider-config-kubernetes": _observed_ready(_provider_config()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), ), - context=structpb.Struct(), - ), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config()), + ready=fnv1.READY_TRUE, + ), + "gpu-observer": fnv1.Resource( + resource=resource.dict_to_struct(_gpu_observer()), + ), + }, ), - Case( - name="observed node pool alone keeps dependents composed", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_unready(_cluster()), - "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ready=fnv1.READY_TRUE, - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), - ), - }, + context=structpb.Struct(), + ), + ), + Case( + name="observed node pool alone keeps dependents composed", + req=_req( + [_GPU_POOL], + observed_resources={ + "cluster": _observed_unready(_cluster()), + "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), ), - context=structpb.Struct(), - ), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), + ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config()), + ready=fnv1.READY_TRUE, + ), + "gpu-observer": fnv1.Resource( + resource=resource.dict_to_struct(_gpu_observer()), + ), + }, ), - Case( - name="fixed-size GPU pool", - req=_req( - [ - v1alpha1.NodePool( - name="gpu-l40s", - role="GPU", - plan="vcg-l40s-16c-180g-48vram", - nodeCount=2, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), - ), - ], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, + context=structpb.Struct(), + ), + ), + Case( + name="fixed-size GPU pool", + req=_req( + [ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + plan="vcg-l40s-16c-180g-48vram", + nodeCount=2, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct( - _node_pool( - label="gpu-l40s", - plan="vcg-l40s-16c-180g-48vram", - node_quantity=2, - labels=[ - {"key": "modelplane.ai/pool", "value": "gpu-l40s"}, - {"key": "modelplane.ai/gpu", "value": "nvidia-l40s"}, - {"key": "nvidia.com/gpu.deploy.device-plugin", "value": "false"}, - ], - taints=_GPU_TAINTS, - ), - ), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), + ], + observed_resources={ + "cluster": _observed_ready(_cluster()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, + ), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct( + _node_pool( + label="gpu-l40s", + plan="vcg-l40s-16c-180g-48vram", + node_quantity=2, + labels=[ + {"key": "modelplane.ai/pool", "value": "gpu-l40s"}, + {"key": "modelplane.ai/gpu", "value": "nvidia-l40s"}, + {"key": "nvidia.com/gpu.deploy.device-plugin", "value": "false"}, + ], + taints=_GPU_TAINTS, ), - }, + ), ), - context=structpb.Struct(), - ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config()), + ready=fnv1.READY_TRUE, + ), + "gpu-observer": fnv1.Resource( + resource=resource.dict_to_struct(_gpu_observer()), + ), + }, ), - Case( - name="minNodeCount sets the autoscaler floor; System pool carries no taint", - req=_req( - [ - v1alpha1.NodePool( - name="workers", - role="System", - plan="vc2-6c-16gb", - nodeCount=2, - minNodeCount=2, - maxNodeCount=5, - ), - ], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, + context=structpb.Struct(), + ), + ), + Case( + name="minNodeCount sets the autoscaler floor; System pool carries no taint", + req=_req( + [ + v1alpha1.NodePool( + name="workers", + role="System", + plan="vc2-6c-16gb", + nodeCount=2, + minNodeCount=2, + maxNodeCount=5, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-workers": fnv1.Resource( - resource=resource.dict_to_struct( - _node_pool( - label="workers", - plan="vc2-6c-16gb", - node_quantity=2, - labels=[{"key": "modelplane.ai/pool", "value": "workers"}], - auto_scaler=True, - min_nodes=2, - max_nodes=5, - ), - ), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), + ], + observed_resources={ + "cluster": _observed_ready(_cluster()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, + ), + "node-pool-workers": fnv1.Resource( + resource=resource.dict_to_struct( + _node_pool( + label="workers", + plan="vc2-6c-16gb", + node_quantity=2, + labels=[{"key": "modelplane.ai/pool", "value": "workers"}], + auto_scaler=True, + min_nodes=2, + max_nodes=5, ), - }, + ), ), - context=structpb.Struct(), - ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config()), + ready=fnv1.READY_TRUE, + ), + "gpu-observer": fnv1.Resource( + resource=resource.dict_to_struct(_gpu_observer()), + ), + }, ), - Case( - name="VultrCluster Ready only once the gpu-observer is Ready", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_ready(_cluster()), - "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN), - "gpu-observer": _observed_ready(_gpu_observer()), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ready=fnv1.READY_TRUE, - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), - ready=fnv1.READY_TRUE, - ), - }, + context=structpb.Struct(), + ), + ), + Case( + name="VultrCluster Ready only once the gpu-observer is Ready", + req=_req( + [_GPU_POOL], + observed_resources={ + "cluster": _observed_ready(_cluster()), + "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN), + "gpu-observer": _observed_ready(_gpu_observer()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, ), - context=structpb.Struct(), - ), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), + ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config()), + ready=fnv1.READY_TRUE, + ), + "gpu-observer": fnv1.Resource( + resource=resource.dict_to_struct(_gpu_observer()), + ready=fnv1.READY_TRUE, + ), + }, ), - ] - - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) + context=structpb.Struct(), + ), + ), +] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes VKE cluster infrastructure.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/nix/checks.nix b/nix/checks.nix index 8aa54827a..04e7d4e8f 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -36,8 +36,9 @@ let # Type-check each function with ty. Each function exports its own 'function' # module, so checking all functions at once would let ty resolve one # function's `function.fn` import to another's package. We check each in - # isolation against a venv that provides its dependencies, plus the protobuf - # type stubs ty needs to resolve the SDK's generated Struct and Duration. + # isolation against a venv that provides its dependencies, pytest, which the + # tests import, and the protobuf type stubs ty needs to resolve the SDK's + # generated Struct and Duration. # # Unlike mkFunctionTest, which runs the function module from the venv, ty # checks the source, so we copy function/ and tests/ from the tree. We also @@ -48,6 +49,7 @@ let let venv = pythonSet.mkVirtualEnv "${name}-ty-env" { ${name} = [ ]; + pytest = [ ]; types-protobuf = [ ]; }; in diff --git a/pyproject.toml b/pyproject.toml index 7e1e0eeba..7ae62a000 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -57,6 +57,7 @@ select = [ "TID", # flake8-tidy-imports (no relative imports) "BLE", # flake8-blind-except (no bare except Exception) "ANN", # flake8-annotations (require type annotations) + "PT", # flake8-pytest-style (pytest idioms, no unittest assertions) "RUF", # ruff-specific rules (incl. unused/malformed noqa) ] @@ -75,6 +76,8 @@ allow-star-arg-any = true # Tests use magic values, many parameters, long hardcoded resource dicts, and # boolean fixture toggles passed positionally. "**/tests/**" = ["E501", "PLR2004", "PLR0913", "FBT001"] +# The pytest style rules are for tests. Elsewhere an assert guards an invariant. +"!**/tests/**" = ["PT"] # fn.py uses gRPC's required PascalCase method name. "functions/*/function/fn.py" = ["N802"] # The docs manifest validator is a CLI script; print is its output. From 948ce18b72b12f83c7b7cb153d8872be81e0b7fa Mon Sep 17 00:00:00 2001 From: Nic Cope Date: Mon, 5 Oct 2026 16:15:52 -0700 Subject: [PATCH 4/7] Write each unit test case out in full The tests built their cases in several ways. Some derived a case from another by copying and mutating it, or patched a request after building it. Others built whole requests and responses with helpers, asserted on individual fields, or computed expected values with the code under test or the SDK. A reader often couldn't tell what a case checked without tracing code. Some case names also claimed conditions their input didn't set up, and as sentences they made long test IDs. This commit rewrites the tests to the rules CONTRIBUTING now states, and says why in a comment wherever a test departs from them. Helpers take everything that varies as a required keyword argument, because a defaulted argument had hidden that a case named for an unpinned replica was pinned. A field-level test became a case in its entry point's table, or was deleted where a case already sent the same request. Each case has a short CamelCase name, which is its test ID, and a one-sentence reason, which pytest prints when it fails. Every distinct RunFunction request and response is unchanged. Writing cases out in full takes the tests from about 22,000 lines to 46,000. Towards #473. Signed-off-by: Nic Cope --- CONTRIBUTING.md | 86 +- .../compose-aks-cluster/tests/test_fn.py | 1116 +- .../compose-civo-cluster/tests/test_fn.py | 932 +- .../compose-eks-cluster/tests/test_fn.py | 4303 +++++--- .../compose-gke-cluster/tests/test_fn.py | 1271 +-- .../compose-inference-class/tests/test_fn.py | 20 +- .../tests/test_fn.py | 7922 ++++++++------ .../tests/test_fn.py | 3791 +++++-- .../compose-metric-mapping/tests/test_fn.py | 171 +- .../compose-model-cache/tests/test_fn.py | 1835 ++-- .../tests/test_cel.py | 445 +- .../compose-model-deployment/tests/test_fn.py | 3699 ++++--- .../tests/test_quantity.py | 502 +- .../tests/test_scheduling.py | 6202 ++++++++--- .../tests/test_semver.py | 379 +- .../compose-model-endpoint/tests/test_fn.py | 309 +- .../tests/test_backends.py | 8258 ++++++++++++--- .../compose-model-replica/tests/test_fn.py | 1307 ++- .../compose-model-route/tests/test_fn.py | 2569 +++-- .../compose-model-service/tests/test_fn.py | 716 +- .../compose-nebius-cluster/tests/test_fn.py | 1699 +-- .../tests/test_collector.py | 1589 ++- .../compose-serving-stack/tests/test_fn.py | 9250 ++++++++++++++--- .../tests/test_stacks.py | 465 +- .../tests/test_fn.py | 404 +- functions/compose-usages/tests/test_fn.py | 424 +- .../compose-vultr-cluster/tests/test_fn.py | 894 +- 27 files changed, 42858 insertions(+), 17700 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 74339614d..fa3d3a7a8 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -281,20 +281,22 @@ can have its own `test_.py` too. The canonical form is a table of `Case`s, each running the function on a `RunFunctionRequest` and comparing the whole `RunFunctionResponse` against an -expected one, rather than asserting on individual fields. `compose-usages` is a -small example. The skeleton: +expected one, rather than asserting on individual fields. `compose-model-cache` +is a good example. The skeleton: ```python @dataclasses.dataclass class Case: name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse COMPOSE_CASES = [ Case( - name="describes what this case exercises", + name="ClusterReady", + reason="Once the cluster is ready, the XR reports Ready.", req=fnv1.RunFunctionRequest(...), want=fnv1.RunFunctionResponse(...), ), @@ -310,30 +312,64 @@ def _to_dict(msg: message.Message) -> dict: def test_compose(case: Case) -> None: """RunFunction composes the resources an XR needs.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) + assert _to_dict(got) == _to_dict(case.want), case.reason ``` -Name a table for the test that runs it, and put it just above that test. Each -case becomes its own test, named for the case, so `pytest -k` can select it. -With `got` on the left, pytest's diff shows the expected lines as `-` and the -actual lines as `+`, the same way round as Go's `cmp.Diff(want, got)`. Tests are -plain functions, with no classes, fixtures, or `conftest.py`. They call the -async `RunFunction` with `asyncio.run` rather than needing a plugin, and check -errors with `pytest.raises(..., match=...)`. - -Build the XR with `resource.dict_to_struct(xr.model_dump(exclude_none=True, -mode="json"))` from a generated Pydantic model; build other observed, desired, -and required resources as plain dicts. Because `want` is the whole response, it -must include the parts the function always emits: `meta.ttl` (60s), an empty -`context`, and any conditions, results, and requirements. Give observed -conditions a fixed `lastTransitionTime` so the input is deterministic. Protobuf -maps (`desired.resources`, `requirements.resources`) compare -order-independently, but repeated fields (`conditions`, `results`, status -arrays) must match the order the function emits. - -Some existing tests (`compose-serving-stack`, the second method in -`compose-eks-cluster`) predate this form and assert on individual fields. Don't -model new tests on them. +Name a table for the test that runs it, and put it just above that test. A +`Case` holds its `name` and `reason`, then the inputs of the call under test, +named for its parameters, then `want`. Each case becomes its own test, with its +`name` as its ID, so `pytest -k` can select it. With `got` on the left, pytest's +diff shows the expected lines as `-` and the actual lines as `+`, the same way +round as Go's `cmp.Diff(want, got)`. Tests are plain functions, with no classes, +fixtures, or `conftest.py`. They call the async `RunFunction` with `asyncio.run` +rather than needing a plugin, and check errors with `pytest.raises(..., +match=...)`. + +Cases are data, so a reader should be able to see everything a case asserts by +reading it: + +- **Write each case out in full.** Repetition between cases is fine. Don't + derive one case from another, or from a shared base, by copying and mutating + it, and don't change a request or response once it's built. Pass + requirements, conditions, and results to the constructor. +- **A resource that appears in three or more cases gets a helper,** the XR + included. Count resources by the role they play, such as "the GPU node pool" + or "an endpoint's Backend". An observed resource plays a different role from + the desired resource it reflects, so it gets its own helper. A helper builds + that one resource and returns the `fnv1.Resource` that carries it, or a dict + where another resource embeds it. Everything that varies between the cases + that use it is a keyword argument with no default, so every call shows every + value that varies. A desired resource's readiness is an `fnv1.Ready` value. A + flag that sets an observed resource's Ready condition is a bool. Write a + resource that appears in one or two cases inline. Never write a helper that + builds a whole request, response, map of resources, or case. +- **Give each case a short name and a reason.** The `name` is a few words of + CamelCase, unique in its table, such as `JobComplete`. The `reason` is one + sentence saying what the case's input sets up and what it expects, and the + test passes it as the assertion's message, so pytest prints it when the case + fails. Put any further comment on a case directly above its `Case(`. A short + comment beside one value can explain that value. +- **Compare the whole output, once.** A test that calls the same entry point + with different data belongs in that entry point's table as another case. +- **Write values as literals,** in requests and expectations alike, including + names the function hashes. An expectation computed by code, whether the code + under test or the SDK's `child_name`, passes whatever that code does. +- **Build the XR from its generated model,** with + `resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json", + by_alias=True))`. Write composed and observed resources as dicts in their wire + form. The generated models include schema defaults, so a model doesn't fix its + own wire form: the SDK sends only the fields a function sets, while the API + server fills in the defaults. +- **Say why where a test departs from a rule,** in a comment beside the + departure. + +Because `want` is the whole response, it must include the parts the function +always emits: `meta.ttl` (60s), an empty `context`, and any conditions, results, +and requirements. Give observed conditions a fixed `lastTransitionTime` so the +input is deterministic. Protobuf maps (`desired.resources`, +`requirements.resources`) compare order-independently, but repeated fields +(`conditions`, `results`, status arrays) must match the order the function +emits. `nix flake check` runs every function's tests, and so does `nix run .#test`, outside the sandbox. Name a function to run only its tests, and pass pytest diff --git a/functions/compose-aks-cluster/tests/test_fn.py b/functions/compose-aks-cluster/tests/test_fn.py index e5a28a466..d9579118a 100644 --- a/functions/compose-aks-cluster/tests/test_fn.py +++ b/functions/compose-aks-cluster/tests/test_fn.py @@ -34,133 +34,151 @@ class Case: """A test case for compose-aks-cluster.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse -# Names derived like the function derives them - the hash suffix depends only -# on the input names. -_CLUSTER_NAME = resource.child_name("modelplane-system", "test-cluster", "aks") -_KUBECONFIG_SECRET_NAME = resource.child_name("test-cluster", "kubeconfig") - - -def _xr( - pools: list[v1alpha1.NodePool], - credentials: v1alpha1.Credentials | None = None, -) -> dict: - """An AKSCluster XR with the given node pools, as a request dict.""" - spec = v1alpha1.Spec(location="westeurope", nodePools=pools) - if credentials is not None: - spec.credentials = credentials - return v1alpha1.AKSCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", +def _xr(*, credentials: v1alpha1.Credentials | None, node_pools: list[v1alpha1.NodePool]) -> fnv1.Resource: + """The observed AKSCluster XR, with the given credentials and node pools.""" + xr = v1alpha1.AKSCluster( + metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), + spec=v1alpha1.Spec( + location="westeurope", + credentials=credentials, + nodePools=node_pools, ), - spec=spec, - ).model_dump(exclude_none=True, mode="json") - - -def _req( - pools: list[v1alpha1.NodePool], - observed_resources: dict[str, fnv1.Resource] | None = None, - credentials: v1alpha1.Credentials | None = None, -) -> fnv1.RunFunctionRequest: - return fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(pools, credentials))), - resources=observed_resources or {}, + ) + return fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json", by_alias=True))) + + +def _desired_xr() -> fnv1.Resource: + """The desired XR, publishing its kubeconfig Secret and cache StorageClass.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + ], + "cache": {"storageClassName": "modelplane-rwx-fs"}, + }, + } ), ) -def _resource_group(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "azure.m.upbound.io/v1beta1", - "kind": "ResourceGroup", - "metadata": {"name": _CLUSTER_NAME}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": {"location": "westeurope"}, - }, - } +def _resource_group(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed ResourceGroup.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "azure.m.upbound.io/v1beta1", + "kind": "ResourceGroup", + "metadata": {"name": "modelplane-system-test-cluster-aks-1173e"}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": {"location": "westeurope"}, + }, + } + ), + ready=ready, + ) -def _virtual_network(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "network.azure.m.upbound.io/v1beta1", - "kind": "VirtualNetwork", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "location": "westeurope", - "addressSpace": ["10.0.0.0/16"], - "resourceGroupNameSelector": {"matchControllerRef": True}, - }, - }, - } +def _virtual_network(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed VirtualNetwork.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "network.azure.m.upbound.io/v1beta1", + "kind": "VirtualNetwork", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "location": "westeurope", + "addressSpace": ["10.0.0.0/16"], + "resourceGroupNameSelector": {"matchControllerRef": True}, + }, + }, + } + ), + ready=ready, + ) -def _subnet(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "network.azure.m.upbound.io/v1beta1", - "kind": "Subnet", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "addressPrefixes": ["10.0.0.0/20"], - "resourceGroupNameSelector": {"matchControllerRef": True}, - "virtualNetworkNameSelector": {"matchControllerRef": True}, - }, - }, - } +def _subnet(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Subnet.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "network.azure.m.upbound.io/v1beta1", + "kind": "Subnet", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "addressPrefixes": ["10.0.0.0/20"], + "resourceGroupNameSelector": {"matchControllerRef": True}, + "virtualNetworkNameSelector": {"matchControllerRef": True}, + }, + }, + } + ), + ready=ready, + ) -def _cluster(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "containerservice.azure.m.upbound.io/v1beta1", - "kind": "KubernetesCluster", - "metadata": {"name": _CLUSTER_NAME}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "location": "westeurope", - "kubernetesVersion": "1.34", - "dnsPrefix": _CLUSTER_NAME, - "nodeResourceGroup": f"{_CLUSTER_NAME}-nodes", - "resourceGroupNameSelector": {"matchControllerRef": True}, - "identity": {"type": "SystemAssigned"}, - "defaultNodePool": { - "name": "system", - "vmSize": "Standard_D4s_v5", - "autoScalingEnabled": True, - "minCount": 1, - "maxCount": 2, - "osDiskSizeGb": 100, - "temporaryNameForRotation": "systemtmp", - "nodeLabels": {"modelplane.ai/pool": "system"}, - "vnetSubnetIdSelector": {"matchControllerRef": True}, - }, - "networkProfile": { - "networkPlugin": "azure", - "networkPluginMode": "overlay", - "podCidr": "10.244.0.0/16", - "serviceCidr": "10.96.0.0/16", - "dnsServiceIp": "10.96.0.10", +def _cluster(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed KubernetesCluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "containerservice.azure.m.upbound.io/v1beta1", + "kind": "KubernetesCluster", + "metadata": {"name": "modelplane-system-test-cluster-aks-1173e"}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "location": "westeurope", + "kubernetesVersion": "1.34", + "dnsPrefix": "modelplane-system-test-cluster-aks-1173e", + "nodeResourceGroup": "modelplane-system-test-cluster-aks-1173e-nodes", + "resourceGroupNameSelector": {"matchControllerRef": True}, + "identity": {"type": "SystemAssigned"}, + "defaultNodePool": { + "name": "system", + "vmSize": "Standard_D4s_v5", + "autoScalingEnabled": True, + "minCount": 1, + "maxCount": 2, + "osDiskSizeGb": 100, + "temporaryNameForRotation": "systemtmp", + "nodeLabels": {"modelplane.ai/pool": "system"}, + "vnetSubnetIdSelector": {"matchControllerRef": True}, + }, + "networkProfile": { + "networkPlugin": "azure", + "networkPluginMode": "overlay", + "podCidr": "10.244.0.0/16", + "serviceCidr": "10.96.0.0/16", + "dnsServiceIp": "10.96.0.10", + }, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, }, - }, - "writeConnectionSecretToRef": {"name": _KUBECONFIG_SECRET_NAME}, - }, - } + } + ), + ready=ready, + ) -def _nodepool_gpu( - cred_kind: str = "ClusterProviderConfig", - cred_name: str = "default", - **for_provider_extra: object, -) -> dict: - """A GPU node pool golden, merged with extra forProvider fields.""" - return { +def _nodepool_gpu(*, cred_kind: str, cred_name: str, zones: list[str] | None, ready: fnv1.Ready) -> fnv1.Resource: + """The composed gpuh100 node pool, in zones if given.""" + nodepool = { "apiVersion": "containerservice.azure.m.upbound.io/v1beta1", "kind": "KubernetesClusterNodePool", "metadata": {"annotations": {"crossplane.io/external-name": "gpuh100"}}, @@ -184,420 +202,679 @@ def _nodepool_gpu( "modelplane.ai/pool": "gpuh100", }, "nodeTaints": ["nvidia.com/gpu=true:NoSchedule"], - **for_provider_extra, }, }, } + if zones is not None: + nodepool["spec"]["forProvider"]["zones"] = zones + return fnv1.Resource(resource=resource.dict_to_struct(nodepool), ready=ready) -def _network_operator_release() -> dict: - return { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "Release", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { +def _provider_config(*, api_version: str) -> fnv1.Resource: + """A Ready ProviderConfig that reaches the cluster through its kubeconfig Secret.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": api_version, "kind": "ProviderConfig", - "name": _KUBECONFIG_SECRET_NAME, - }, - "forProvider": { - "chart": { - "name": "network-operator", - "repository": "https://helm.ngc.nvidia.com/nvidia", - "version": "26.4.0", - }, - "namespace": "network-operator", - "values": { - "deployCR": True, - "ofedDriver": {"deploy": True}, - "rdmaSharedDevicePlugin": {"deploy": True}, - # The driver and device plugin must tolerate the GPU taint - # to run on the InfiniBand nodes. - "daemonsets": { - "tolerations": [ - { - "key": "nvidia.com/gpu", - "operator": "Exists", - "effect": "NoSchedule", - }, - ], + "metadata": {"name": "test-cluster-kubeconfig-55b57"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", + "key": "kubeconfig", + }, }, }, - }, - }, - } - - -def _storage_class() -> dict: - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { - "kind": "ProviderConfig", - "name": _KUBECONFIG_SECRET_NAME, - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "storage.k8s.io/v1", - "kind": "StorageClass", - "metadata": {"name": "modelplane-rwx-fs"}, - "provisioner": "file.csi.azure.com", - "parameters": {"skuName": "Premium_LRS"}, - "mountOptions": [ - "dir_mode=0777", - "file_mode=0777", - "uid=0", - "gid=0", - "mfsymlinks", - "cache=strict", - "actimeo=30", - "nosharesock", - ], - "reclaimPolicy": "Delete", - "allowVolumeExpansion": True, - "volumeBindingMode": "WaitForFirstConsumer", - }, - }, - }, - } - - -def _provider_config(api_version: str, kind: str) -> dict: - return { - "apiVersion": api_version, - "kind": kind, - "metadata": {"name": _KUBECONFIG_SECRET_NAME}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "name": _KUBECONFIG_SECRET_NAME, - "namespace": "modelplane-system", - "key": "kubeconfig", - }, - }, - }, - } - - -def _status() -> dict: - return { - "status": { - "secrets": [ - { - "type": "Kubeconfig", - "name": _KUBECONFIG_SECRET_NAME, - "key": "kubeconfig", - }, - ], - "cache": {"storageClassName": "modelplane-rwx-fs"}, - }, - } - - -def _observed_ready(desired: dict) -> fnv1.Resource: - """An observed variant of a desired resource with a Ready=True condition.""" - observed = { - **desired, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - }, - } - return fnv1.Resource(resource=resource.dict_to_struct(observed)) - + } + ), + ready=fnv1.READY_TRUE, + ) -_GPU_POOL = v1alpha1.NodePool( - name="gpuh100", - role="GPU", - vmSize="Standard_ND96isr_H100_v5", - diskSizeGb=200, - nodeCount=1, - minNodeCount=1, - maxNodeCount=4, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), -) -_GPU_POOL_INFINIBAND = v1alpha1.NodePool( - name="gpuh100", - role="GPU", - vmSize="Standard_ND96isr_H100_v5", - diskSizeGb=200, - nodeCount=1, - minNodeCount=1, - maxNodeCount=4, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), - fabric="InfiniBand", -) +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) COMPOSE_CASES = [ + # The StorageClass isn't composed yet: the cluster isn't observed, so the + # ProviderConfigs can't reach it. Case( - name="first pass composes infra; gated resources wait for the cluster", - req=_req([_GPU_POOL]), + name="FirstPass", + reason="With nothing observed, an AKSCluster composes its Azure infrastructure and ProviderConfigs, but not the StorageClass.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpuh100", + role="GPU", + vmSize="Standard_ND96isr_H100_v5", + diskSizeGb=200, + nodeCount=1, + minNodeCount=1, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + ), + ], + ), + ), + ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - # The StorageClass isn't composed yet: the cluster - # isn't observed, so the ProviderConfigs can't - # reach it. - "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "resource-group": _resource_group( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "virtual-network": _virtual_network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodepool-gpuh100": _nodepool_gpu( + cred_kind="ClusterProviderConfig", cred_name="default", zones=None, ready=fnv1.READY_UNSPECIFIED ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), }, ), context=structpb.Struct(), ), ), Case( - name="zones pass through to the node pool", - req=_req( - [ - v1alpha1.NodePool( - name="gpuh100", - role="GPU", - vmSize="Standard_ND96isr_H100_v5", - diskSizeGb=200, - nodeCount=1, - minNodeCount=1, - maxNodeCount=4, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), - zones=[v1alpha1.Zone("1")], + name="NodePoolZones", + reason="An AKSCluster passes a pool's zones through to its KubernetesClusterNodePool.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpuh100", + role="GPU", + vmSize="Standard_ND96isr_H100_v5", + diskSizeGb=200, + nodeCount=1, + minNodeCount=1, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + zones=[v1alpha1.Zone("1")], + ), + ], ), - ] + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "nodepool-gpuh100": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu(zones=["1"])), + "resource-group": _resource_group( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "virtual-network": _virtual_network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodepool-gpuh100": _nodepool_gpu( + cred_kind="ClusterProviderConfig", + cred_name="default", + zones=["1"], + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), }, ), context=structpb.Struct(), ), ), Case( - name="InfiniBand pool composes the network operator once the cluster is observed", - req=_req( - [_GPU_POOL_INFINIBAND], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, + name="InfiniBandFirstPass", + reason="With nothing observed, an AKSCluster with an InfiniBand pool doesn't compose the network operator release yet.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpuh100", + role="GPU", + vmSize="Standard_ND96isr_H100_v5", + diskSizeGb=200, + nodeCount=1, + minNodeCount=1, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + fabric="InfiniBand", + ), + ], + ), + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, + "resource-group": _resource_group( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), - "release-network-operator": fnv1.Resource( - resource=resource.dict_to_struct(_network_operator_release()), + "virtual-network": _virtual_network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - "storage-class-rwx-fs": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class()), - ready=fnv1.READY_TRUE, + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "nodepool-gpuh100": _nodepool_gpu( + cred_kind="ClusterProviderConfig", cred_name="default", zones=None, ready=fnv1.READY_UNSPECIFIED ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), }, ), context=structpb.Struct(), ), ), Case( - name="InfiniBand pool before the cluster is observed gates the network operator", - req=_req([_GPU_POOL_INFINIBAND]), + name="InfiniBandClusterObserved", + reason="With the cluster observed, an AKSCluster with an InfiniBand pool composes the network operator release alongside the StorageClass.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpuh100", + role="GPU", + vmSize="Standard_ND96isr_H100_v5", + diskSizeGb=200, + nodeCount=1, + minNodeCount=1, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + fabric="InfiniBand", + ), + ], + ), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "containerservice.azure.m.upbound.io/v1beta1", + "kind": "KubernetesCluster", + "metadata": {"name": "modelplane-system-test-cluster-aks-1173e"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "location": "westeurope", + "kubernetesVersion": "1.34", + "dnsPrefix": "modelplane-system-test-cluster-aks-1173e", + "nodeResourceGroup": "modelplane-system-test-cluster-aks-1173e-nodes", + "resourceGroupNameSelector": {"matchControllerRef": True}, + "identity": {"type": "SystemAssigned"}, + "defaultNodePool": { + "name": "system", + "vmSize": "Standard_D4s_v5", + "autoScalingEnabled": True, + "minCount": 1, + "maxCount": 2, + "osDiskSizeGb": 100, + "temporaryNameForRotation": "systemtmp", + "nodeLabels": {"modelplane.ai/pool": "system"}, + "vnetSubnetIdSelector": {"matchControllerRef": True}, + }, + "networkProfile": { + "networkPlugin": "azure", + "networkPluginMode": "overlay", + "podCidr": "10.244.0.0/16", + "serviceCidr": "10.96.0.0/16", + "dnsServiceIp": "10.96.0.10", + }, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ), + }, + ), + ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), - "provider-config-kubernetes": fnv1.Resource( + "resource-group": _resource_group( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "virtual-network": _virtual_network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "nodepool-gpuh100": _nodepool_gpu( + cred_kind="ClusterProviderConfig", cred_name="default", zones=None, ready=fnv1.READY_UNSPECIFIED + ), + "release-network-operator": fnv1.Resource( resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "forProvider": { + "chart": { + "name": "network-operator", + "repository": "https://helm.ngc.nvidia.com/nvidia", + "version": "26.4.0", + }, + "namespace": "network-operator", + "values": { + "deployCR": True, + "ofedDriver": {"deploy": True}, + "rdmaSharedDevicePlugin": {"deploy": True}, + # The driver and device plugin must + # tolerate the GPU taint to run on + # the InfiniBand nodes. + "daemonsets": { + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + }, + ], + }, + }, + }, + }, + } ), - ready=fnv1.READY_TRUE, ), - "provider-config-helm": fnv1.Resource( + "storage-class-rwx-fs": fnv1.Resource( resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "storage.k8s.io/v1", + "kind": "StorageClass", + "metadata": {"name": "modelplane-rwx-fs"}, + "provisioner": "file.csi.azure.com", + "parameters": {"skuName": "Premium_LRS"}, + "mountOptions": [ + "dir_mode=0777", + "file_mode=0777", + "uid=0", + "gid=0", + "mfsymlinks", + "cache=strict", + "actimeo=30", + "nosharesock", + ], + "reclaimPolicy": "Delete", + "allowVolumeExpansion": True, + "volumeBindingMode": "WaitForFirstConsumer", + }, + }, + }, + } ), ready=fnv1.READY_TRUE, ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), }, ), context=structpb.Struct(), ), ), + # The ProviderConfigs reach the cluster through its kubeconfig, not the + # cloud credentials. Case( - name="custom credentials flow through to all cloud MRs", - req=_req( - [_GPU_POOL], - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="my-azure-account", + name="CustomCredentials", + reason="The ProviderConfig an AKSCluster's spec.credentials names becomes every Azure managed resource's providerConfigRef.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-azure-account"), + node_pools=[ + v1alpha1.NodePool( + name="gpuh100", + role="GPU", + vmSize="Standard_ND96isr_H100_v5", + diskSizeGb=200, + nodeCount=1, + minNodeCount=1, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + ), + ], + ), ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "resource-group": fnv1.Resource( - resource=resource.dict_to_struct(_resource_group("ProviderConfig", "my-azure-account")), + "resource-group": _resource_group( + cred_kind="ProviderConfig", cred_name="my-azure-account", ready=fnv1.READY_UNSPECIFIED ), - "virtual-network": fnv1.Resource( - resource=resource.dict_to_struct(_virtual_network("ProviderConfig", "my-azure-account")), + "virtual-network": _virtual_network( + cred_kind="ProviderConfig", cred_name="my-azure-account", ready=fnv1.READY_UNSPECIFIED ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet("ProviderConfig", "my-azure-account")), + "subnet": _subnet( + cred_kind="ProviderConfig", cred_name="my-azure-account", ready=fnv1.READY_UNSPECIFIED ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster("ProviderConfig", "my-azure-account")), + "cluster": _cluster( + cred_kind="ProviderConfig", cred_name="my-azure-account", ready=fnv1.READY_UNSPECIFIED ), - "nodepool-gpuh100": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu("ProviderConfig", "my-azure-account")), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "nodepool-gpuh100": _nodepool_gpu( + cred_kind="ProviderConfig", + cred_name="my-azure-account", + zones=None, + ready=fnv1.READY_UNSPECIFIED, ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), }, ), context=structpb.Struct(), ), ), Case( - name="marks managed resources ready from observed conditions", - req=_req( - [_GPU_POOL], - observed_resources={ - "resource-group": _observed_ready(_resource_group()), - "virtual-network": _observed_ready(_virtual_network()), - "subnet": _observed_ready(_subnet()), - "cluster": _observed_ready(_cluster()), - "nodepool-gpuh100": _observed_ready(_nodepool_gpu()), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + name="ResourcesReady", + reason="With the cluster and its other Azure resources observed Ready, an AKSCluster marks them ready and composes the StorageClass.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpuh100", + role="GPU", + vmSize="Standard_ND96isr_H100_v5", + diskSizeGb=200, + nodeCount=1, + minNodeCount=1, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + ), + ], + ), resources={ "resource-group": fnv1.Resource( - resource=resource.dict_to_struct(_resource_group()), - ready=fnv1.READY_TRUE, + resource=resource.dict_to_struct( + { + "apiVersion": "azure.m.upbound.io/v1beta1", + "kind": "ResourceGroup", + "metadata": {"name": "modelplane-system-test-cluster-aks-1173e"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": {"location": "westeurope"}, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), ), "virtual-network": fnv1.Resource( - resource=resource.dict_to_struct(_virtual_network()), - ready=fnv1.READY_TRUE, + resource=resource.dict_to_struct( + { + "apiVersion": "network.azure.m.upbound.io/v1beta1", + "kind": "VirtualNetwork", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "location": "westeurope", + "addressSpace": ["10.0.0.0/16"], + "resourceGroupNameSelector": {"matchControllerRef": True}, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), ), "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ready=fnv1.READY_TRUE, + resource=resource.dict_to_struct( + { + "apiVersion": "network.azure.m.upbound.io/v1beta1", + "kind": "Subnet", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "addressPrefixes": ["10.0.0.0/20"], + "resourceGroupNameSelector": {"matchControllerRef": True}, + "virtualNetworkNameSelector": {"matchControllerRef": True}, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), ), "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - # The cluster is observed, so the StorageClass is - # composed too. - "storage-class-rwx-fs": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class()), - ready=fnv1.READY_TRUE, + resource=resource.dict_to_struct( + { + "apiVersion": "containerservice.azure.m.upbound.io/v1beta1", + "kind": "KubernetesCluster", + "metadata": {"name": "modelplane-system-test-cluster-aks-1173e"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "location": "westeurope", + "kubernetesVersion": "1.34", + "dnsPrefix": "modelplane-system-test-cluster-aks-1173e", + "nodeResourceGroup": "modelplane-system-test-cluster-aks-1173e-nodes", + "resourceGroupNameSelector": {"matchControllerRef": True}, + "identity": {"type": "SystemAssigned"}, + "defaultNodePool": { + "name": "system", + "vmSize": "Standard_D4s_v5", + "autoScalingEnabled": True, + "minCount": 1, + "maxCount": 2, + "osDiskSizeGb": 100, + "temporaryNameForRotation": "systemtmp", + "nodeLabels": {"modelplane.ai/pool": "system"}, + "vnetSubnetIdSelector": {"matchControllerRef": True}, + }, + "networkProfile": { + "networkPlugin": "azure", + "networkPluginMode": "overlay", + "podCidr": "10.244.0.0/16", + "serviceCidr": "10.96.0.0/16", + "dnsServiceIp": "10.96.0.10", + }, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), ), "nodepool-gpuh100": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu()), - ready=fnv1.READY_TRUE, - ), - "provider-config-kubernetes": fnv1.Resource( resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + { + "apiVersion": "containerservice.azure.m.upbound.io/v1beta1", + "kind": "KubernetesClusterNodePool", + "metadata": {"annotations": {"crossplane.io/external-name": "gpuh100"}}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "managementPolicies": ["Observe", "Create", "Update", "Delete"], + "initProvider": {"nodeCount": 1}, + "forProvider": { + "kubernetesClusterIdSelector": {"matchControllerRef": True}, + "vnetSubnetIdSelector": {"matchControllerRef": True}, + "mode": "User", + "vmSize": "Standard_ND96isr_H100_v5", + "osDiskSizeGb": 200, + "orchestratorVersion": "1.34", + "autoScalingEnabled": True, + "minCount": 1, + "maxCount": 4, + "gpuDriver": "Install", + "nodeLabels": { + "modelplane.ai/gpu": "nvidia-h100", + "modelplane.ai/pool": "gpuh100", + }, + "nodeTaints": ["nvidia.com/gpu=true:NoSchedule"], + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } ), - ready=fnv1.READY_TRUE, ), - "provider-config-helm": fnv1.Resource( + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "resource-group": _resource_group( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "virtual-network": _virtual_network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "subnet": _subnet(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "cluster": _cluster(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "storage-class-rwx-fs": fnv1.Resource( resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "storage.k8s.io/v1", + "kind": "StorageClass", + "metadata": {"name": "modelplane-rwx-fs"}, + "provisioner": "file.csi.azure.com", + "parameters": {"skuName": "Premium_LRS"}, + "mountOptions": [ + "dir_mode=0777", + "file_mode=0777", + "uid=0", + "gid=0", + "mfsymlinks", + "cache=strict", + "actimeo=30", + "nosharesock", + ], + "reclaimPolicy": "Delete", + "allowVolumeExpansion": True, + "volumeBindingMode": "WaitForFirstConsumer", + }, + }, + }, + } ), ready=fnv1.READY_TRUE, ), + "nodepool-gpuh100": _nodepool_gpu( + cred_kind="ClusterProviderConfig", cred_name="default", zones=None, ready=fnv1.READY_TRUE + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), }, ), context=structpb.Struct(), @@ -606,13 +883,8 @@ def _observed_ready(desired: dict) -> fnv1.Resource: ] -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - @pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: - """The function composes AKS cluster infrastructure.""" + """RunFunction composes AKS cluster infrastructure.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-civo-cluster/tests/test_fn.py b/functions/compose-civo-cluster/tests/test_fn.py index a6bed72dd..c0ae9ae55 100644 --- a/functions/compose-civo-cluster/tests/test_fn.py +++ b/functions/compose-civo-cluster/tests/test_fn.py @@ -35,295 +35,303 @@ class Case: """A test case for compose-civo-cluster.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse -# Name of the cluster's connection secret. Derived like the function derives -# it - the hash suffix depends only on the parent and child names. -_KUBECONFIG_SECRET_NAME = resource.child_name("test-cluster", "kubeconfig") - -# The system node pool injected inline into every cluster. -_SYSTEM_POOL = { - "label": "system", - "size": "g4p.kube.small", - "nodeCount": 2, - "labels": {"modelplane.ai/pool": "system"}, -} - -# The taint every GPU pool carries. -_GPU_TAINT = [ - {"key": "nvidia.com/gpu", "value": "true", "effect": "NoSchedule"}, -] - -# Civo IDs the provider writes to external-name annotations once resources -# exist. The autoscaler addresses node groups by pool ID. -_CLUSTER_ID = "11111111-2222-3333-4444-555555555555" -_GPU_POOL_ID = "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee" - - -def _xr(pools: list[v1alpha1.NodePool]) -> dict: - """A CivoCluster XR with the given node pools, as a request dict.""" - return v1alpha1.CivoCluster( +def _xr(*, credentials: v1alpha1.Credentials | None, node_pools: list[v1alpha1.NodePool]) -> fnv1.Resource: + """The observed CivoCluster XR in LON1, with the given credentials and node pools.""" + xr = v1alpha1.CivoCluster( metadata=metav1.ObjectMeta( name="test-cluster", namespace="modelplane-system", ), spec=v1alpha1.Spec( region="LON1", - nodePools=pools, + credentials=credentials, + nodePools=node_pools, ), - ).model_dump(exclude_none=True, mode="json") - - -def _req( - pools: list[v1alpha1.NodePool], - observed_resources: dict[str, fnv1.Resource] | None = None, -) -> fnv1.RunFunctionRequest: - return fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(pools))), - resources=observed_resources or {}, + ) + return fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json", by_alias=True))) + + +def _desired_xr() -> fnv1.Resource: + """The desired XR, publishing the cluster's kubeconfig Secret.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + ], + }, + } ), ) -def _network( - cred_kind: str = "ClusterProviderConfig", - cred_name: str = "default", -) -> dict: - """A Network golden.""" - return { - "apiVersion": "vpc.civo.m.upbound.io/v1beta1", - "kind": "Network", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "label": "test-cluster", - "region": "LON1", - }, - }, - } - - -def _firewall( - cred_kind: str = "ClusterProviderConfig", - cred_name: str = "default", -) -> dict: - """A Firewall golden with Civo's default rules.""" - return { - "apiVersion": "vpc.civo.m.upbound.io/v1beta1", - "kind": "Firewall", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "name": "test-cluster", - "region": "LON1", - "createDefaultRules": True, - "networkIdSelector": {"matchControllerRef": True}, - }, - }, - } +def _network(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The composed Network.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vpc.civo.m.upbound.io/v1beta1", + "kind": "Network", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "label": "test-cluster", + "region": "LON1", + }, + }, + } + ), + ) -def _cluster( - cred_kind: str = "ClusterProviderConfig", - cred_name: str = "default", -) -> dict: - """A Cluster golden with only the system pool.""" - return { - "apiVersion": "kubernetes.civo.m.upbound.io/v1beta1", - "kind": "Cluster", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "name": "test-cluster", - "region": "LON1", - "cni": "cilium", - "applications": "-traefik2-nodeport", - "writeKubeconfig": True, - "networkIdSelector": {"matchControllerRef": True}, - "firewallIdSelector": {"matchControllerRef": True}, - "pools": _SYSTEM_POOL, - }, - "writeConnectionSecretToRef": {"name": _KUBECONFIG_SECRET_NAME}, - }, - } +def _firewall(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The composed Firewall, with Civo's default rules.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vpc.civo.m.upbound.io/v1beta1", + "kind": "Firewall", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "name": "test-cluster", + "region": "LON1", + "createDefaultRules": True, + "networkIdSelector": {"matchControllerRef": True}, + }, + }, + } + ), + ) -def _provider_config_helm() -> dict: - """A provider-helm ProviderConfig golden pointing at the kubeconfig.""" - return { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "ProviderConfig", - "metadata": { - "name": _KUBECONFIG_SECRET_NAME, - "namespace": "modelplane-system", - }, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": _KUBECONFIG_SECRET_NAME, - "key": "kubeconfig", +def _cluster(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Civo Cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.civo.m.upbound.io/v1beta1", + "kind": "Cluster", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "name": "test-cluster", + "region": "LON1", + "cni": "cilium", + "applications": "-traefik2-nodeport", + "writeKubeconfig": True, + "networkIdSelector": {"matchControllerRef": True}, + "firewallIdSelector": {"matchControllerRef": True}, + # The system node pool the function adds inline to every cluster. + "pools": { + "label": "system", + "size": "g4p.kube.small", + "nodeCount": 2, + "labels": {"modelplane.ai/pool": "system"}, + }, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, }, - }, - }, - } + } + ), + ready=ready, + ) -def _autoscaler(groups: list[dict]) -> dict: - """The cluster autoscaler Release golden for the given node groups.""" - return { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "Release", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { - "kind": "ProviderConfig", - "name": _KUBECONFIG_SECRET_NAME, - }, - "forProvider": { - "chart": { - "name": "cluster-autoscaler", - "repository": "https://kubernetes.github.io/autoscaler", - "version": "9.57.0", +def _observed_cluster(*, cred_kind: str, cred_name: str, ready: bool) -> fnv1.Resource: + """The Civo Cluster as observed, with its Civo ID, and a Ready condition that's True if ready and False if not.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.civo.m.upbound.io/v1beta1", + "kind": "Cluster", + "metadata": { + "annotations": {"crossplane.io/external-name": "11111111-2222-3333-4444-555555555555"}, }, - "namespace": "kube-system", - "values": { - "cloudProvider": "civo", - "autoscalingGroups": groups, - "secretKeyRefNameOverride": "civo-api-access", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "name": "test-cluster", + "region": "LON1", + "cni": "cilium", + "applications": "-traefik2-nodeport", + "writeKubeconfig": True, + "networkIdSelector": {"matchControllerRef": True}, + "firewallIdSelector": {"matchControllerRef": True}, + "pools": { + "label": "system", + "size": "g4p.kube.small", + "nodeCount": 2, + "labels": {"modelplane.ai/pool": "system"}, + }, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, }, - }, - }, - } + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True" if ready else "False", + "reason": "Available" if ready else "Unavailable", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ) -def _node_pool( - label: str, - size: str, - node_count: int, - labels: dict[str, str], - taint: list | None = None, - cred_kind: str = "ClusterProviderConfig", - cred_name: str = "default", +def _node_pool_gpu( *, - autoscaled: bool = False, -) -> dict: - """A NodePool golden. Autoscaled pools seed nodeCount via initProvider - so the autoscaler owns it after creation; fixed pools keep it in - forProvider.""" - fp: dict[str, Any] = { - "label": label, - "size": size, + management_policies: list[str] | None, + init_node_count: int | None, + node_count: int | None, + ready: fnv1.Ready, +) -> fnv1.Resource: + """The composed NodePool for the gpu-l40s pool. + + An autoscaled pool's nodeCount is in initProvider, and its management + policies leave out LateInitialize, so the cluster autoscaler owns the count + once the pool exists. A fixed-size pool's nodeCount is in forProvider. + """ + for_provider: dict[str, Any] = { + "label": "gpu-l40s", + "size": "an.g1.l40s.kube.x1", "region": "LON1", - "labels": labels, + "labels": { + "modelplane.ai/pool": "gpu-l40s", + "modelplane.ai/gpu": "nvidia-l40s", + }, "clusterIdSelector": {"matchControllerRef": True}, + "taint": [{"key": "nvidia.com/gpu", "value": "true", "effect": "NoSchedule"}], } - if taint: - fp["taint"] = taint + if node_count is not None: + for_provider["nodeCount"] = node_count spec: dict[str, Any] = { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": fp, - } - if autoscaled: - spec["managementPolicies"] = ["Observe", "Create", "Update", "Delete"] - spec["initProvider"] = {"nodeCount": node_count} - else: - fp["nodeCount"] = node_count - return { - "apiVersion": "kubernetes.civo.m.upbound.io/v1beta1", - "kind": "NodePool", - "spec": spec, + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": for_provider, } + if management_policies is not None: + spec["managementPolicies"] = management_policies + if init_node_count is not None: + spec["initProvider"] = {"nodeCount": init_node_count} + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.civo.m.upbound.io/v1beta1", + "kind": "NodePool", + "spec": spec, + } + ), + ready=ready, + ) -def _status() -> dict: - return { - "status": { - "secrets": [ - { - "type": "Kubeconfig", - "name": _KUBECONFIG_SECRET_NAME, - "key": "kubeconfig", +def _provider_config() -> fnv1.Resource: + """The composed provider-helm ProviderConfig for the cluster, which is always ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "metadata": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", }, - ], - }, - } - - -def _observed(desired: dict, ready: str, external_name: str | None = None) -> fnv1.Resource: - """An observed variant of a desired resource with a Ready condition and - optionally the external-name annotation the provider sets.""" - observed: dict[str, Any] = { - **desired, - "status": { - "conditions": [ - { - "type": "Ready", - "status": ready, - "reason": "Available" if ready == "True" else "Unavailable", - "lastTransitionTime": "2024-01-01T00:00:00Z", + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + }, }, - ], - }, - } - if external_name is not None: - metadata = {**observed.get("metadata", {})} - metadata["annotations"] = {"crossplane.io/external-name": external_name} - observed["metadata"] = metadata - return fnv1.Resource(resource=resource.dict_to_struct(observed)) - - -def _observed_ready(desired: dict, external_name: str | None = None) -> fnv1.Resource: - return _observed(desired, "True", external_name) - + } + ), + ready=fnv1.READY_TRUE, + ) -def _observed_unready(desired: dict, external_name: str | None = None) -> fnv1.Resource: - return _observed(desired, "False", external_name) +def _autoscaler_release(*, autoscaling_groups: list[dict]) -> fnv1.Resource: + """The composed cluster autoscaler Release, scaling the given node groups.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "forProvider": { + "chart": { + "name": "cluster-autoscaler", + "repository": "https://kubernetes.github.io/autoscaler", + "version": "9.57.0", + }, + "namespace": "kube-system", + "values": { + "cloudProvider": "civo", + "autoscalingGroups": autoscaling_groups, + "secretKeyRefNameOverride": "civo-api-access", + }, + }, + }, + } + ), + ) -_GPU_POOL = v1alpha1.NodePool( - name="gpu-l40s", - role="GPU", - size="an.g1.l40s.kube.x1", - maxNodeCount=4, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), -) -_GPU_POOL_GOLDEN = _node_pool( - label="gpu-l40s", - size="an.g1.l40s.kube.x1", - node_count=1, - labels={ - "modelplane.ai/pool": "gpu-l40s", - "modelplane.ai/gpu": "nvidia-l40s", - }, - taint=_GPU_TAINT, - autoscaled=True, -) +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) COMPOSE_CASES = [ Case( - name="network, firewall and cluster composed first; node pools withheld until cluster Ready", - req=_req([_GPU_POOL]), + name="FirstPass", + reason="With nothing observed, a CivoCluster composes the network, firewall and cluster, and withholds the node pools, ProviderConfig and autoscaler until the cluster is Ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + size="an.g1.l40s.kube.x1", + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], + ), + ), + ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ), - "firewall": fnv1.Resource( - resource=resource.dict_to_struct(_firewall()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), + "network": _network(cred_kind="ClusterProviderConfig", cred_name="default"), + "firewall": _firewall(cred_kind="ClusterProviderConfig", cred_name="default"), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), }, ), @@ -331,259 +339,312 @@ def _observed_unready(desired: dict, external_name: str | None = None) -> fnv1.R ), ), Case( - name="node pools and provider config composed once cluster is Ready; autoscaler has no groups until pool IDs observed", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_ready(_cluster(), external_name=_CLUSTER_ID), - }, + name="ClusterReady", + reason="With the cluster observed Ready, a CivoCluster composes the node pools, ProviderConfig and autoscaler, which has no node groups until a pool's Civo ID is observed.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + size="an.g1.l40s.kube.x1", + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], + ), + resources={ + "cluster": _observed_cluster(cred_kind="ClusterProviderConfig", cred_name="default", ready=True), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ), - "firewall": fnv1.Resource( - resource=resource.dict_to_struct(_firewall()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - "release-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler([])), - ), + "network": _network(cred_kind="ClusterProviderConfig", cred_name="default"), + "firewall": _firewall(cred_kind="ClusterProviderConfig", cred_name="default"), + "cluster": _cluster(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "node-pool-gpu-l40s": _node_pool_gpu( + management_policies=["Observe", "Create", "Update", "Delete"], + init_node_count=1, + node_count=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-helm": _provider_config(), + "release-cluster-autoscaler": _autoscaler_release(autoscaling_groups=[]), }, ), context=structpb.Struct(), ), ), + # The autoscaler addresses node groups by the pool ID the provider writes + # to the NodePool's external-name annotation once the pool exists. The pool + # leaves nodeCount at its default of 1, so this case can't tell a floor + # taken from nodeCount from one fixed at 1. Case( - name="autoscaler release composed from observed pool IDs", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_ready(_cluster(), external_name=_CLUSTER_ID), - "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN, external_name=_GPU_POOL_ID), - }, + name="PoolIDObserved", + reason="With the GPU pool observed Ready under its Civo ID, a CivoCluster marks the pool ready and has the autoscaler scale that ID from one node to its maxNodeCount.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + size="an.g1.l40s.kube.x1", + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], + ), + resources={ + "cluster": _observed_cluster(cred_kind="ClusterProviderConfig", cred_name="default", ready=True), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.civo.m.upbound.io/v1beta1", + "kind": "NodePool", + "metadata": { + "annotations": { + "crossplane.io/external-name": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", + }, + }, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "managementPolicies": ["Observe", "Create", "Update", "Delete"], + "initProvider": {"nodeCount": 1}, + "forProvider": { + "label": "gpu-l40s", + "size": "an.g1.l40s.kube.x1", + "region": "LON1", + "labels": { + "modelplane.ai/pool": "gpu-l40s", + "modelplane.ai/gpu": "nvidia-l40s", + }, + "clusterIdSelector": {"matchControllerRef": True}, + "taint": [{"key": "nvidia.com/gpu", "value": "true", "effect": "NoSchedule"}], + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ), - "firewall": fnv1.Resource( - resource=resource.dict_to_struct(_firewall()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), + "network": _network(cred_kind="ClusterProviderConfig", cred_name="default"), + "firewall": _firewall(cred_kind="ClusterProviderConfig", cred_name="default"), + "cluster": _cluster(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "node-pool-gpu-l40s": _node_pool_gpu( + management_policies=["Observe", "Create", "Update", "Delete"], + init_node_count=1, + node_count=None, ready=fnv1.READY_TRUE, ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - "release-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct( - _autoscaler( - [{"name": _GPU_POOL_ID, "minSize": 1, "maxSize": 4}], - ), - ), + "provider-config-helm": _provider_config(), + "release-cluster-autoscaler": _autoscaler_release( + autoscaling_groups=[ + {"name": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", "minSize": 1, "maxSize": 4} + ], ), }, ), context=structpb.Struct(), ), ), + # The observed ProviderConfig shows the dependents were composed before. + # Dropping them would delete the NodePools, and deprovision their nodes. Case( - name="dependents kept when the cluster Ready condition transiently regresses", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_unready(_cluster(), external_name=_CLUSTER_ID), - "provider-config-helm": _observed_ready(_provider_config_helm()), - }, + name="ClusterReadyRegressed", + reason="With the cluster's Ready condition regressed to False but its ProviderConfig observed, a CivoCluster keeps the node pools, ProviderConfig and autoscaler composed.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + size="an.g1.l40s.kube.x1", + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], + ), + resources={ + "cluster": _observed_cluster(cred_kind="ClusterProviderConfig", cred_name="default", ready=False), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "metadata": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", + }, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ), - "firewall": fnv1.Resource( - resource=resource.dict_to_struct(_firewall()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - "release-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler([])), - ), + "network": _network(cred_kind="ClusterProviderConfig", cred_name="default"), + "firewall": _firewall(cred_kind="ClusterProviderConfig", cred_name="default"), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "node-pool-gpu-l40s": _node_pool_gpu( + management_policies=["Observe", "Create", "Update", "Delete"], + init_node_count=1, + node_count=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-helm": _provider_config(), + "release-cluster-autoscaler": _autoscaler_release(autoscaling_groups=[]), }, ), context=structpb.Struct(), ), ), Case( - name="fixed-size GPU pool composes the autoscaler release with no node groups", - req=_req( - [ - v1alpha1.NodePool( - name="gpu-l40s", - role="GPU", - size="an.g1.l40s.kube.x1", - nodeCount=2, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + name="FixedSizePool", + reason="A CivoCluster with a GPU pool that sets nodeCount but no maxNodeCount puts that nodeCount in forProvider under the default management policies, and still composes the autoscaler with no node groups.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + size="an.g1.l40s.kube.x1", + nodeCount=2, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], ), - ], - observed_resources={ - "cluster": _observed_ready(_cluster(), external_name=_CLUSTER_ID), - }, + resources={ + "cluster": _observed_cluster(cred_kind="ClusterProviderConfig", cred_name="default", ready=True), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ), - "firewall": fnv1.Resource( - resource=resource.dict_to_struct(_firewall()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct( - _node_pool( - label="gpu-l40s", - size="an.g1.l40s.kube.x1", - node_count=2, - labels={ - "modelplane.ai/pool": "gpu-l40s", - "modelplane.ai/gpu": "nvidia-l40s", - }, - taint=_GPU_TAINT, - ), - ), - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - "release-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler([])), - ), + "network": _network(cred_kind="ClusterProviderConfig", cred_name="default"), + "firewall": _firewall(cred_kind="ClusterProviderConfig", cred_name="default"), + "cluster": _cluster(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "node-pool-gpu-l40s": _node_pool_gpu( + management_policies=None, + init_node_count=None, + node_count=2, + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-helm": _provider_config(), + "release-cluster-autoscaler": _autoscaler_release(autoscaling_groups=[]), }, ), context=structpb.Struct(), ), ), + # The helm ProviderConfig and the Release reach the cluster through its + # kubeconfig, so they don't carry the Civo credentials. Case( - name="System pool carries no taint; credentials override propagates", + name="CustomCredentials", + reason="A CivoCluster naming its own ProviderConfig gets it on every Civo resource, and its System-role pool gets no GPU label or taint.", req=fnv1.RunFunctionRequest( observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.CivoCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - region="LON1", - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="team-a", - ), - nodePools=[ - v1alpha1.NodePool( - name="workers", - role="System", - size="g4p.kube.small", - nodeCount=2, - ), - ], - ), - ).model_dump(exclude_none=True, mode="json"), - ), + composite=_xr( + credentials=v1alpha1.Credentials(type="ProviderConfig", name="team-a"), + node_pools=[ + v1alpha1.NodePool( + name="workers", + role="System", + size="g4p.kube.small", + nodeCount=2, + ), + ], ), resources={ - "cluster": _observed_ready( - _cluster(cred_kind="ProviderConfig", cred_name="team-a"), - external_name=_CLUSTER_ID, - ), + "cluster": _observed_cluster(cred_kind="ProviderConfig", cred_name="team-a", ready=True), }, ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct( - _network(cred_kind="ProviderConfig", cred_name="team-a"), - ), - ), - "firewall": fnv1.Resource( - resource=resource.dict_to_struct( - _firewall(cred_kind="ProviderConfig", cred_name="team-a"), - ), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct( - _cluster(cred_kind="ProviderConfig", cred_name="team-a"), - ), - ready=fnv1.READY_TRUE, - ), + "network": _network(cred_kind="ProviderConfig", cred_name="team-a"), + "firewall": _firewall(cred_kind="ProviderConfig", cred_name="team-a"), + "cluster": _cluster(cred_kind="ProviderConfig", cred_name="team-a", ready=fnv1.READY_TRUE), "node-pool-workers": fnv1.Resource( resource=resource.dict_to_struct( - _node_pool( - label="workers", - size="g4p.kube.small", - node_count=2, - labels={"modelplane.ai/pool": "workers"}, - cred_kind="ProviderConfig", - cred_name="team-a", - ), + { + "apiVersion": "kubernetes.civo.m.upbound.io/v1beta1", + "kind": "NodePool", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "team-a"}, + "forProvider": { + "label": "workers", + "size": "g4p.kube.small", + "region": "LON1", + "labels": {"modelplane.ai/pool": "workers"}, + "clusterIdSelector": {"matchControllerRef": True}, + "nodeCount": 2, + }, + }, + } ), ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - "release-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler([])), - ), + "provider-config-helm": _provider_config(), + "release-cluster-autoscaler": _autoscaler_release(autoscaling_groups=[]), }, ), context=structpb.Struct(), @@ -592,13 +653,8 @@ def _observed_unready(desired: dict, external_name: str | None = None) -> fnv1.R ] -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - @pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: - """The function composes Civo cluster infrastructure.""" + """RunFunction composes a Civo cluster's infrastructure.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-eks-cluster/tests/test_fn.py b/functions/compose-eks-cluster/tests/test_fn.py index 9658abec0..d08a6a7f6 100644 --- a/functions/compose-eks-cluster/tests/test_fn.py +++ b/functions/compose-eks-cluster/tests/test_fn.py @@ -17,7 +17,6 @@ import asyncio import dataclasses import json -from typing import Any import pytest from crossplane.function import resource @@ -35,1608 +34,2906 @@ class Case: """A test case for compose-eks-cluster.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse -_KUBECONFIG_SECRET = "test-cluster-kubeconfig-55b57" -_SUBNET_A = "test-cluster-subnet-us-west-2a-952dc" -_SUBNET_B = "test-cluster-subnet-us-west-2b-2b80f" -_SUBNET_C = "test-cluster-subnet-us-west-2c-03273" -_PRIVATE_SUBNET_A = "test-cluster-private-subnet-us-west-2a-6a89f" -_PRIVATE_SUBNET_B = "test-cluster-private-subnet-us-west-2b-b7832" -_PRIVATE_SUBNET_C = "test-cluster-private-subnet-us-west-2c-ef57d" - - -def _xr(credentials: v1alpha1.Credentials | None = None) -> v1alpha1.EKSCluster: - return v1alpha1.EKSCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), +def _xr(*, credentials: v1alpha1.Credentials | None, node_pools: list[v1alpha1.NodePool]) -> fnv1.Resource: + """The observed EKSCluster XR in us-west-2, with the given credentials and node pools.""" + xr = v1alpha1.EKSCluster( + metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), spec=v1alpha1.Spec( region="us-west-2", credentials=credentials, - nodePools=[ - v1alpha1.NodePool( - name="gpu-l4", - role="GPU", - instanceType="g6.xlarge", - nodeCount=1, - minNodeCount=0, - maxNodeCount=4, - gpu=v1alpha1.Gpu( - acceleratorType="nvidia-l4", - ), - zones=[v1alpha1.Zone("us-west-2a"), v1alpha1.Zone("us-west-2b")], - ), - ], + nodePools=node_pools, ), ) + return fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json", by_alias=True))) -# Launch template name is derived the same way the function derives it, so -# the test can't drift from the function's child_name hashing. -_LAUNCH_TEMPLATE_NAME = resource.child_name("test-cluster", "lt-gpu-h200") -_CAPACITY_RESERVATION_ID = "cr-0123456789abcdef0" - - -def _xr_capacity_block() -> v1alpha1.EKSCluster: - return v1alpha1.EKSCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - region="us-west-2", - nodePools=[ - v1alpha1.NodePool( - name="gpu-h200", - role="GPU", - instanceType="p5en.48xlarge", - nodeCount=2, - minNodeCount=0, - maxNodeCount=2, - diskSizeGb=1024, - gpu=v1alpha1.Gpu( - acceleratorType="nvidia-h200", - ), - capacityBlock=v1alpha1.CapacityBlock( - capacityReservationId=_CAPACITY_RESERVATION_ID, - ), - zones=[v1alpha1.Zone("us-west-2a")], - ), - ], +def _desired_xr() -> fnv1.Resource: + """The desired XR, publishing its kubeconfig Secret and cache StorageClass.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + # `type` is emitted because the function sets it explicitly + # on the Status model, so update_status (exclude_unset) + # keeps it rather than dropping it as an unset field. + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + ], + # write_status always publishes the effective RWX + # StorageClass name, even before the managed class + # materialises on the workload cluster. + "cache": {"storageClassName": "modelplane-rwx-efs"}, + }, + } ), ) -def _launch_template(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "LaunchTemplate", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "name": _LAUNCH_TEMPLATE_NAME, - "instanceType": "p5en.48xlarge", - "blockDeviceMappings": [ - {"deviceName": "/dev/xvda", "ebs": {"volumeSize": 1024}}, - ], - "instanceMarketOptions": {"marketType": "capacity-block"}, - "capacityReservationSpecification": { - "capacityReservationPreference": "capacity-reservations-only", - "capacityReservationTarget": { - "capacityReservationId": _CAPACITY_RESERVATION_ID, +def _vpc(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The VPC.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "VPC", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "cidrBlock": "10.0.0.0/16", + "enableDnsHostnames": True, + "enableDnsSupport": True, }, }, - }, - }, - } - - -def _gpu_node_group_capacity_block(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "NodeGroup", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "managementPolicies": ["Observe", "Create", "Update", "Delete"], - "initProvider": {"scalingConfig": {"desiredSize": 2}}, - "forProvider": { - "region": "us-west-2", - "amiType": "AL2023_x86_64_NVIDIA", - "clusterNameSelector": {"matchControllerRef": True}, - "nodeRoleArnSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "node"}, + } + ), + ) + + +def _subnet(*, name: str, az: str, cidr: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """A public subnet in az.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "Subnet", + "metadata": { + "name": name, + "labels": {"modelplane.ai/zone": az, "modelplane.ai/subnet-tier": "public"}, }, - "capacityType": "CAPACITY_BLOCK", - "launchTemplate": { - "name": _LAUNCH_TEMPLATE_NAME, - "version": "$Latest", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "availabilityZone": az, + "cidrBlock": cidr, + "mapPublicIpOnLaunch": True, + "tags": {"kubernetes.io/role/elb": "1"}, + "vpcIdSelector": {"matchControllerRef": True}, + }, }, - "subnetIdRefs": [{"name": _PRIVATE_SUBNET_A}], - "scalingConfig": {"minSize": 0, "maxSize": 2}, - "labels": { - "modelplane.ai/gpu": "nvidia-h200", - "modelplane.ai/pool": "gpu-h200", + } + ), + ) + + +def _private_subnet(*, name: str, az: str, cidr: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """A private subnet in az.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "Subnet", + "metadata": { + "name": name, + "labels": {"modelplane.ai/zone": az, "modelplane.ai/subnet-tier": "private"}, }, - "taint": [ - { - "key": "nvidia.com/gpu", - "value": "true", - "effect": "NO_SCHEDULE", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "availabilityZone": az, + "cidrBlock": cidr, + "mapPublicIpOnLaunch": False, + "vpcIdSelector": {"matchControllerRef": True}, }, - ], - }, - }, - } + }, + } + ), + ) -# EFA launch template and security-group object names, derived the same way the -# function derives them, so the test can't drift from the child_name hashing. -_EFA_LAUNCH_TEMPLATE_NAME = resource.child_name("test-cluster", "lt-gpu-h200") -_EFA_SECURITY_GROUP_NAME = resource.child_name("test-cluster", "efa-sg") +def _internet_gateway(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The VPC's internet gateway.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "InternetGateway", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "vpcIdSelector": {"matchControllerRef": True}, + }, + }, + } + ), + ) -def _xr_efa() -> v1alpha1.EKSCluster: - return v1alpha1.EKSCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - region="us-west-2", - nodePools=[ - v1alpha1.NodePool( - name="gpu-h200", - role="GPU", - instanceType="p5en.48xlarge", - nodeCount=2, - minNodeCount=0, - maxNodeCount=2, - diskSizeGb=1024, - gpu=v1alpha1.Gpu( - acceleratorType="nvidia-h200", - ), - fabric="EFA", - zones=[v1alpha1.Zone("us-west-2a")], - ), - ], +def _route_table(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The public route table.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "RouteTable", + "metadata": {"labels": {"modelplane.ai/subnet-tier": "public"}}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "vpcIdSelector": {"matchControllerRef": True}, + }, + }, + } ), ) -def _efa_network_interface(card: int, security_groups: list[str] | None = None) -> dict: - ni: dict[str, Any] = { - "networkCardIndex": card, - "deviceIndex": 0 if card == 0 else 1, - "interfaceType": "efa" if card == 0 else "efa-only", - } - if security_groups: - ni["securityGroups"] = security_groups - return ni +def _route_default(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The public route table's default route, through the internet gateway.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "Route", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "destinationCidrBlock": "0.0.0.0/0", + "gatewayIdSelector": {"matchControllerRef": True}, + "routeTableIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/subnet-tier": "public"}, + }, + }, + }, + } + ), + ) -def _launch_template_efa(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - # p5en.48xlarge has 16 network cards; one EFA interface per card. - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "LaunchTemplate", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "name": _EFA_LAUNCH_TEMPLATE_NAME, - "instanceType": "p5en.48xlarge", - "blockDeviceMappings": [ - {"deviceName": "/dev/xvda", "ebs": {"volumeSize": 1024}}, - ], - "networkInterfaces": [_efa_network_interface(card) for card in range(16)], - }, - }, - } - - -def _efa_security_group(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "SecurityGroup", - "metadata": { - "name": _EFA_SECURITY_GROUP_NAME, - "labels": {"modelplane.ai/fabric": "EFA"}, - }, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "name": "test-cluster-efa", - "description": "EFA OS-bypass traffic between gang nodes", - "vpcIdSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _efa_security_group_ingress(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "SecurityGroupIngressRule", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "ipProtocol": "-1", - "referencedSecurityGroupIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/fabric": "EFA"}, - }, - "securityGroupIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/fabric": "EFA"}, - }, - }, - }, - } - - -def _efa_security_group_egress(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "SecurityGroupEgressRule", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "ipProtocol": "-1", - "referencedSecurityGroupIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/fabric": "EFA"}, - }, - "securityGroupIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/fabric": "EFA"}, - }, - }, - }, - } - - -def _gpu_node_group_efa(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "NodeGroup", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "managementPolicies": ["Observe", "Create", "Update", "Delete"], - "initProvider": {"scalingConfig": {"desiredSize": 2}}, - "forProvider": { - "region": "us-west-2", - "amiType": "AL2023_x86_64_NVIDIA", - "clusterNameSelector": {"matchControllerRef": True}, - "nodeRoleArnSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "node"}, - }, - "launchTemplate": { - "name": _EFA_LAUNCH_TEMPLATE_NAME, - "version": "$Latest", - }, - "subnetIdRefs": [{"name": _PRIVATE_SUBNET_A}], - "scalingConfig": {"minSize": 0, "maxSize": 2}, - "labels": { - "modelplane.ai/gpu": "nvidia-h200", - "modelplane.ai/pool": "gpu-h200", - }, - "taint": [ - { - "key": "nvidia.com/gpu", - "value": "true", - "effect": "NO_SCHEDULE", +def _nat_eip(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The NAT gateway's elastic IP.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "EIP", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "domain": "vpc", }, - ], - }, - }, - } - - -def _ready_condition() -> dict: - return { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - } - - -def _vpc(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "VPC", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "cidrBlock": "10.0.0.0/16", - "enableDnsHostnames": True, - "enableDnsSupport": True, - }, - }, - } - - -def _subnet( - name: str, az: str, cidr: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default" -) -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "Subnet", - "metadata": { - "name": name, - "labels": {"modelplane.ai/zone": az, "modelplane.ai/subnet-tier": "public"}, - }, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "availabilityZone": az, - "cidrBlock": cidr, - "mapPublicIpOnLaunch": True, - "tags": {"kubernetes.io/role/elb": "1"}, - "vpcIdSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _private_subnet( - name: str, az: str, cidr: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default" -) -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "Subnet", - "metadata": { - "name": name, - "labels": {"modelplane.ai/zone": az, "modelplane.ai/subnet-tier": "private"}, - }, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "availabilityZone": az, - "cidrBlock": cidr, - "mapPublicIpOnLaunch": False, - "vpcIdSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _internet_gateway(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "InternetGateway", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "vpcIdSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _route_table(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "RouteTable", - "metadata": {"labels": {"modelplane.ai/subnet-tier": "public"}}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "vpcIdSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _route_default(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "Route", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "destinationCidrBlock": "0.0.0.0/0", - "gatewayIdSelector": {"matchControllerRef": True}, - "routeTableIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/subnet-tier": "public"}, }, - }, - }, - } - - -def _nat_eip(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "EIP", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "domain": "vpc", - }, - }, - } - - -def _nat_gateway(az: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "NATGateway", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "allocationIdSelector": {"matchControllerRef": True}, - "subnetIdSelector": { - "matchControllerRef": True, - "matchLabels": { - "modelplane.ai/zone": az, - "modelplane.ai/subnet-tier": "public", + } + ), + ) + + +def _nat_gateway(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The NAT gateway, in the first AZ's public subnet.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "NATGateway", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "allocationIdSelector": {"matchControllerRef": True}, + "subnetIdSelector": { + "matchControllerRef": True, + "matchLabels": { + "modelplane.ai/zone": "us-west-2a", + "modelplane.ai/subnet-tier": "public", + }, + }, }, }, - }, - }, - } - - -def _private_route_table(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "RouteTable", - "metadata": {"labels": {"modelplane.ai/subnet-tier": "private"}}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "vpcIdSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _private_route_default(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "Route", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "destinationCidrBlock": "0.0.0.0/0", - "natGatewayIdSelector": {"matchControllerRef": True}, - "routeTableIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/subnet-tier": "private"}, - }, - }, - }, - } - - -def _route_table_association(az: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "RouteTableAssociation", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "routeTableIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/subnet-tier": "public"}, - }, - "subnetIdSelector": { - "matchControllerRef": True, - "matchLabels": { - "modelplane.ai/zone": az, - "modelplane.ai/subnet-tier": "public", + } + ), + ) + + +def _private_route_table(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The private route table.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "RouteTable", + "metadata": {"labels": {"modelplane.ai/subnet-tier": "private"}}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "vpcIdSelector": {"matchControllerRef": True}, }, }, - }, - }, - } - - -def _private_route_table_association( - az: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default" -) -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "RouteTableAssociation", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "routeTableIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/subnet-tier": "private"}, - }, - "subnetIdSelector": { - "matchControllerRef": True, - "matchLabels": { - "modelplane.ai/zone": az, - "modelplane.ai/subnet-tier": "private", + } + ), + ) + + +def _private_route_default(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The private route table's default route, through the NAT gateway.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "Route", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "destinationCidrBlock": "0.0.0.0/0", + "natGatewayIdSelector": {"matchControllerRef": True}, + "routeTableIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/subnet-tier": "private"}, + }, }, }, - }, - }, - } - - -def _role(role: str, assume_policy: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "iam.aws.m.upbound.io/v1beta1", - "kind": "Role", - "metadata": {"labels": {"modelplane.ai/iam-role": role}}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": {"assumeRolePolicy": assume_policy}, - }, - } - - -def _role_policy_attachment( - role: str, arn: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default" -) -> dict: - return { - "apiVersion": "iam.aws.m.upbound.io/v1beta1", - "kind": "RolePolicyAttachment", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "policyArn": arn, - "roleSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": role}, - }, - }, - }, - } - - -_ASSUME_CLUSTER = ( - '{"Version":"2012-10-17","Statement":[{"Effect":"Allow",' - '"Principal":{"Service":"eks.amazonaws.com"},' - '"Action":"sts:AssumeRole"}]}' -) -_ASSUME_NODE = ( - '{"Version":"2012-10-17","Statement":[{"Effect":"Allow",' - '"Principal":{"Service":"ec2.amazonaws.com"},' - '"Action":"sts:AssumeRole"}]}' -) -_ASSUME_POD_IDENTITY = ( - '{"Version":"2012-10-17","Statement":[{"Effect":"Allow",' - '"Principal":{"Service":"pods.eks.amazonaws.com"},' - '"Action":["sts:AssumeRole","sts:TagSession"]}]}' -) -_POLICY_EFS_CSI = "arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy" -_POLICY_CLUSTER_AUTOSCALER = ( - '{"Version":"2012-10-17","Statement":[' - '{"Effect":"Allow","Action":[' - '"autoscaling:DescribeAutoScalingGroups",' - '"autoscaling:DescribeAutoScalingInstances",' - '"autoscaling:DescribeLaunchConfigurations",' - '"autoscaling:DescribeScalingActivities",' - '"ec2:DescribeImages",' - '"ec2:DescribeInstanceTypes",' - '"ec2:DescribeLaunchTemplateVersions",' - '"ec2:GetInstanceTypesFromInstanceRequirements",' - '"eks:DescribeNodegroup"' - '],"Resource":["*"]},' - '{"Effect":"Allow","Action":[' - '"autoscaling:SetDesiredCapacity",' - '"autoscaling:TerminateInstanceInAutoScalingGroup"' - '],"Resource":["*"]}]}' -) - - -def _eks_cluster(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "Cluster", - "metadata": {"name": "modelplane-system-test-cluster-eks-0865f"}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "version": "1.36", - "roleArnSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "cluster"}, - }, - "accessConfig": { - "authenticationMode": "API_AND_CONFIG_MAP", - "bootstrapClusterCreatorAdminPermissions": True, - }, - "vpcConfig": { - "endpointPrivateAccess": True, - "endpointPublicAccess": True, - "subnetIdSelector": {"matchControllerRef": True}, - }, - }, - }, - } - - -def _cluster_auth(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "ClusterAuth", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "clusterNameSelector": {"matchControllerRef": True}, - }, - "writeConnectionSecretToRef": {"name": _KUBECONFIG_SECRET}, - }, - } - - -def _system_node_group(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "NodeGroup", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "managementPolicies": ["Observe", "Create", "Update", "Delete"], - "initProvider": {"scalingConfig": {"desiredSize": 1}}, - "forProvider": { - "region": "us-west-2", - "amiType": "AL2023_x86_64_STANDARD", - "instanceTypes": ["m6i.xlarge"], - "clusterNameSelector": {"matchControllerRef": True}, - "nodeRoleArnSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "node"}, + } + ), + ) + + +def _route_table_association(*, az: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The association of az's public subnet with the public route table.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "RouteTableAssociation", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "routeTableIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/subnet-tier": "public"}, + }, + "subnetIdSelector": { + "matchControllerRef": True, + "matchLabels": { + "modelplane.ai/zone": az, + "modelplane.ai/subnet-tier": "public", + }, + }, + }, }, - "subnetIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/subnet-tier": "private"}, + } + ), + ) + + +def _private_route_table_association(*, az: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The association of az's private subnet with the private route table.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "RouteTableAssociation", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "routeTableIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/subnet-tier": "private"}, + }, + "subnetIdSelector": { + "matchControllerRef": True, + "matchLabels": { + "modelplane.ai/zone": az, + "modelplane.ai/subnet-tier": "private", + }, + }, + }, }, - "scalingConfig": {"minSize": 1, "maxSize": 2}, - "labels": {"modelplane.ai/pool": "system"}, - }, - }, - } - - -def _gpu_node_group(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "NodeGroup", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "managementPolicies": ["Observe", "Create", "Update", "Delete"], - "initProvider": {"scalingConfig": {"desiredSize": 1}}, - "forProvider": { - "region": "us-west-2", - "amiType": "AL2023_x86_64_NVIDIA", - "instanceTypes": ["g6.xlarge"], - "diskSize": 100, - "clusterNameSelector": {"matchControllerRef": True}, - "nodeRoleArnSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "node"}, + } + ), + ) + + +def _cluster_role(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The IAM role the EKS control plane assumes.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "iam.aws.m.upbound.io/v1beta1", + "kind": "Role", + "metadata": {"labels": {"modelplane.ai/iam-role": "cluster"}}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "assumeRolePolicy": ( + '{"Version":"2012-10-17","Statement":[{"Effect":"Allow",' + '"Principal":{"Service":"eks.amazonaws.com"},' + '"Action":"sts:AssumeRole"}]}' + ), + }, }, - "subnetIdRefs": [{"name": _PRIVATE_SUBNET_A}, {"name": _PRIVATE_SUBNET_B}], - "scalingConfig": {"minSize": 0, "maxSize": 4}, - "labels": { - "modelplane.ai/gpu": "nvidia-l4", - "modelplane.ai/pool": "gpu-l4", + } + ), + ) + + +def _node_role(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The IAM role the nodes assume.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "iam.aws.m.upbound.io/v1beta1", + "kind": "Role", + "metadata": {"labels": {"modelplane.ai/iam-role": "node"}}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "assumeRolePolicy": ( + '{"Version":"2012-10-17","Statement":[{"Effect":"Allow",' + '"Principal":{"Service":"ec2.amazonaws.com"},' + '"Action":"sts:AssumeRole"}]}' + ), + }, }, - "taint": [ - { - "key": "nvidia.com/gpu", - "value": "true", - "effect": "NO_SCHEDULE", + } + ), + ) + + +def _pod_identity_role(*, role: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """An IAM role a ServiceAccount assumes through EKS Pod Identity.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "iam.aws.m.upbound.io/v1beta1", + "kind": "Role", + "metadata": {"labels": {"modelplane.ai/iam-role": role}}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "assumeRolePolicy": ( + '{"Version":"2012-10-17","Statement":[{"Effect":"Allow",' + '"Principal":{"Service":"pods.eks.amazonaws.com"},' + '"Action":["sts:AssumeRole","sts:TagSession"]}]}' + ), }, - ], - }, - }, - } - - -def _addon(name: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "Addon", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "addonName": name, - "clusterNameSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _efs_filesystem(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "efs.aws.m.upbound.io/v1beta1", - "kind": "FileSystem", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": {"region": "us-west-2", "throughputMode": "elastic", "encrypted": True}, - }, - } - - -def _efs_security_group(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "SecurityGroup", - "metadata": {"labels": {"modelplane.ai/sg-role": "efs"}}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "name": "test-cluster-efs", - "description": "NFS access to the ModelCache EFS mount targets", - "vpcIdSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _efs_security_group_ingress(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "SecurityGroupIngressRule", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "ipProtocol": "tcp", - "fromPort": 2049, - "toPort": 2049, - "cidrIpv4": "10.0.0.0/16", - "securityGroupIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/sg-role": "efs"}, }, - }, - }, - } - - -def _efs_mount_target(subnet_name: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "efs.aws.m.upbound.io/v1beta1", - "kind": "MountTarget", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "fileSystemIdSelector": {"matchControllerRef": True}, - "subnetIdRef": {"name": subnet_name}, - "securityGroupsSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/sg-role": "efs"}, + } + ), + ) + + +def _role_policy_attachment(*, role: str, arn: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """An attachment of the managed policy arn to the role labelled role.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "iam.aws.m.upbound.io/v1beta1", + "kind": "RolePolicyAttachment", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "policyArn": arn, + "roleSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": role}, + }, + }, }, - }, - }, - } - - -def _pod_identity_association(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "PodIdentityAssociation", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "namespace": "kube-system", - "serviceAccount": "efs-csi-controller-sa", - "clusterNameSelector": {"matchControllerRef": True}, - "roleArnSelector": {"matchControllerRef": True, "matchLabels": {"modelplane.ai/iam-role": "efs-csi"}}, - }, - }, - } - - -def _storage_class_object(filesystem_id: str) -> dict: - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { - "kind": "ProviderConfig", - "name": _KUBECONFIG_SECRET, - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "storage.k8s.io/v1", - "kind": "StorageClass", - "metadata": {"name": "modelplane-rwx-efs"}, - "provisioner": "efs.csi.aws.com", - "parameters": { - "provisioningMode": "efs-ap", - "fileSystemId": filesystem_id, - "directoryPerms": "700", + } + ), + ) + + +def _eks_cluster(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed EKS Cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "Cluster", + "metadata": {"name": "modelplane-system-test-cluster-eks-0865f"}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "version": "1.36", + "roleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "cluster"}, + }, + "accessConfig": { + "authenticationMode": "API_AND_CONFIG_MAP", + "bootstrapClusterCreatorAdminPermissions": True, + }, + "vpcConfig": { + "endpointPrivateAccess": True, + "endpointPublicAccess": True, + "subnetIdSelector": {"matchControllerRef": True}, + }, }, - "volumeBindingMode": "Immediate", }, + } + ), + ready=ready, + ) + + +def _observed_eks_cluster(*, cluster_security_group_id: str | None, ready_condition: bool) -> fnv1.Resource: + """The observed EKS Cluster, with its security group id if given, and a True Ready condition only if ready_condition.""" + status: dict = {} + if cluster_security_group_id is not None: + status["atProvider"] = {"vpcConfig": {"clusterSecurityGroupId": cluster_security_group_id}} + if ready_condition: + status["conditions"] = [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", }, - }, - } - - -def _autoscaler_policy(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "iam.aws.m.upbound.io/v1beta1", - "kind": "Policy", - "metadata": {"labels": {"modelplane.ai/iam-role": "cluster-autoscaler"}}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": {"policy": _POLICY_CLUSTER_AUTOSCALER}, - }, - } - - -def _autoscaler_attachment(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "iam.aws.m.upbound.io/v1beta1", - "kind": "RolePolicyAttachment", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "policyArnSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "cluster-autoscaler"}, - }, - "roleSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "cluster-autoscaler"}, + ] + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "Cluster", + "metadata": {"name": "modelplane-system-test-cluster-eks-0865f"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "version": "1.36", + "roleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "cluster"}, + }, + "accessConfig": { + "authenticationMode": "API_AND_CONFIG_MAP", + "bootstrapClusterCreatorAdminPermissions": True, + }, + "vpcConfig": { + "endpointPrivateAccess": True, + "endpointPublicAccess": True, + "subnetIdSelector": {"matchControllerRef": True}, + }, + }, }, - }, - }, - } - - -def _autoscaler_pod_identity(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "PodIdentityAssociation", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "namespace": "kube-system", - "serviceAccount": "cluster-autoscaler", - "clusterNameSelector": {"matchControllerRef": True}, - "roleArnSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "cluster-autoscaler"}, + "status": status, + } + ), + ) + + +def _cluster_auth(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed ClusterAuth.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "ClusterAuth", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "clusterNameSelector": {"matchControllerRef": True}, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, }, - }, - }, - } - - -def _autoscaler_release() -> dict: - return { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "Release", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": {"kind": "ProviderConfig", "name": _KUBECONFIG_SECRET}, - "forProvider": { - "chart": { - "name": "cluster-autoscaler", - "repository": "https://kubernetes.github.io/autoscaler", - "version": "9.57.0", + } + ), + ready=ready, + ) + + +def _nodegroup_system(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The composed system node group every cluster gets.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "managementPolicies": ["Observe", "Create", "Update", "Delete"], + "initProvider": {"scalingConfig": {"desiredSize": 1}}, + "forProvider": { + "region": "us-west-2", + "amiType": "AL2023_x86_64_STANDARD", + "instanceTypes": ["m6i.xlarge"], + "clusterNameSelector": {"matchControllerRef": True}, + "nodeRoleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "node"}, + }, + "subnetIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/subnet-tier": "private"}, + }, + "scalingConfig": {"minSize": 1, "maxSize": 2}, + "labels": {"modelplane.ai/pool": "system"}, + }, }, - "namespace": "kube-system", - "values": { - "cloudProvider": "aws", - "awsRegion": "us-west-2", - "autoDiscovery": {"clusterName": "modelplane-system-test-cluster-eks-0865f"}, - "rbac": {"serviceAccount": {"name": "cluster-autoscaler"}}, - "extraArgs": {"balance-similar-node-groups": True}, + } + ), + ) + + +def _nodegroup_gpu(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The composed gpu-l4 node group, which needs no launch template.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "managementPolicies": ["Observe", "Create", "Update", "Delete"], + "initProvider": {"scalingConfig": {"desiredSize": 1}}, + "forProvider": { + "region": "us-west-2", + "amiType": "AL2023_x86_64_NVIDIA", + "instanceTypes": ["g6.xlarge"], + "diskSize": 100, + "clusterNameSelector": {"matchControllerRef": True}, + "nodeRoleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "node"}, + }, + "subnetIdRefs": [ + {"name": "test-cluster-private-subnet-us-west-2a-6a89f"}, + {"name": "test-cluster-private-subnet-us-west-2b-b7832"}, + ], + "scalingConfig": {"minSize": 0, "maxSize": 4}, + "labels": { + "modelplane.ai/gpu": "nvidia-l4", + "modelplane.ai/pool": "gpu-l4", + }, + "taint": [ + { + "key": "nvidia.com/gpu", + "value": "true", + "effect": "NO_SCHEDULE", + }, + ], + }, }, - }, - }, - } - - -def _efa_dra_driver_release() -> dict: - return { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "Release", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": {"kind": "ProviderConfig", "name": _KUBECONFIG_SECRET}, - "forProvider": { - "chart": { - "name": "aws-dranet", - "repository": "https://aws.github.io/eks-charts", - "version": "1.0.0", + } + ), + ) + + +def _launch_template_efa(*, security_groups: list[str] | None) -> fnv1.Resource: + """The gpu-h200 EFA launch template, with security_groups on every interface if given.""" + # p5en.48xlarge has 16 network cards; one EFA interface per card. + interfaces: list[dict] = [ + {"networkCardIndex": 0, "deviceIndex": 0, "interfaceType": "efa"}, + {"networkCardIndex": 1, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 2, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 3, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 4, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 5, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 6, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 7, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 8, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 9, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 10, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 11, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 12, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 13, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 14, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 15, "deviceIndex": 1, "interfaceType": "efa-only"}, + ] + if security_groups is not None: + for interface in interfaces: + interface["securityGroups"] = security_groups + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "LaunchTemplate", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "name": "test-cluster-lt-gpu-h200-83c00", + "instanceType": "p5en.48xlarge", + "blockDeviceMappings": [ + {"deviceName": "/dev/xvda", "ebs": {"volumeSize": 1024}}, + ], + "networkInterfaces": interfaces, + }, }, - "namespace": "kube-system", - "values": { - "tolerations": [ - {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}, - ], + } + ), + ) + + +def _efa_security_group() -> fnv1.Resource: + """The composed EFA security group.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "SecurityGroup", + "metadata": { + "name": "test-cluster-efa-sg-602a9", + "labels": {"modelplane.ai/fabric": "EFA"}, }, - }, - }, - } - - -def _provider_config(api_version: str) -> dict: - return { - "apiVersion": api_version, - "kind": "ProviderConfig", - "metadata": {"name": _KUBECONFIG_SECRET}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "name": _KUBECONFIG_SECRET, - "namespace": "modelplane-system", - "key": "kubeconfig", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "name": "test-cluster-efa", + "description": "EFA OS-bypass traffic between gang nodes", + "vpcIdSelector": {"matchControllerRef": True}, + }, }, - }, - }, - } + } + ), + ) -def _expected_status() -> dict: - # `type` is emitted because the function sets it explicitly on the Status - # model, so update_status (exclude_unset) keeps it rather than dropping it - # as an unset field. - status = { - "secrets": [ +def _efa_security_group_ingress() -> fnv1.Resource: + """The EFA security group's rule admitting all traffic from itself.""" + return fnv1.Resource( + resource=resource.dict_to_struct( { - "type": "Kubeconfig", - "name": _KUBECONFIG_SECRET, - "key": "kubeconfig", - }, - ], - # write_status always publishes the effective RWX StorageClass name, - # even before the managed class materialises on the workload cluster. - "cache": {"storageClassName": "modelplane-rwx-efs"}, - } - return {"status": status} - - -def _expected_resources() -> dict: - return { - "vpc": fnv1.Resource(resource=resource.dict_to_struct(_vpc())), - "subnet-0": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_A, "us-west-2a", "10.0.0.0/20")), + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "SecurityGroupIngressRule", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "ipProtocol": "-1", + "referencedSecurityGroupIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/fabric": "EFA"}, + }, + "securityGroupIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/fabric": "EFA"}, + }, + }, + }, + } ), - "subnet-1": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_B, "us-west-2b", "10.0.16.0/20")), + ) + + +def _efa_security_group_egress() -> fnv1.Resource: + """The EFA security group's rule allowing all traffic to itself.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "SecurityGroupEgressRule", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "ipProtocol": "-1", + "referencedSecurityGroupIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/fabric": "EFA"}, + }, + "securityGroupIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/fabric": "EFA"}, + }, + }, + }, + } ), - "subnet-2": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_C, "us-west-2c", "10.0.32.0/20")), + ) + + +def _nodegroup_gpu_efa() -> fnv1.Resource: + """The composed gpu-h200 EFA node group, which takes its instance type from the launch template.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "managementPolicies": ["Observe", "Create", "Update", "Delete"], + "initProvider": {"scalingConfig": {"desiredSize": 2}}, + "forProvider": { + "region": "us-west-2", + "amiType": "AL2023_x86_64_NVIDIA", + "clusterNameSelector": {"matchControllerRef": True}, + "nodeRoleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "node"}, + }, + "launchTemplate": { + "name": "test-cluster-lt-gpu-h200-83c00", + "version": "$Latest", + }, + "subnetIdRefs": [{"name": "test-cluster-private-subnet-us-west-2a-6a89f"}], + "scalingConfig": {"minSize": 0, "maxSize": 2}, + "labels": { + "modelplane.ai/gpu": "nvidia-h200", + "modelplane.ai/pool": "gpu-h200", + }, + "taint": [ + { + "key": "nvidia.com/gpu", + "value": "true", + "effect": "NO_SCHEDULE", + }, + ], + }, + }, + } ), - "private-subnet-0": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_A, "us-west-2a", "10.0.48.0/20")), + ) + + +def _addon(*, name: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The EKS addon called name.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "Addon", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "addonName": name, + "clusterNameSelector": {"matchControllerRef": True}, + }, + }, + } ), - "private-subnet-1": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_B, "us-west-2b", "10.0.64.0/20")), + ) + + +def _efs_filesystem(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed EFS filesystem backing ModelCache.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "efs.aws.m.upbound.io/v1beta1", + "kind": "FileSystem", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": {"region": "us-west-2", "throughputMode": "elastic", "encrypted": True}, + }, + } ), - "private-subnet-2": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_C, "us-west-2c", "10.0.80.0/20")), - ), - "internet-gateway": fnv1.Resource(resource=resource.dict_to_struct(_internet_gateway())), - "nat-eip": fnv1.Resource(resource=resource.dict_to_struct(_nat_eip())), - "nat-gateway": fnv1.Resource(resource=resource.dict_to_struct(_nat_gateway("us-west-2a"))), - "route-table": fnv1.Resource(resource=resource.dict_to_struct(_route_table())), - "route-default": fnv1.Resource(resource=resource.dict_to_struct(_route_default())), - "private-route-table": fnv1.Resource(resource=resource.dict_to_struct(_private_route_table())), - "private-route-default": fnv1.Resource(resource=resource.dict_to_struct(_private_route_default())), - "route-table-association-0": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2a")), - ), - "route-table-association-1": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2b")), - ), - "route-table-association-2": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2c")), - ), - "private-route-table-association-0": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2a")), - ), - "private-route-table-association-1": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2b")), - ), - "private-route-table-association-2": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2c")), - ), - "iam-role-cluster": fnv1.Resource( - resource=resource.dict_to_struct(_role("cluster", _ASSUME_CLUSTER)), - ), - "iam-attach-cluster-policy": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("cluster", "arn:aws:iam::aws:policy/AmazonEKSClusterPolicy"), - ), - ), - "iam-role-node": fnv1.Resource(resource=resource.dict_to_struct(_role("node", _ASSUME_NODE))), - "iam-attach-node-worker": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy"), - ), - ), - "iam-attach-node-cni": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy"), - ), - ), - "iam-attach-node-ecr": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly"), - ), - ), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_eks_cluster())), - "cluster-auth": fnv1.Resource(resource=resource.dict_to_struct(_cluster_auth())), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_system_node_group())), - "nodegroup-gpu-l4": fnv1.Resource(resource=resource.dict_to_struct(_gpu_node_group())), - "addon-vpc-cni": fnv1.Resource(resource=resource.dict_to_struct(_addon("vpc-cni"))), - "addon-kube-proxy": fnv1.Resource(resource=resource.dict_to_struct(_addon("kube-proxy"))), - "addon-coredns": fnv1.Resource(resource=resource.dict_to_struct(_addon("coredns"))), - "efs-filesystem": fnv1.Resource(resource=resource.dict_to_struct(_efs_filesystem())), - "efs-security-group": fnv1.Resource(resource=resource.dict_to_struct(_efs_security_group())), - "efs-security-group-ingress": fnv1.Resource(resource=resource.dict_to_struct(_efs_security_group_ingress())), - "efs-mount-target-0": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_A))), - "efs-mount-target-1": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_B))), - "efs-mount-target-2": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_C))), - "iam-role-efs-csi": fnv1.Resource(resource=resource.dict_to_struct(_role("efs-csi", _ASSUME_POD_IDENTITY))), - "iam-attach-efs-csi": fnv1.Resource( - resource=resource.dict_to_struct(_role_policy_attachment("efs-csi", _POLICY_EFS_CSI)), - ), - "addon-eks-pod-identity-agent": fnv1.Resource( - resource=resource.dict_to_struct(_addon("eks-pod-identity-agent")), - ), - "pod-identity-efs-csi": fnv1.Resource(resource=resource.dict_to_struct(_pod_identity_association())), - "addon-aws-efs-csi-driver": fnv1.Resource(resource=resource.dict_to_struct(_addon("aws-efs-csi-driver"))), - "iam-policy-cluster-autoscaler": fnv1.Resource(resource=resource.dict_to_struct(_autoscaler_policy())), - "iam-role-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_role("cluster-autoscaler", _ASSUME_POD_IDENTITY)), - ), - "iam-attach-cluster-autoscaler": fnv1.Resource(resource=resource.dict_to_struct(_autoscaler_attachment())), - "pod-identity-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler_pod_identity()), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config("kubernetes.m.crossplane.io/v1alpha1")), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config("helm.m.crossplane.io/v1beta1")), - ready=fnv1.READY_TRUE, - ), - } + ready=ready, + ) -def _compose_cases() -> list[Case]: - """The cases test_compose runs.""" - # Second pass: cluster and cluster-auth observed Ready, function flips - # those two desired resources ready while still emitting everything. - ready_resources = _expected_resources() - ready_resources["cluster"] = fnv1.Resource( - resource=ready_resources["cluster"].resource, - ready=fnv1.READY_TRUE, - ) - ready_resources["cluster-auth"] = fnv1.Resource( - resource=ready_resources["cluster-auth"].resource, - ready=fnv1.READY_TRUE, - ) - ready_resources["efs-filesystem"] = fnv1.Resource( - resource=ready_resources["efs-filesystem"].resource, - ready=fnv1.READY_TRUE, +def _efs_security_group(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The security group on the EFS mount targets.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "SecurityGroup", + "metadata": {"labels": {"modelplane.ai/sg-role": "efs"}}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "name": "test-cluster-efs", + "description": "NFS access to the ModelCache EFS mount targets", + "vpcIdSelector": {"matchControllerRef": True}, + }, + }, + } + ), ) - # Once the EFS filesystem id is observed, the managed StorageClass Object - # is composed (and marked ready) against the cluster's own ProviderConfig. - ready_resources["storage-class-rwx-efs"] = fnv1.Resource( - resource=resource.dict_to_struct(_storage_class_object("fs-0abc123")), - ready=fnv1.READY_TRUE, + + +def _efs_security_group_ingress(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The EFS security group's rule admitting NFS from the VPC.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "SecurityGroupIngressRule", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "ipProtocol": "tcp", + "fromPort": 2049, + "toPort": 2049, + "cidrIpv4": "10.0.0.0/16", + "securityGroupIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/sg-role": "efs"}, + }, + }, + }, + } + ), ) - # With the cluster observed, the autoscaler Helm release is composed (it's - # gated on the cluster existing so provider-helm can reach it). It carries - # no Ready condition yet, so it stays not-ready this pass. - ready_resources["release-cluster-autoscaler"] = fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler_release()), + + +def _efs_mount_target(*, subnet_name: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """An EFS mount target in the named subnet.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "efs.aws.m.upbound.io/v1beta1", + "kind": "MountTarget", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "fileSystemIdSelector": {"matchControllerRef": True}, + "subnetIdRef": {"name": subnet_name}, + "securityGroupsSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/sg-role": "efs"}, + }, + }, + }, + } + ), ) - return [ - Case( - name="first pass composes infra resources; none ready", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr().model_dump(exclude_none=True, mode="json"), - ), - ), - ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources=_expected_resources(), - ), - context=structpb.Struct(), - ), + +def _pod_identity_association(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The Pod Identity association for the EFS CSI controller.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "PodIdentityAssociation", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "namespace": "kube-system", + "serviceAccount": "efs-csi-controller-sa", + "clusterNameSelector": {"matchControllerRef": True}, + "roleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "efs-csi"}, + }, + }, + }, + } ), - Case( - name="second pass with observed cluster ready marks cluster resources ready", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr().model_dump(exclude_none=True, mode="json"), - ), - ), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_eks_cluster(), - "status": {"conditions": [_ready_condition()]}, - }, - ), - ), - "cluster-auth": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_cluster_auth(), - "status": {"conditions": [_ready_condition()]}, - }, - ), - ), - "efs-filesystem": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_efs_filesystem(), - "metadata": {"annotations": {"crossplane.io/external-name": "fs-0abc123"}}, - "status": {"conditions": [_ready_condition()]}, - }, - ), + ) + + +def _autoscaler_policy(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The IAM policy the cluster autoscaler needs.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "iam.aws.m.upbound.io/v1beta1", + "kind": "Policy", + "metadata": {"labels": {"modelplane.ai/iam-role": "cluster-autoscaler"}}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "policy": ( + '{"Version":"2012-10-17","Statement":[' + '{"Effect":"Allow","Action":[' + '"autoscaling:DescribeAutoScalingGroups",' + '"autoscaling:DescribeAutoScalingInstances",' + '"autoscaling:DescribeLaunchConfigurations",' + '"autoscaling:DescribeScalingActivities",' + '"ec2:DescribeImages",' + '"ec2:DescribeInstanceTypes",' + '"ec2:DescribeLaunchTemplateVersions",' + '"ec2:GetInstanceTypesFromInstanceRequirements",' + '"eks:DescribeNodegroup"' + '],"Resource":["*"]},' + '{"Effect":"Allow","Action":[' + '"autoscaling:SetDesiredCapacity",' + '"autoscaling:TerminateInstanceInAutoScalingGroup"' + '],"Resource":["*"]}]}' ), }, - ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources=ready_resources, - ), - context=structpb.Struct(), - ), + }, + } ), - ] + ) -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) +def _autoscaler_attachment(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The attachment of the autoscaler's policy to its role.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "iam.aws.m.upbound.io/v1beta1", + "kind": "RolePolicyAttachment", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "policyArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "cluster-autoscaler"}, + }, + "roleSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "cluster-autoscaler"}, + }, + }, + }, + } + ), + ) -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) -def test_compose(case: Case) -> None: - """The function composes EKS cluster infrastructure.""" - got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) - - -def test_compose_capacity_block() -> None: - """A Capacity Block pool composes a launch template and a CAPACITY_BLOCK node group.""" - # The GPU node group must not set instanceTypes (EKS takes the type - # from the launch template), must set capacityType=CAPACITY_BLOCK, and - # must reference the launch template. The launch template targets the - # reservation via the capacity-block market type. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr_capacity_block().model_dump(exclude_none=True, mode="json"), - ), - ), +def _autoscaler_pod_identity(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The Pod Identity association for the cluster autoscaler.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "PodIdentityAssociation", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "namespace": "kube-system", + "serviceAccount": "cluster-autoscaler", + "clusterNameSelector": {"matchControllerRef": True}, + "roleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "cluster-autoscaler"}, + }, + }, + }, + } ), ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - resources = got.desired.resources - - # The launch template is composed and targets the reservation. - assert "launch-template-gpu-h200" in resources - assert resource.struct_to_dict(resources["launch-template-gpu-h200"].resource) == _launch_template() - - # The GPU node group uses CAPACITY_BLOCK + the launch template and - # carries no instanceTypes. - assert resource.struct_to_dict(resources["nodegroup-gpu-h200"].resource) == _gpu_node_group_capacity_block() - - -def test_compose_efa() -> None: - """An EFA GPU pool composes EFA infrastructure end to end.""" - # The node group's launch template carries one EFA interface per network - # card (card 0 keeps device index 0 for the node's IP traffic, the rest - # device index 1 for RDMA), the cluster gets an EFA security group with - # self-referencing all-traffic ingress and egress rules, and the node - # group references the launch template instead of setting instanceTypes. - want_resources = { - "vpc": fnv1.Resource(resource=resource.dict_to_struct(_vpc())), - "subnet-0": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_A, "us-west-2a", "10.0.0.0/20")), - ), - "subnet-1": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_B, "us-west-2b", "10.0.16.0/20")), - ), - "subnet-2": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_C, "us-west-2c", "10.0.32.0/20")), - ), - "private-subnet-0": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_A, "us-west-2a", "10.0.48.0/20")), - ), - "private-subnet-1": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_B, "us-west-2b", "10.0.64.0/20")), - ), - "private-subnet-2": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_C, "us-west-2c", "10.0.80.0/20")), - ), - "internet-gateway": fnv1.Resource(resource=resource.dict_to_struct(_internet_gateway())), - "nat-eip": fnv1.Resource(resource=resource.dict_to_struct(_nat_eip())), - "nat-gateway": fnv1.Resource(resource=resource.dict_to_struct(_nat_gateway("us-west-2a"))), - "route-table": fnv1.Resource(resource=resource.dict_to_struct(_route_table())), - "route-default": fnv1.Resource(resource=resource.dict_to_struct(_route_default())), - "private-route-table": fnv1.Resource(resource=resource.dict_to_struct(_private_route_table())), - "private-route-default": fnv1.Resource(resource=resource.dict_to_struct(_private_route_default())), - "route-table-association-0": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2a")), - ), - "route-table-association-1": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2b")), - ), - "route-table-association-2": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2c")), - ), - "private-route-table-association-0": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2a")), - ), - "private-route-table-association-1": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2b")), - ), - "private-route-table-association-2": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2c")), - ), - "iam-role-cluster": fnv1.Resource( - resource=resource.dict_to_struct(_role("cluster", _ASSUME_CLUSTER)), - ), - "iam-attach-cluster-policy": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("cluster", "arn:aws:iam::aws:policy/AmazonEKSClusterPolicy"), - ), - ), - "iam-role-node": fnv1.Resource(resource=resource.dict_to_struct(_role("node", _ASSUME_NODE))), - "iam-attach-node-worker": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy"), - ), - ), - "iam-attach-node-cni": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy"), - ), - ), - "iam-attach-node-ecr": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly"), - ), - ), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_eks_cluster())), - "cluster-auth": fnv1.Resource(resource=resource.dict_to_struct(_cluster_auth())), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_system_node_group())), - "launch-template-gpu-h200": fnv1.Resource(resource=resource.dict_to_struct(_launch_template_efa())), - "efa-security-group": fnv1.Resource(resource=resource.dict_to_struct(_efa_security_group())), - "efa-security-group-ingress": fnv1.Resource( - resource=resource.dict_to_struct(_efa_security_group_ingress()), - ), - "efa-security-group-egress": fnv1.Resource( - resource=resource.dict_to_struct(_efa_security_group_egress()), - ), - "nodegroup-gpu-h200": fnv1.Resource(resource=resource.dict_to_struct(_gpu_node_group_efa())), - "addon-vpc-cni": fnv1.Resource(resource=resource.dict_to_struct(_addon("vpc-cni"))), - "addon-kube-proxy": fnv1.Resource(resource=resource.dict_to_struct(_addon("kube-proxy"))), - "addon-coredns": fnv1.Resource(resource=resource.dict_to_struct(_addon("coredns"))), - "efs-filesystem": fnv1.Resource(resource=resource.dict_to_struct(_efs_filesystem())), - "efs-security-group": fnv1.Resource(resource=resource.dict_to_struct(_efs_security_group())), - "efs-security-group-ingress": fnv1.Resource( - resource=resource.dict_to_struct(_efs_security_group_ingress()), - ), - "efs-mount-target-0": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_A))), - "efs-mount-target-1": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_B))), - "efs-mount-target-2": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_C))), - "iam-role-efs-csi": fnv1.Resource( - resource=resource.dict_to_struct(_role("efs-csi", _ASSUME_POD_IDENTITY)), - ), - "iam-attach-efs-csi": fnv1.Resource( - resource=resource.dict_to_struct(_role_policy_attachment("efs-csi", _POLICY_EFS_CSI)), - ), - "addon-eks-pod-identity-agent": fnv1.Resource( - resource=resource.dict_to_struct(_addon("eks-pod-identity-agent")), - ), - "pod-identity-efs-csi": fnv1.Resource(resource=resource.dict_to_struct(_pod_identity_association())), - "addon-aws-efs-csi-driver": fnv1.Resource( - resource=resource.dict_to_struct(_addon("aws-efs-csi-driver")), - ), - "iam-policy-cluster-autoscaler": fnv1.Resource(resource=resource.dict_to_struct(_autoscaler_policy())), - "iam-role-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_role("cluster-autoscaler", _ASSUME_POD_IDENTITY)), - ), - "iam-attach-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler_attachment()), - ), - "pod-identity-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler_pod_identity()), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config("kubernetes.m.crossplane.io/v1alpha1")), - ready=fnv1.READY_TRUE, + +def _autoscaler_release() -> fnv1.Resource: + """The cluster autoscaler's Helm release.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster-kubeconfig-55b57"}, + "forProvider": { + "chart": { + "name": "cluster-autoscaler", + "repository": "https://kubernetes.github.io/autoscaler", + "version": "9.57.0", + }, + "namespace": "kube-system", + "values": { + "cloudProvider": "aws", + "awsRegion": "us-west-2", + "autoDiscovery": {"clusterName": "modelplane-system-test-cluster-eks-0865f"}, + "rbac": {"serviceAccount": {"name": "cluster-autoscaler"}}, + "extraArgs": {"balance-similar-node-groups": True}, + }, + }, + }, + } ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config("helm.m.crossplane.io/v1beta1")), - ready=fnv1.READY_TRUE, + ) + + +def _provider_config(*, api_version: str) -> fnv1.Resource: + """A Ready ProviderConfig that reaches the cluster through its kubeconfig Secret.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": api_version, + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster-kubeconfig-55b57"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", + "key": "kubeconfig", + }, + }, + }, + } ), - } + ready=fnv1.READY_TRUE, + ) + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - case = Case( - name="an EFA pool composes EFA launch template, security group, and rules", + +COMPOSE_CASES = [ + Case( + name="FirstPass", + reason="With nothing observed, an EKSCluster composes its AWS infrastructure and ProviderConfigs, and marks only the ProviderConfigs ready.", req=fnv1.RunFunctionRequest( observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr_efa().model_dump(exclude_none=True, mode="json"), - ), + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-l4", + role="GPU", + instanceType="g6.xlarge", + nodeCount=1, + minNodeCount=0, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l4"), + zones=[v1alpha1.Zone("us-west-2a"), v1alpha1.Zone("us-west-2b")], + ), + ], ), ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources=want_resources, + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ClusterProviderConfig", cred_name="default"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "internet-gateway": _internet_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-eip": _nat_eip(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-gateway": _nat_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-table": _route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-default": _route_default(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-table": _private_route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-default": _private_route_default( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster": _cluster_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-node": _node_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "cluster": _eks_cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster-auth": _cluster_auth( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-system": _nodegroup_system(cred_kind="ClusterProviderConfig", cred_name="default"), + "nodegroup-gpu-l4": _nodegroup_gpu(cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-filesystem": _efs_filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "efs-security-group": _efs_security_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, ), context=structpb.Struct(), ), - ) - - got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) - - -def test_compose_efa_cluster_security_group() -> None: - """Once both security groups are observed, every interface carries them.""" - # A launch template with networkInterfaces makes its security groups - # authoritative, so the interfaces must carry both the EFA security group - # and the EKS cluster security group or the node never joins. Both are set - # as raw IDs in securityGroups (not securityGroupRefs): the provider's - # reference resolver no-ops once that field is populated, so a ref mixed - # with a literal would be dropped. The EFA group's ID comes from its - # observed external name, the cluster group's from the observed cluster's - # status, so both appear only once their resources report them. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr_efa().model_dump(exclude_none=True, mode="json"), + ), + # The observed filesystem id lets the function compose the StorageClass + # Object, which is ready, against the cluster's own ProviderConfig. The + # observed cluster lets it compose the autoscaler Helm release, since + # provider-helm can now reach the cluster. The release isn't observed yet, so + # it isn't ready. + Case( + name="ResourcesReady", + reason="With the cluster, ClusterAuth and EFS filesystem observed Ready, an EKSCluster marks them ready and composes the StorageClass and autoscaler release.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-l4", + role="GPU", + instanceType="g6.xlarge", + nodeCount=1, + minNodeCount=0, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l4"), + zones=[v1alpha1.Zone("us-west-2a"), v1alpha1.Zone("us-west-2b")], + ), + ], ), - ), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_eks_cluster(), - "status": { - "atProvider": { - "vpcConfig": {"clusterSecurityGroupId": "sg-0cluster"}, + resources={ + "cluster": _observed_eks_cluster(cluster_security_group_id=None, ready_condition=True), + "cluster-auth": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "ClusterAuth", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "clusterNameSelector": {"matchControllerRef": True}, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, }, - }, - }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), ), - ), - "efa-security-group": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_efa_security_group(), - "metadata": { - **_efa_security_group()["metadata"], - "annotations": {"crossplane.io/external-name": "sg-0efa"}, - }, - }, + "efs-filesystem": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "efs.aws.m.upbound.io/v1beta1", + "kind": "FileSystem", + "metadata": {"annotations": {"crossplane.io/external-name": "fs-0abc123"}}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "throughputMode": "elastic", + "encrypted": True, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), ), - ), - }, - ), - ) - - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - lt = resource.struct_to_dict(got.desired.resources["launch-template-gpu-h200"].resource) - interfaces = lt["spec"]["forProvider"]["networkInterfaces"] - - # Every interface carries both SGs as raw IDs (EFA first, then cluster) - # and no securityGroupRefs; no interface requests a public IP (nodes are - # in private subnets). - assert interfaces[0]["interfaceType"] == "efa" - for ni in interfaces: - assert "securityGroupRefs" not in ni - assert ni["securityGroups"] == ["sg-0efa", "sg-0cluster"] - assert "associatePublicIpAddress" not in ni - for ni in interfaces[1:]: - assert ni["interfaceType"] == "efa-only" - - -def test_compose_efa_dra_driver() -> None: - """An EFA pool installs the EFA DRA driver Helm release, and a pool without EFA doesn't.""" - # Like the autoscaler, the release is gated on the cluster being observed - # so provider-helm can reach it. A pool without the EFA fabric installs no - # driver even once the cluster is observed. - observed_cluster = { - "cluster": fnv1.Resource( - resource=resource.dict_to_struct( - {**_eks_cluster(), "status": {"conditions": [_ready_condition()]}}, + }, ), ), - } - - got_efa = asyncio.run( - fn.FunctionRunner().RunFunction( - fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_xr_efa().model_dump(exclude_none=True, mode="json")), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ClusterProviderConfig", cred_name="default"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", ), - resources=observed_cluster, - ), - ), - None, - ) - ) - assert "release-efa-dra-driver" in got_efa.desired.resources - assert ( - resource.struct_to_dict(got_efa.desired.resources["release-efa-dra-driver"].resource) - == _efa_dra_driver_release() - ) - - got_none = asyncio.run( - fn.FunctionRunner().RunFunction( - fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_xr().model_dump(exclude_none=True, mode="json")), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", ), - resources=observed_cluster, - ), - ), - None, - ) - ) - assert "release-efa-dra-driver" not in got_none.desired.resources - - -def test_custom_credentials() -> None: - """Custom credentials flow through to all cloud MRs.""" - # When spec.credentials is set with a custom type and name, every cloud - # provider MR (VPC, subnets, IAM roles, EKS cluster, node groups, addons, - # EFS resources, autoscaler IAM resources) carries the corresponding - # providerConfigRef. The kubeconfig-based resources (provider-config-kubernetes, - # provider-config-helm, release-*, storage-class-*) are unaffected. - ck = "ProviderConfig" - cn = "my-aws-account" - creds = v1alpha1.Credentials(type=ck, name=cn) - - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr(credentials=creds).model_dump(exclude_none=True, mode="json"), - ), - ), - ), - ) - - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - rs = got.desired.resources - - cloud_checks = { - "vpc": _vpc(ck, cn), - "subnet-0": _subnet(_SUBNET_A, "us-west-2a", "10.0.0.0/20", ck, cn), - "subnet-1": _subnet(_SUBNET_B, "us-west-2b", "10.0.16.0/20", ck, cn), - "subnet-2": _subnet(_SUBNET_C, "us-west-2c", "10.0.32.0/20", ck, cn), - "private-subnet-0": _private_subnet(_PRIVATE_SUBNET_A, "us-west-2a", "10.0.48.0/20", ck, cn), - "private-subnet-1": _private_subnet(_PRIVATE_SUBNET_B, "us-west-2b", "10.0.64.0/20", ck, cn), - "private-subnet-2": _private_subnet(_PRIVATE_SUBNET_C, "us-west-2c", "10.0.80.0/20", ck, cn), - "internet-gateway": _internet_gateway(ck, cn), - "nat-eip": _nat_eip(ck, cn), - "nat-gateway": _nat_gateway("us-west-2a", ck, cn), - "route-table": _route_table(ck, cn), - "route-default": _route_default(ck, cn), - "private-route-table": _private_route_table(ck, cn), - "private-route-default": _private_route_default(ck, cn), - "route-table-association-0": _route_table_association("us-west-2a", ck, cn), - "route-table-association-1": _route_table_association("us-west-2b", ck, cn), - "route-table-association-2": _route_table_association("us-west-2c", ck, cn), - "private-route-table-association-0": _private_route_table_association("us-west-2a", ck, cn), - "private-route-table-association-1": _private_route_table_association("us-west-2b", ck, cn), - "private-route-table-association-2": _private_route_table_association("us-west-2c", ck, cn), - "iam-role-cluster": _role("cluster", _ASSUME_CLUSTER, ck, cn), - "iam-attach-cluster-policy": _role_policy_attachment( - "cluster", "arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", ck, cn - ), - "iam-role-node": _role("node", _ASSUME_NODE, ck, cn), - "iam-attach-node-worker": _role_policy_attachment( - "node", "arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", ck, cn - ), - "iam-attach-node-cni": _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", ck, cn), - "iam-attach-node-ecr": _role_policy_attachment( - "node", "arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", ck, cn - ), - "cluster": _eks_cluster(ck, cn), - "cluster-auth": _cluster_auth(ck, cn), - "nodegroup-system": _system_node_group(ck, cn), - "nodegroup-gpu-l4": _gpu_node_group(ck, cn), - "addon-vpc-cni": _addon("vpc-cni", ck, cn), - "addon-kube-proxy": _addon("kube-proxy", ck, cn), - "addon-coredns": _addon("coredns", ck, cn), - "efs-filesystem": _efs_filesystem(ck, cn), - "efs-security-group": _efs_security_group(ck, cn), - "efs-security-group-ingress": _efs_security_group_ingress(ck, cn), - "efs-mount-target-0": _efs_mount_target(_PRIVATE_SUBNET_A, ck, cn), - "efs-mount-target-1": _efs_mount_target(_PRIVATE_SUBNET_B, ck, cn), - "efs-mount-target-2": _efs_mount_target(_PRIVATE_SUBNET_C, ck, cn), - "iam-role-efs-csi": _role("efs-csi", _ASSUME_POD_IDENTITY, ck, cn), - "iam-attach-efs-csi": _role_policy_attachment("efs-csi", _POLICY_EFS_CSI, ck, cn), - "addon-eks-pod-identity-agent": _addon("eks-pod-identity-agent", ck, cn), - "pod-identity-efs-csi": _pod_identity_association(ck, cn), - "addon-aws-efs-csi-driver": _addon("aws-efs-csi-driver", ck, cn), - "iam-policy-cluster-autoscaler": _autoscaler_policy(ck, cn), - "iam-role-cluster-autoscaler": _role("cluster-autoscaler", _ASSUME_POD_IDENTITY, ck, cn), - "iam-attach-cluster-autoscaler": _autoscaler_attachment(ck, cn), - "pod-identity-cluster-autoscaler": _autoscaler_pod_identity(ck, cn), - } - - for key, want in cloud_checks.items(): - assert key in rs, f"resource {key!r} not found in desired" - got_dict = resource.struct_to_dict(rs[key].resource) - assert got_dict == want, f"resource {key!r} mismatch" - - # kubeconfig-based resources must NOT carry the cloud providerConfigRef - for key in ("provider-config-kubernetes", "provider-config-helm"): - got_dict = resource.struct_to_dict(rs[key].resource) - assert "providerConfigRef" not in got_dict.get("spec", {}), f"{key} should not have providerConfigRef" + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "internet-gateway": _internet_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-eip": _nat_eip(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-gateway": _nat_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-table": _route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-default": _route_default(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-table": _private_route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-default": _private_route_default( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster": _cluster_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-node": _node_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "cluster": _eks_cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "cluster-auth": _cluster_auth( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "nodegroup-system": _nodegroup_system(cred_kind="ClusterProviderConfig", cred_name="default"), + "nodegroup-gpu-l4": _nodegroup_gpu(cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-filesystem": _efs_filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "efs-security-group": _efs_security_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "storage-class-rwx-efs": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "storage.k8s.io/v1", + "kind": "StorageClass", + "metadata": {"name": "modelplane-rwx-efs"}, + "provisioner": "efs.csi.aws.com", + "parameters": { + "provisioningMode": "efs-ap", + "fileSystemId": "fs-0abc123", + "directoryPerms": "700", + }, + "volumeBindingMode": "Immediate", + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "release-cluster-autoscaler": _autoscaler_release(), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + ), + ), + # The launch template targets the reservation through the capacity-block + # market type. The node group sets no instanceTypes, because EKS takes the + # type from the launch template. + Case( + name="CapacityBlock", + reason="An EKSCluster with a Capacity Block pool composes a launch template targeting its reservation and a CAPACITY_BLOCK node group that uses the template.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-h200", + role="GPU", + instanceType="p5en.48xlarge", + nodeCount=2, + minNodeCount=0, + maxNodeCount=2, + diskSizeGb=1024, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h200"), + capacityBlock=v1alpha1.CapacityBlock(capacityReservationId="cr-0123456789abcdef0"), + zones=[v1alpha1.Zone("us-west-2a")], + ), + ], + ), + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ClusterProviderConfig", cred_name="default"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "internet-gateway": _internet_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-eip": _nat_eip(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-gateway": _nat_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-table": _route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-default": _route_default(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-table": _private_route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-default": _private_route_default( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster": _cluster_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-node": _node_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "cluster": _eks_cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster-auth": _cluster_auth( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-system": _nodegroup_system(cred_kind="ClusterProviderConfig", cred_name="default"), + "launch-template-gpu-h200": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "LaunchTemplate", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "name": "test-cluster-lt-gpu-h200-83c00", + "instanceType": "p5en.48xlarge", + "blockDeviceMappings": [ + {"deviceName": "/dev/xvda", "ebs": {"volumeSize": 1024}}, + ], + "instanceMarketOptions": {"marketType": "capacity-block"}, + "capacityReservationSpecification": { + "capacityReservationPreference": "capacity-reservations-only", + "capacityReservationTarget": { + "capacityReservationId": "cr-0123456789abcdef0", + }, + }, + }, + }, + } + ), + ), + "nodegroup-gpu-h200": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "managementPolicies": ["Observe", "Create", "Update", "Delete"], + "initProvider": {"scalingConfig": {"desiredSize": 2}}, + "forProvider": { + "region": "us-west-2", + "amiType": "AL2023_x86_64_NVIDIA", + "clusterNameSelector": {"matchControllerRef": True}, + "nodeRoleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "node"}, + }, + "capacityType": "CAPACITY_BLOCK", + "launchTemplate": { + "name": "test-cluster-lt-gpu-h200-83c00", + "version": "$Latest", + }, + "subnetIdRefs": [{"name": "test-cluster-private-subnet-us-west-2a-6a89f"}], + "scalingConfig": {"minSize": 0, "maxSize": 2}, + "labels": { + "modelplane.ai/gpu": "nvidia-h200", + "modelplane.ai/pool": "gpu-h200", + }, + "taint": [ + { + "key": "nvidia.com/gpu", + "value": "true", + "effect": "NO_SCHEDULE", + }, + ], + }, + }, + } + ), + ), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-filesystem": _efs_filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "efs-security-group": _efs_security_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + ), + ), + # The node group's launch template carries one EFA interface per network card + # (card 0 keeps device index 0 for the node's IP traffic, the rest device + # index 1 for RDMA), the cluster gets an EFA security group with + # self-referencing all-traffic ingress and egress rules, and the node group + # references the launch template instead of setting instanceTypes. + Case( + name="EFAFirstPass", + reason="With nothing observed, an EKSCluster with an EFA pool composes an EFA launch template, still without security groups, and an EFA security group with its rules.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-h200", + role="GPU", + instanceType="p5en.48xlarge", + nodeCount=2, + minNodeCount=0, + maxNodeCount=2, + diskSizeGb=1024, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h200"), + fabric="EFA", + zones=[v1alpha1.Zone("us-west-2a")], + ), + ], + ), + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ClusterProviderConfig", cred_name="default"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "internet-gateway": _internet_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-eip": _nat_eip(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-gateway": _nat_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-table": _route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-default": _route_default(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-table": _private_route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-default": _private_route_default( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster": _cluster_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-node": _node_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "cluster": _eks_cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster-auth": _cluster_auth( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-system": _nodegroup_system(cred_kind="ClusterProviderConfig", cred_name="default"), + "launch-template-gpu-h200": _launch_template_efa(security_groups=None), + "efa-security-group": _efa_security_group(), + "efa-security-group-ingress": _efa_security_group_ingress(), + "efa-security-group-egress": _efa_security_group_egress(), + "nodegroup-gpu-h200": _nodegroup_gpu_efa(), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-filesystem": _efs_filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "efs-security-group": _efs_security_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + ), + ), + # Both security group IDs are observed, the EFA group's in its external name + # and the cluster group's in the cluster's status, so every interface carries + # both, EFA first, and none asks for a public IP. The cluster is observed but + # not Ready, so both Helm releases are composed and the cluster isn't marked + # ready. + Case( + name="SecurityGroupsObserved", + reason="With the EFA security group and the cluster observed, an EKSCluster puts both their security group ids on every EFA interface in the launch template.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-h200", + role="GPU", + instanceType="p5en.48xlarge", + nodeCount=2, + minNodeCount=0, + maxNodeCount=2, + diskSizeGb=1024, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h200"), + fabric="EFA", + zones=[v1alpha1.Zone("us-west-2a")], + ), + ], + ), + resources={ + "cluster": _observed_eks_cluster(cluster_security_group_id="sg-0cluster", ready_condition=False), + "efa-security-group": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "SecurityGroup", + "metadata": { + "name": "test-cluster-efa-sg-602a9", + "labels": {"modelplane.ai/fabric": "EFA"}, + "annotations": {"crossplane.io/external-name": "sg-0efa"}, + }, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "name": "test-cluster-efa", + "description": "EFA OS-bypass traffic between gang nodes", + "vpcIdSelector": {"matchControllerRef": True}, + }, + }, + } + ), + ), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ClusterProviderConfig", cred_name="default"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "internet-gateway": _internet_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-eip": _nat_eip(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-gateway": _nat_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-table": _route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-default": _route_default(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-table": _private_route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-default": _private_route_default( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster": _cluster_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-node": _node_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "cluster": _eks_cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster-auth": _cluster_auth( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-system": _nodegroup_system(cred_kind="ClusterProviderConfig", cred_name="default"), + "launch-template-gpu-h200": _launch_template_efa(security_groups=["sg-0efa", "sg-0cluster"]), + "efa-security-group": _efa_security_group(), + "efa-security-group-ingress": _efa_security_group_ingress(), + "efa-security-group-egress": _efa_security_group_egress(), + "nodegroup-gpu-h200": _nodegroup_gpu_efa(), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-filesystem": _efs_filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "efs-security-group": _efs_security_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "release-cluster-autoscaler": _autoscaler_release(), + "release-efa-dra-driver": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "forProvider": { + "chart": { + "name": "aws-dranet", + "repository": "https://aws.github.io/eks-charts", + "version": "1.0.0", + }, + "namespace": "kube-system", + "values": { + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}, + ], + }, + }, + }, + } + ) + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + ), + ), + # Like the autoscaler, the EFA DRA driver release is gated on the cluster being + # observed so provider-helm can reach it. + Case( + name="EFAClusterObserved", + reason="With the cluster observed, an EKSCluster with an EFA pool composes the EFA DRA driver release alongside the autoscaler release.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-h200", + role="GPU", + instanceType="p5en.48xlarge", + nodeCount=2, + minNodeCount=0, + maxNodeCount=2, + diskSizeGb=1024, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h200"), + fabric="EFA", + zones=[v1alpha1.Zone("us-west-2a")], + ), + ], + ), + resources={ + "cluster": _observed_eks_cluster(cluster_security_group_id=None, ready_condition=True), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ClusterProviderConfig", cred_name="default"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "internet-gateway": _internet_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-eip": _nat_eip(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-gateway": _nat_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-table": _route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-default": _route_default(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-table": _private_route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-default": _private_route_default( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster": _cluster_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-node": _node_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "cluster": _eks_cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "cluster-auth": _cluster_auth( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-system": _nodegroup_system(cred_kind="ClusterProviderConfig", cred_name="default"), + "launch-template-gpu-h200": _launch_template_efa(security_groups=None), + "efa-security-group": _efa_security_group(), + "efa-security-group-ingress": _efa_security_group_ingress(), + "efa-security-group-egress": _efa_security_group_egress(), + "nodegroup-gpu-h200": _nodegroup_gpu_efa(), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-filesystem": _efs_filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "efs-security-group": _efs_security_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "release-cluster-autoscaler": _autoscaler_release(), + "release-efa-dra-driver": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "forProvider": { + "chart": { + "name": "aws-dranet", + "repository": "https://aws.github.io/eks-charts", + "version": "1.0.0", + }, + "namespace": "kube-system", + "values": { + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}, + ], + }, + }, + }, + } + ) + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + ), + ), + Case( + name="NoFabricClusterObserved", + reason="With the cluster observed, an EKSCluster whose pool has no fabric composes the autoscaler release but no EFA DRA driver release.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-l4", + role="GPU", + instanceType="g6.xlarge", + nodeCount=1, + minNodeCount=0, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l4"), + zones=[v1alpha1.Zone("us-west-2a"), v1alpha1.Zone("us-west-2b")], + ), + ], + ), + resources={ + "cluster": _observed_eks_cluster(cluster_security_group_id=None, ready_condition=True), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ClusterProviderConfig", cred_name="default"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "internet-gateway": _internet_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-eip": _nat_eip(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-gateway": _nat_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-table": _route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-default": _route_default(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-table": _private_route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-default": _private_route_default( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster": _cluster_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-node": _node_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "cluster": _eks_cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "cluster-auth": _cluster_auth( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-system": _nodegroup_system(cred_kind="ClusterProviderConfig", cred_name="default"), + "nodegroup-gpu-l4": _nodegroup_gpu(cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-filesystem": _efs_filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "efs-security-group": _efs_security_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "release-cluster-autoscaler": _autoscaler_release(), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + ), + ), + # The ProviderConfigs reach the cluster through its kubeconfig, so they don't + # carry the cloud credentials. + Case( + name="CustomCredentials", + reason="The ProviderConfig an EKSCluster's spec.credentials names becomes every AWS managed resource's providerConfigRef.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-aws-account"), + node_pools=[ + v1alpha1.NodePool( + name="gpu-l4", + role="GPU", + instanceType="g6.xlarge", + nodeCount=1, + minNodeCount=0, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l4"), + zones=[v1alpha1.Zone("us-west-2a"), v1alpha1.Zone("us-west-2b")], + ), + ], + ), + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "internet-gateway": _internet_gateway(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "nat-eip": _nat_eip(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "nat-gateway": _nat_gateway(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "route-table": _route_table(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "route-default": _route_default(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "private-route-table": _private_route_table(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "private-route-default": _private_route_default( + cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "iam-role-cluster": _cluster_role(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "iam-role-node": _node_role(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "cluster": _eks_cluster( + cred_kind="ProviderConfig", cred_name="my-aws-account", ready=fnv1.READY_UNSPECIFIED + ), + "cluster-auth": _cluster_auth( + cred_kind="ProviderConfig", cred_name="my-aws-account", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-system": _nodegroup_system(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "nodegroup-gpu-l4": _nodegroup_gpu(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ProviderConfig", cred_name="my-aws-account"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ProviderConfig", cred_name="my-aws-account"), + "efs-filesystem": _efs_filesystem( + cred_kind="ProviderConfig", cred_name="my-aws-account", ready=fnv1.READY_UNSPECIFIED + ), + "efs-security-group": _efs_security_group(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + ), + ), +] + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes EKS cluster infrastructure.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-gke-cluster/tests/test_fn.py b/functions/compose-gke-cluster/tests/test_fn.py index 07898ac6b..c3ccd534d 100644 --- a/functions/compose-gke-cluster/tests/test_fn.py +++ b/functions/compose-gke-cluster/tests/test_fn.py @@ -34,39 +34,14 @@ class Case: """A test case for compose-gke-cluster.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse -_DEFAULT_CRED_KIND = "ClusterProviderConfig" -_DEFAULT_CRED_NAME = "default" - -_GCP_PROVIDER_CONFIG = { - "apiVersion": "gcp.m.upbound.io/v1beta1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "default"}, - "spec": { - "projectID": "my-gcp-project", - "credentials": { - "source": "Secret", - "secretRef": { - "name": "gcp-credentials", - "namespace": "crossplane-system", - "key": "credentials", - }, - }, - }, -} - -_GCP_PROVIDER_CONFIG_SELECTOR = fnv1.ResourceSelector( - api_version="gcp.m.upbound.io/v1beta1", - kind="ClusterProviderConfig", - match_name="default", -) - - -def _gke_xr(credentials: v1alpha1.Credentials | None = None) -> v1alpha1.GKECluster: - return v1alpha1.GKECluster( +def _xr(*, credentials: v1alpha1.Credentials | None, node_pools: list[v1alpha1.NodePool]) -> fnv1.Resource: + """The observed GKECluster XR, with the given credentials and node pools.""" + xr = v1alpha1.GKECluster( metadata=metav1.ObjectMeta( name="test-cluster", namespace="modelplane-system", @@ -74,638 +49,702 @@ def _gke_xr(credentials: v1alpha1.Credentials | None = None) -> v1alpha1.GKEClus spec=v1alpha1.Spec( region="us-central1", credentials=credentials, - nodePools=[ - v1alpha1.NodePool( - name="gpu-pool", - role="GPU", - machineType="a2-highgpu-8g", - gpu=v1alpha1.Gpu( - acceleratorType="nvidia-tesla-a100", - acceleratorCount=8, - ), - ), - ], + nodePools=node_pools, ), ) + return fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json", by_alias=True))) -def _network(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "compute.gcp.m.upbound.io/v1beta1", - "kind": "Network", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "autoCreateSubnetworks": False, - }, - }, - } - - -def _projectservice_filestore(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ProjectService", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "service": "file.googleapis.com", - "disableOnDestroy": False, - }, - }, - } - - -def _subnet(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "compute.gcp.m.upbound.io/v1beta1", - "kind": "Subnetwork", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-central1", - "networkSelector": {"matchControllerRef": True}, - "ipCidrRange": "10.0.0.0/24", - "secondaryIpRange": [ - {"rangeName": "pods", "ipCidrRange": "10.1.0.0/16"}, - {"rangeName": "services", "ipCidrRange": "10.2.0.0/16"}, - ], - }, - }, - } - - -def _cluster(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "container.gcp.m.upbound.io/v1beta1", - "kind": "Cluster", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "location": "us-central1", - "deletionProtection": False, - "removeDefaultNodePool": True, - "initialNodeCount": 1, - "minMasterVersion": "1.35", - "networkSelector": {"matchControllerRef": True}, - "subnetworkSelector": {"matchControllerRef": True}, - "ipAllocationPolicy": { - "clusterSecondaryRangeName": "pods", - "servicesSecondaryRangeName": "services", - }, - "releaseChannel": {"channel": "REGULAR"}, - "workloadIdentityConfig": { - "workloadPool": "my-gcp-project.svc.id.goog", - }, - "addonsConfig": { - "gcpFilestoreCsiDriverConfig": {"enabled": True}, - }, - }, - "writeConnectionSecretToRef": { - "name": "test-cluster-kubeconfig-55b57", - }, - }, - } - - -def _nodepool_system(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "container.gcp.m.upbound.io/v1beta1", - "kind": "NodePool", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "location": "us-central1", - "clusterSelector": {"matchControllerRef": True}, - "initialNodeCount": 1, - "autoscaling": {"minNodeCount": 1, "maxNodeCount": 2}, - "nodeConfig": { - "machineType": "e2-standard-4", - "imageType": "COS_CONTAINERD", - "oauthScopes": [ - "https://www.googleapis.com/auth/cloud-platform", - ], - "labels": {"modelplane.ai/pool": "system"}, - }, - }, - }, - } - - -def _nodepool_gpu(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "container.gcp.m.upbound.io/v1beta1", - "kind": "NodePool", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "location": "us-central1", - "clusterSelector": {"matchControllerRef": True}, - "initialNodeCount": 1, - "autoscaling": {"minNodeCount": 0, "maxNodeCount": 8}, - "nodeConfig": { - "machineType": "a2-highgpu-8g", - "diskSizeGb": 100, - "imageType": "COS_CONTAINERD", - "oauthScopes": [ - "https://www.googleapis.com/auth/cloud-platform", - ], - "guestAccelerator": [ +def _desired_xr() -> fnv1.Resource: + """The desired XR, publishing its connection Secrets and cache StorageClass.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "secrets": [ { - "type": "nvidia-tesla-a100", - "count": 8, - "gpuDriverInstallationConfig": { - "gpuDriverVersion": "DEFAULT", - }, + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + { + "type": "GoogleApplicationCredentials", + "name": "test-cluster-sa-key-3295c", + "key": "private_key", }, ], - "labels": { - "modelplane.ai/gpu": "nvidia-tesla-a100", - "modelplane.ai/pool": "gpu-pool", - "cloud.google.com/gke-nvidia-gpu-dra-driver": "true", + "cache": {"storageClassName": "modelplane-rwx"}, + }, + } + ), + ) + + +def _gcp_provider_config(*, kind: str, name: str, namespace: str | None) -> fnv1.Resource: + """The GCP provider config the function requires, in a namespace if given.""" + metadata = {"name": name} + if namespace is not None: + metadata["namespace"] = namespace + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "gcp.m.upbound.io/v1beta1", + "kind": kind, + "metadata": metadata, + "spec": { + "projectID": "my-gcp-project", + "credentials": { + "source": "Secret", + "secretRef": { + "name": "gcp-credentials", + "namespace": "crossplane-system", + "key": "credentials", + }, }, }, - }, - }, - } - - -def _service_account(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ServiceAccount", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "displayName": "Crossplane GKECluster test-cluster", - }, - }, - } - - -def _service_account_key(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ServiceAccountKey", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "serviceAccountIdSelector": {"matchControllerRef": True}, - }, - "writeConnectionSecretToRef": { - "name": "test-cluster-sa-key-3295c", - }, - }, - } - - -def _iam_binding( - sa_email: str = "test-sa@my-gcp-project.iam.gserviceaccount.com", - cred_kind: str = _DEFAULT_CRED_KIND, - cred_name: str = _DEFAULT_CRED_NAME, -) -> dict: - return { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ProjectIAMMember", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "role": "roles/container.admin", - "member": f"serviceAccount:{sa_email}", - "project": "my-gcp-project", - }, - }, - } - - -def _provider_config_kubernetes() -> dict: - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ProviderConfig", - "metadata": {"name": "test-cluster-kubeconfig-55b57"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "name": "test-cluster-kubeconfig-55b57", - "namespace": "modelplane-system", - "key": "kubeconfig", + } + ), + ) + + +def _network(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed VPC Network.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.gcp.m.upbound.io/v1beta1", + "kind": "Network", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "autoCreateSubnetworks": False, + }, }, - }, - "identity": { - "type": "GoogleApplicationCredentials", - "source": "Secret", - "secretRef": { - "name": "test-cluster-sa-key-3295c", - "namespace": "modelplane-system", - "key": "private_key", + } + ), + ready=ready, + ) + + +def _projectservice_filestore(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The composed ProjectService that enables the Filestore API.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ProjectService", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "service": "file.googleapis.com", + "disableOnDestroy": False, + }, }, - }, - }, - } - - -def _provider_config_helm() -> dict: - return { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "ProviderConfig", - "metadata": {"name": "test-cluster-kubeconfig-55b57"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "name": "test-cluster-kubeconfig-55b57", - "namespace": "modelplane-system", - "key": "kubeconfig", + } + ), + ) + + +def _subnet(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The composed Subnetwork, with secondary ranges for pods and services.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.gcp.m.upbound.io/v1beta1", + "kind": "Subnetwork", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-central1", + "networkSelector": {"matchControllerRef": True}, + "ipCidrRange": "10.0.0.0/24", + "secondaryIpRange": [ + {"rangeName": "pods", "ipCidrRange": "10.1.0.0/16"}, + {"rangeName": "services", "ipCidrRange": "10.2.0.0/16"}, + ], + }, }, - }, - "identity": { - "type": "GoogleApplicationCredentials", - "source": "Secret", - "secretRef": { - "name": "test-cluster-sa-key-3295c", - "namespace": "modelplane-system", - "key": "private_key", + } + ), + ) + + +def _cluster(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The composed GKE Cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "container.gcp.m.upbound.io/v1beta1", + "kind": "Cluster", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "location": "us-central1", + "deletionProtection": False, + "removeDefaultNodePool": True, + "initialNodeCount": 1, + "minMasterVersion": "1.35", + "networkSelector": {"matchControllerRef": True}, + "subnetworkSelector": {"matchControllerRef": True}, + "ipAllocationPolicy": { + "clusterSecondaryRangeName": "pods", + "servicesSecondaryRangeName": "services", + }, + "releaseChannel": {"channel": "REGULAR"}, + "workloadIdentityConfig": { + "workloadPool": "my-gcp-project.svc.id.goog", + }, + "addonsConfig": { + "gcpFilestoreCsiDriverConfig": {"enabled": True}, + }, + }, + "writeConnectionSecretToRef": { + "name": "test-cluster-kubeconfig-55b57", + }, }, - }, - }, - } - - -def _storage_class_rwx(network_name: str) -> dict: - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { - "kind": "ProviderConfig", - "name": "test-cluster-kubeconfig-55b57", - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "storage.k8s.io/v1", - "kind": "StorageClass", - "metadata": {"name": "modelplane-rwx"}, - "provisioner": "filestore.csi.storage.gke.io", - "parameters": { - "tier": "enterprise", - "network": network_name, + } + ), + ) + + +def _nodepool_system(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The composed system NodePool the function adds to every cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "container.gcp.m.upbound.io/v1beta1", + "kind": "NodePool", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "location": "us-central1", + "clusterSelector": {"matchControllerRef": True}, + "initialNodeCount": 1, + "autoscaling": {"minNodeCount": 1, "maxNodeCount": 2}, + "nodeConfig": { + "machineType": "e2-standard-4", + "imageType": "COS_CONTAINERD", + "oauthScopes": [ + "https://www.googleapis.com/auth/cloud-platform", + ], + "labels": {"modelplane.ai/pool": "system"}, + }, }, - "volumeBindingMode": "Immediate", - "allowVolumeExpansion": True, }, - }, - }, - } - - -def _expected_status() -> dict: - return { - "status": { - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-55b57", - "key": "kubeconfig", + } + ), + ) + + +def _nodepool_gpu(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The composed NodePool for the XR's gpu-pool.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "container.gcp.m.upbound.io/v1beta1", + "kind": "NodePool", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "location": "us-central1", + "clusterSelector": {"matchControllerRef": True}, + "initialNodeCount": 1, + "autoscaling": {"minNodeCount": 0, "maxNodeCount": 8}, + "nodeConfig": { + "machineType": "a2-highgpu-8g", + "diskSizeGb": 100, + "imageType": "COS_CONTAINERD", + "oauthScopes": [ + "https://www.googleapis.com/auth/cloud-platform", + ], + "guestAccelerator": [ + { + "type": "nvidia-tesla-a100", + "count": 8, + "gpuDriverInstallationConfig": { + "gpuDriverVersion": "DEFAULT", + }, + }, + ], + "labels": { + "modelplane.ai/gpu": "nvidia-tesla-a100", + "modelplane.ai/pool": "gpu-pool", + "cloud.google.com/gke-nvidia-gpu-dra-driver": "true", + }, + }, + }, }, - { - "type": "GoogleApplicationCredentials", - "name": "test-cluster-sa-key-3295c", - "key": "private_key", + } + ), + ) + + +def _service_account(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed GCP ServiceAccount.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ServiceAccount", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "displayName": "Crossplane GKECluster test-cluster", + }, }, - ], - "cache": {"storageClassName": "modelplane-rwx"}, - }, - } - - -def _compose_cases() -> list[Case]: - """The cases for test_compose.""" - req1 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr().model_dump(exclude_none=True, mode="json"), - ), - ), + } ), + ready=ready, ) - req1.required_resources["gcp-provider-config"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(_GCP_PROVIDER_CONFIG)) + + +def _service_account_key(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The composed ServiceAccountKey.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ServiceAccountKey", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "serviceAccountIdSelector": {"matchControllerRef": True}, + }, + "writeConnectionSecretToRef": { + "name": "test-cluster-sa-key-3295c", + }, + }, + } + ), ) - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ), - "projectservice-filestore": fnv1.Resource( - resource=resource.dict_to_struct(_projectservice_filestore()), - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "nodepool-system": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_system()), - ), - "nodepool-gpu-pool": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu()), - ), - "service-account": fnv1.Resource( - resource=resource.dict_to_struct(_service_account()), - ), - "service-account-key": fnv1.Resource( - resource=resource.dict_to_struct(_service_account_key()), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_kubernetes()), - ready=fnv1.READY_TRUE, + +def _provider_config(*, api_version: str) -> fnv1.Resource: + """The provider-kubernetes or provider-helm ProviderConfig for the composed cluster, which is always ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": api_version, + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster-kubeconfig-55b57"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", + "key": "kubeconfig", + }, + }, + "identity": { + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": { + "name": "test-cluster-sa-key-3295c", + "namespace": "modelplane-system", + "key": "private_key", + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +COMPOSE_CASES = [ + Case( + name="FirstPass", + reason="With nothing observed, a GKECluster composes its GCP infrastructure and ProviderConfigs, but not the IAM binding or StorageClass.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-pool", + role="GPU", + machineType="a2-highgpu-8g", + gpu=v1alpha1.Gpu(acceleratorType="nvidia-tesla-a100", acceleratorCount=8), + ), + ], ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, + ), + required_resources={ + "gcp-provider-config": fnv1.Resources( + items=[_gcp_provider_config(kind="ClusterProviderConfig", name="default", namespace=None)], ), }, ), - context=structpb.Struct(), - ) - want1.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) - - req2 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr().model_dump(exclude_none=True, mode="json"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "network": _network( + cred_kind="ClusterProviderConfig", + cred_name="default", + ready=fnv1.READY_UNSPECIFIED, + ), + "projectservice-filestore": _projectservice_filestore( + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet": _subnet(cred_kind="ClusterProviderConfig", cred_name="default"), + "cluster": _cluster(cred_kind="ClusterProviderConfig", cred_name="default"), + "nodepool-system": _nodepool_system(cred_kind="ClusterProviderConfig", cred_name="default"), + "nodepool-gpu-pool": _nodepool_gpu(cred_kind="ClusterProviderConfig", cred_name="default"), + "service-account": _service_account( + cred_kind="ClusterProviderConfig", + cred_name="default", + ready=fnv1.READY_UNSPECIFIED, + ), + "service-account-key": _service_account_key(cred_kind="ClusterProviderConfig", cred_name="default"), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gcp-provider-config": fnv1.ResourceSelector( + api_version="gcp.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", + ), + }, + ), + ), + ), + # An XR with no composed resources would aggregate to trivially ready, so + # the function marks it not ready. + Case( + name="ProviderConfigMissing", + reason="With the GCP provider config missing and no cluster observed to take the project from, a GKECluster composes nothing and isn't ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-pool", + role="GPU", + machineType="a2-highgpu-8g", + gpu=v1alpha1.Gpu(acceleratorType="nvidia-tesla-a100", acceleratorCount=8), + ), + ], ), ), - resources={ - "service-account": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ServiceAccount", - "spec": { - "forProvider": {}, - }, - "status": { - "atProvider": { - "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", - }, - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - }, - } + required_resources={"gcp-provider-config": fnv1.Resources()}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for GCP ClusterProviderConfig default", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gcp-provider-config": fnv1.ResourceSelector( + api_version="gcp.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", ), + }, + ), + ), + ), + Case( + name="ResourcesReady", + reason="With the service account and network observed Ready, a GKECluster marks them ready and composes the IAM binding and StorageClass from their email and external name.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-pool", + role="GPU", + machineType="a2-highgpu-8g", + gpu=v1alpha1.Gpu(acceleratorType="nvidia-tesla-a100", acceleratorCount=8), + ), + ], ), - "network": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "compute.gcp.m.upbound.io/v1beta1", - "kind": "Network", - # The external-name annotation carries the - # provider-generated VPC name, which the - # function pins the Filestore StorageClass to. - "metadata": { - "annotations": {"crossplane.io/external-name": "test-cluster-abc12"}, - }, - "spec": { - "forProvider": { - "autoCreateSubnetworks": False, + resources={ + "service-account": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ServiceAccount", + "spec": {"forProvider": {}}, + "status": { + "atProvider": { + "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", + }, + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], }, - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", + } + ), + ), + "network": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.gcp.m.upbound.io/v1beta1", + "kind": "Network", + "metadata": { + # The external-name annotation carries the + # provider-generated VPC name, which the + # function pins the Filestore StorageClass to. + "annotations": {"crossplane.io/external-name": "test-cluster-abc12"}, + }, + "spec": { + "forProvider": { + "autoCreateSubnetworks": False, }, - ], - }, - } + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), ), + }, + ), + required_resources={ + "gcp-provider-config": fnv1.Resources( + items=[_gcp_provider_config(kind="ClusterProviderConfig", name="default", namespace=None)], ), }, ), - ) - req2.required_resources["gcp-provider-config"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(_GCP_PROVIDER_CONFIG)) - ) - - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "network": _network( + cred_kind="ClusterProviderConfig", + cred_name="default", + ready=fnv1.READY_TRUE, + ), + "projectservice-filestore": _projectservice_filestore( + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + # Composed once the observed Network's external name gives + # the VPC to pin Filestore to. A StorageClass has no Ready + # condition, hence SuccessfulCreate, and the Object omits + # Delete so it dies with the cluster rather than wedging + # teardown on the deleted kubeconfig Secret. + "storage-class-rwx": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "storage.k8s.io/v1", + "kind": "StorageClass", + "metadata": {"name": "modelplane-rwx"}, + "provisioner": "filestore.csi.storage.gke.io", + "parameters": { + "tier": "enterprise", + "network": "test-cluster-abc12", + }, + "volumeBindingMode": "Immediate", + "allowVolumeExpansion": True, + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "subnet": _subnet(cred_kind="ClusterProviderConfig", cred_name="default"), + "cluster": _cluster(cred_kind="ClusterProviderConfig", cred_name="default"), + "nodepool-system": _nodepool_system(cred_kind="ClusterProviderConfig", cred_name="default"), + "nodepool-gpu-pool": _nodepool_gpu(cred_kind="ClusterProviderConfig", cred_name="default"), + "service-account": _service_account( + cred_kind="ClusterProviderConfig", + cred_name="default", + ready=fnv1.READY_TRUE, + ), + "service-account-key": _service_account_key(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-binding": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ProjectIAMMember", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "role": "roles/container.admin", + "member": "serviceAccount:test-sa@my-gcp-project.iam.gserviceaccount.com", + "project": "my-gcp-project", + }, + }, + } + ), + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, ), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ready=fnv1.READY_TRUE, - ), - "projectservice-filestore": fnv1.Resource( - resource=resource.dict_to_struct(_projectservice_filestore()), - ), - # With the network name known, the managed Filestore - # StorageClass is composed against the cluster's own - # provider-kubernetes ProviderConfig, pinned to the - # observed VPC. StorageClass has no Ready condition, - # so readiness is SuccessfulCreate. It's orphaned (no - # Delete policy) so it dies with the cluster instead of - # wedging on a deleted kubeconfig Secret during teardown. - "storage-class-rwx": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class_rwx("test-cluster-abc12")), - ready=fnv1.READY_TRUE, - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "nodepool-system": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_system()), - ), - "nodepool-gpu-pool": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu()), - ), - "service-account": fnv1.Resource( - resource=resource.dict_to_struct(_service_account()), - ready=fnv1.READY_TRUE, - ), - "service-account-key": fnv1.Resource( - resource=resource.dict_to_struct(_service_account_key()), - ), - "iam-binding": fnv1.Resource( - resource=resource.dict_to_struct(_iam_binding()), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_kubernetes()), - ready=fnv1.READY_TRUE, + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gcp-provider-config": fnv1.ResourceSelector( + api_version="gcp.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", + ), + }, + ), + ), + ), + # The kubeconfig-based ProviderConfigs carry no providerConfigRef, so they're + # unaffected. + Case( + name="CustomCredentials", + reason="The ProviderConfig a GKECluster's spec.credentials names becomes every GCP managed resource's providerConfigRef.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-gcp-account"), + node_pools=[ + v1alpha1.NodePool( + name="gpu-pool", + role="GPU", + machineType="a2-highgpu-8g", + gpu=v1alpha1.Gpu(acceleratorType="nvidia-tesla-a100", acceleratorCount=8), + ), + ], ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, + resources={ + "service-account": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ServiceAccount", + "spec": {"forProvider": {}}, + "status": { + "atProvider": { + "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", + }, + }, + } + ), + ), + }, + ), + required_resources={ + "gcp-provider-config": fnv1.Resources( + items=[ + _gcp_provider_config( + kind="ProviderConfig", + name="my-gcp-account", + namespace="crossplane-system", + ), + ], ), }, ), - context=structpb.Struct(), - ) - want2.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) - - # The ProviderConfig resolved to nothing and no cluster is observed to - # take the project from, so nothing can be composed. The XR is marked - # not ready rather than left to aggregate to trivially ready. - req3 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr().model_dump(exclude_none=True, mode="json"), - ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "network": _network( + cred_kind="ProviderConfig", + cred_name="my-gcp-account", + ready=fnv1.READY_UNSPECIFIED, + ), + "projectservice-filestore": _projectservice_filestore( + cred_kind="ProviderConfig", + cred_name="my-gcp-account", + ), + "subnet": _subnet(cred_kind="ProviderConfig", cred_name="my-gcp-account"), + "cluster": _cluster(cred_kind="ProviderConfig", cred_name="my-gcp-account"), + "nodepool-system": _nodepool_system(cred_kind="ProviderConfig", cred_name="my-gcp-account"), + "nodepool-gpu-pool": _nodepool_gpu(cred_kind="ProviderConfig", cred_name="my-gcp-account"), + "service-account": _service_account( + cred_kind="ProviderConfig", + cred_name="my-gcp-account", + ready=fnv1.READY_UNSPECIFIED, + ), + "service-account-key": _service_account_key(cred_kind="ProviderConfig", cred_name="my-gcp-account"), + "iam-binding": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ProjectIAMMember", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "my-gcp-account"}, + "forProvider": { + "role": "roles/container.admin", + "member": "serviceAccount:test-sa@my-gcp-project.iam.gserviceaccount.com", + "project": "my-gcp-project", + }, + }, + } + ), + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, ), - ), - ) - req3.required_resources["gcp-provider-config"].SetInParent() - - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for GCP ClusterProviderConfig default", + context=structpb.Struct(), + # A ProviderConfig, unlike a ClusterProviderConfig, is namespaced, + # so the function requires it from the XR's namespace. Crossplane + # wouldn't return the request's crossplane-system one for this + # selector, but the function doesn't check its namespace. + requirements=fnv1.Requirements( + resources={ + "gcp-provider-config": fnv1.ResourceSelector( + api_version="gcp.m.upbound.io/v1beta1", + kind="ProviderConfig", + match_name="my-gcp-account", + namespace="modelplane-system", + ), + }, ), - ], - context=structpb.Struct(), - ) - want3.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) - - return [ - Case(name="first pass composes infra resources; IAM binding gated", req=req1, want=want1), - Case(name="a missing ProviderConfig composes nothing and isn't ready", req=req3, want=want3), - Case( - name="second pass with observed SA email composes IAM binding and marks ready resources", - req=req2, - want=want2, ), - ] + ), +] -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction composes GKE cluster infrastructure.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) - - -def test_custom_credentials() -> None: - """Custom credentials flow through to all cloud MRs.""" - # When spec.credentials is set with a custom type and name, every cloud - # provider MR (Network, ProjectService, Subnetwork, Cluster, NodePools, - # ServiceAccount, ServiceAccountKey, IAM binding) carries the corresponding - # providerConfigRef. The kubeconfig-based resources (provider-config-kubernetes, - # provider-config-helm, storage-class-rwx) are unaffected. - ck = "ProviderConfig" - cn = "my-gcp-account" - creds = v1alpha1.Credentials(type=ck, name=cn) - custom_pc = { - "apiVersion": "gcp.m.upbound.io/v1beta1", - "kind": "ProviderConfig", - "metadata": {"name": cn, "namespace": "crossplane-system"}, - "spec": { - "projectID": "my-gcp-project", - "credentials": { - "source": "Secret", - "secretRef": { - "name": "gcp-credentials", - "namespace": "crossplane-system", - "key": "credentials", - }, - }, - }, - } - - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr(credentials=creds).model_dump(exclude_none=True, mode="json"), - ), - ), - resources={ - "service-account": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ServiceAccount", - "spec": { - "forProvider": {}, - }, - "status": { - "atProvider": { - "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", - }, - }, - } - ), - ), - }, - ), - ) - req.required_resources["gcp-provider-config"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(custom_pc)) - ) - - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - rs = got.desired.resources - - cloud_checks = { - "network": _network(ck, cn), - "projectservice-filestore": _projectservice_filestore(ck, cn), - "subnet": _subnet(ck, cn), - "cluster": _cluster(ck, cn), - "nodepool-system": _nodepool_system(ck, cn), - "nodepool-gpu-pool": _nodepool_gpu(ck, cn), - "service-account": _service_account(ck, cn), - "service-account-key": _service_account_key(ck, cn), - "iam-binding": _iam_binding("test-sa@my-gcp-project.iam.gserviceaccount.com", ck, cn), - } - - got_cloud = {key: resource.struct_to_dict(r.resource) for key, r in rs.items() if key in cloud_checks} - assert got_cloud == cloud_checks - - # kubeconfig-based resources must NOT carry the cloud providerConfigRef - for key in ("provider-config-kubernetes", "provider-config-helm"): - got_dict = resource.struct_to_dict(rs[key].resource) - assert "providerConfigRef" not in got_dict.get("spec", {}), f"{key} should not have providerConfigRef" - - custom_selector = fnv1.ResourceSelector( - api_version="gcp.m.upbound.io/v1beta1", - kind="ProviderConfig", - match_name=cn, - namespace="modelplane-system", - ) - assert _to_dict(got.requirements.resources["gcp-provider-config"]) == _to_dict(custom_selector) + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-inference-class/tests/test_fn.py b/functions/compose-inference-class/tests/test_fn.py index 327807286..fb9a0539d 100644 --- a/functions/compose-inference-class/tests/test_fn.py +++ b/functions/compose-inference-class/tests/test_fn.py @@ -34,13 +34,20 @@ class Case: """A test case for compose-inference-class.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + COMPOSE_CASES = [ Case( - name="marks XR ready with Accepted condition and empty status", + name="DRAGPU", + reason="An InferenceClass describing one DRA GPU is ready and Accepted, with an empty status.", req=fnv1.RunFunctionRequest( observed=fnv1.State( composite=fnv1.Resource( @@ -59,7 +66,7 @@ class Case: ), ], ), - ).model_dump(exclude_none=True, mode="json") + ).model_dump(exclude_none=True, mode="json", by_alias=True) ), ), ), @@ -72,6 +79,7 @@ class Case: ready=fnv1.READY_TRUE, ), ), + context=structpb.Struct(), conditions=[ fnv1.Condition( type="Accepted", @@ -79,19 +87,13 @@ class Case: reason="Available", ), ], - context=structpb.Struct(), ), ), ] -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - @pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction marks the InferenceClass ready.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-inference-cluster/tests/test_fn.py b/functions/compose-inference-cluster/tests/test_fn.py index 4120990da..dfbcb7600 100644 --- a/functions/compose-inference-cluster/tests/test_fn.py +++ b/functions/compose-inference-cluster/tests/test_fn.py @@ -15,7 +15,6 @@ """Tests for the compose-inference-cluster function.""" import asyncio -import copy import dataclasses import json @@ -29,198 +28,338 @@ from models.ai.modelplane.inferencecluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -# The internal name Modelplane derives for this cluster's gateway, which -# compose-inference-gateway resolves. Built from the SDK's own child_name, so the -# namespace and suffix are asserted independently of the function under test. -_GATEWAY_HOSTNAME = f"{resource.child_name('gateway', 'test-cluster')}.modelplane-system.svc.cluster.local" - @dataclasses.dataclass -class Case: - """A test case for compose-inference-cluster.""" +class ComposeCase: + """A test case for RunFunction.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse -_ACTIVATION_API_VERSION = "apiextensions.crossplane.io/v1alpha1" +@dataclasses.dataclass +class GatewayHostnameCase: + """A test case for _gateway_hostname.""" + name: str + reason: str + cluster_name: str + want: str -def _observe_activated(req: fnv1.RunFunctionRequest, kinds: tuple[str, ...]) -> None: - """Observe the composed activation policy with the kinds in status.activated, - so the function composes the cluster XR rather than waiting for activation.""" - req.observed.resources["activation"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": _ACTIVATION_API_VERSION, - "kind": "ManagedResourceActivationPolicy", - "status": {"activated": list(kinds)}, - }, - ), + +def _inference_cluster(*, cluster: v1alpha1.Cluster, node_pools: list[v1alpha1.NodePool] | None) -> fnv1.Resource: + """The observed InferenceCluster XR, test-cluster, with node_pools unless they're None.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), + spec=v1alpha1.Spec(cluster=cluster, nodePools=node_pools), + ).model_dump(exclude_none=True, mode="json", by_alias=True) ), ) -def _want_activation(want: fnv1.RunFunctionResponse, kinds: tuple[str, ...]) -> None: - """Add the activation policy the function composes for a cloud cluster.""" - want.desired.resources["activation"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": _ACTIVATION_API_VERSION, - "kind": "ManagedResourceActivationPolicy", - "spec": {"activate": list(kinds)}, +def _desired_inference_cluster(*, gpu_pools: list[dict], cache: dict | None, gateway: dict | None) -> fnv1.Resource: + """The desired InferenceCluster XR's status, with its cache and gateway unless they're None.""" + status: dict = { + "providerConfigRef": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "namespace": "modelplane-system", + "gpuPools": gpu_pools, + } + if cache is not None: + status["cache"] = cache + if gateway is not None: + status["gateway"] = gateway + return fnv1.Resource(resource=resource.dict_to_struct({"status": status})) + + +def _inference_class(*, name: str, count: int, memory: str, provisioning: dict) -> fnv1.Resource: + """An InferenceClass of count DRA-claimed NVIDIA GPUs with memory each, as its class requirement returns it.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceClass", + "metadata": {"name": name}, + "spec": { + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": count, + "capacity": {"memory": {"value": memory}}, + }, + ], + "provisioning": provisioning, }, - ), - ready=fnv1.READY_TRUE, - ), + } + ) ) -def _eks_ready_extras(want: fnv1.RunFunctionResponse, storage_class: str) -> None: - """Apply the EKS-ready deltas on top of the EKS first-pass response: mark the - EKSCluster ready and relay the backing cluster's status.cache up to the - InferenceCluster's status.cache.storageClassName.""" - want.desired.resources["eks-cluster"].ready = fnv1.READY_TRUE - status = want.desired.composite.resource.fields["status"].struct_value - status.fields["cache"].struct_value.fields["storageClassName"].string_value = storage_class +def _inference_gateway(*, name: str, cluster: str, status: dict | None) -> fnv1.Resource: + """An InferenceGateway on cluster, as the gateways requirement returns it, with status unless it's None.""" + gateway: dict = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceGateway", + "metadata": {"name": name}, + "spec": {"clusterName": cluster}, + } + if status is not None: + gateway["status"] = status + return fnv1.Resource(resource=resource.dict_to_struct(gateway)) -def _gateways_selector() -> fnv1.ResourceSelector: - """Every InferenceGateway. A cluster gateway accepts client certificates - from each of their CAs, which is how an InferenceGateway proves itself.""" - return fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway") +def _model_cache(*, name: str, namespace: str, cluster: str) -> fnv1.Resource: + """A ModelCache staged onto cluster, as the model-caches requirement returns it.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelCache", + "metadata": {"name": name, "namespace": namespace}, + "spec": {"source": "HuggingFace"}, + "status": {"clusters": [{"name": cluster, "phase": "Ready"}]}, + } + ) + ) -def _replicas_selector(cluster_name: str) -> fnv1.ResourceSelector: - """The ModelReplica guard requirement: replicas scheduled to a cluster, - across all namespaces.""" - sel = fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica") - sel.match_labels.labels.update({"modelplane.ai/cluster": cluster_name}) - return sel +def _observed_activation_policy(*, activated: list[str]) -> fnv1.Resource: + """The observed ManagedResourceActivationPolicy, reporting the kinds it has activated.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "apiextensions.crossplane.io/v1alpha1", + "kind": "ManagedResourceActivationPolicy", + "status": {"activated": activated}, + } + ) + ) -def _routes_selector(cluster_name: str) -> fnv1.ResourceSelector: - """The ModelRoute requirement: routes scheduled to a cluster, across all - namespaces. Their teams' namespaces are mirrored alongside the replicas'.""" - sel = fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelRoute") - sel.match_labels.labels.update({"modelplane.ai/cluster": cluster_name}) - return sel +def _observed_serving_stack(*, gateway: dict) -> fnv1.Resource: + """The observed ServingStack, Ready, with the gateway status it has published.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": {"name": "test-cluster-serving-stack-fd00b"}, + "status": {"conditions": [{"type": "Ready", "status": "True"}], "gateway": gateway}, + } + ) + ) -def _caches_selector() -> fnv1.ResourceSelector: - """Every ModelCache. A cache fans out to many clusters, so it can't be - label-selected to one; the function filters by status.clusters[].""" - return fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache") +def _activation_policy(*, activate: list[str], ready: fnv1.Ready) -> fnv1.Resource: + """The composed ManagedResourceActivationPolicy, activating a cloud's managed resource kinds.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "apiextensions.crossplane.io/v1alpha1", + "kind": "ManagedResourceActivationPolicy", + "spec": {"activate": activate}, + } + ), + ready=ready, + ) -def _replica_item(name: str, namespace: str) -> fnv1.Resource: - """An observed ModelReplica labelled for test-cluster.""" +def _gke_cluster(*, credentials: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The composed GKECluster with l4-pool, using credentials unless they're None.""" + spec: dict = { + "region": "us-central1", + "kubernetesVersion": "1.35", + "nodePools": [ + { + "name": "l4-pool", + "role": "GPU", + "machineType": "g2-standard-48", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": {"acceleratorType": "nvidia-l4", "acceleratorCount": 1}, + "zones": ["us-central1-a"], + }, + ], + } + if credentials is not None: + spec["credentials"] = credentials return fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": name, - "namespace": namespace, - "labels": {"modelplane.ai/cluster": "test-cluster"}, - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": spec, } - ) + ), + ready=ready, ) -def _route_item(name: str, namespace: str) -> fnv1.Resource: - """An observed ModelRoute labelled for test-cluster.""" +def _eks_cluster( + *, zones: list[str], capacity_block: dict | None, fabric: str | None, ready: fnv1.Ready +) -> fnv1.Resource: + """The composed EKSCluster with l4-pool in zones, setting its capacityBlock and fabric unless they're None.""" + pool: dict = { + "name": "l4-pool", + "role": "GPU", + "instanceType": "g6.xlarge", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": {"acceleratorType": "nvidia-l4"}, + "zones": zones, + } + if capacity_block is not None: + pool["capacityBlock"] = capacity_block + if fabric is not None: + pool["fabric"] = fabric return fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelRoute", - "metadata": { - "name": name, - "namespace": namespace, - "labels": {"modelplane.ai/cluster": "test-cluster"}, - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "EKSCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": {"region": "us-west-2", "kubernetesVersion": "1.36", "nodePools": [pool]}, } - ) + ), + ready=ready, ) -def _cache_item(name: str, namespace: str, clusters: list[str]) -> fnv1.Resource: - """An observed ModelCache staging onto the named clusters (status.clusters). - - Caches carry no cluster label; the function reads status.clusters[] to tell - which cluster a cache lands on, so only those naming this one are mirrored.""" +def _vultr_cluster(*, credentials: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The composed VultrCluster with l40s-pool, using credentials unless they're None.""" + spec: dict = { + "region": "ewr", + "kubernetesVersion": "v1.36.2+1", + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "plan": "vcg-l40s-16c-180g-48vram", + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": {"acceleratorType": "nvidia-l40s"}, + }, + ], + } + if credentials is not None: + spec["credentials"] = credentials return fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelCache", - "metadata": {"name": name, "namespace": namespace}, - "spec": {"source": "HuggingFace"}, - "status": {"clusters": [{"name": c, "phase": "Ready"} for c in clusters]}, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "VultrCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": spec, } - ) + ), + ready=ready, ) -def _namespace_object(ns: str, name: str) -> fnv1.Resource: - """The mirrored namespace the function composes for a team with a replica or - route on the cluster, labelled for the gateways' route selector and kept (no - Delete). Marked ready so a new team's namespace can't flap the cluster's - readiness. +def _civo_cluster(*, node_pools: list[dict], credentials: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The composed CivoCluster in LON1 with node_pools, using credentials unless they're None.""" + spec: dict = {"region": "LON1", "nodePools": node_pools} + if credentials is not None: + spec["credentials"] = credentials + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "CivoCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": spec, + } + ), + ready=ready, + ) + - name is spelled out rather than computed with child_name, since the other - functions that land objects in it hardcode the same derivation, and a test - computing it the same way would pass whichever way any of them drifted.""" +def _cluster_provider_config(*, kubeconfig: str, identity: dict | None) -> fnv1.Resource: + """The ClusterProviderConfig reaching test-cluster with the kubeconfig Secret, as identity unless it's None.""" + spec: dict = { + "credentials": { + "source": "Secret", + "secretRef": {"namespace": "modelplane-system", "name": kubeconfig, "key": "kubeconfig"}, + }, + } + if identity is not None: + spec["identity"] = identity return fnv1.Resource( resource=resource.dict_to_struct( { "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { - "kind": "ClusterProviderConfig", - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "v1", - "kind": "Namespace", - "metadata": {"name": name, "labels": {"modelplane.ai/namespace": ns}}, - }, - }, - }, + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": spec, } ), ready=fnv1.READY_TRUE, ) -def _gateway_item(name: str, cluster: str) -> fnv1.Resource: - """An observed InferenceGateway running on the named cluster, with no client - CA published yet so it doesn't change the ServingStack's gateway spec.""" +def _serving_stack( + *, cloud: str, secrets: list[dict], client_cas: list[dict] | None, gpu: dict | None, ready: fnv1.Ready +) -> fnv1.Resource: + """The composed ServingStack, its gateway accepting client_cas and its spec.gpu set to gpu unless they're None.""" + gateway: dict = {"hostname": "gateway-test-cluster-09532.modelplane-system.svc.cluster.local"} + if client_cas is not None: + gateway["clientCAs"] = client_cas + spec: dict = {"cloud": cloud, "gateway": gateway, "stack": "Standard", "secrets": secrets} + if gpu is not None: + spec["gpu"] = gpu return fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceGateway", - "metadata": {"name": name}, - "spec": {"clusterName": cluster}, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": {"name": "test-cluster-serving-stack-fd00b", "namespace": "modelplane-system"}, + "spec": spec, } - ) + ), + ready=ready, + ) + + +def _backend_usage(*, cluster_kind: str) -> fnv1.Resource: + """The composed Usage holding the cluster_kind cluster XR until the ServingStack is gone.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "of": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": cluster_kind, + "resourceSelector": {"matchControllerRef": True}, + }, + "by": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "resourceSelector": {"matchControllerRef": True}, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, ) -def _guard_clusterusage(reason: str) -> fnv1.Resource: - """The reason-only ClusterUsage the guard composes for test-cluster.""" +def _guard_cluster_usage(*, reason: str) -> fnv1.Resource: + """The reason-only ClusterUsage the deletion guard composes for test-cluster.""" return fnv1.Resource( resource=resource.dict_to_struct( { @@ -241,1737 +380,1933 @@ def _guard_clusterusage(reason: str) -> fnv1.Resource: ) -def _guard_case( - base_req: fnv1.RunFunctionRequest, base_want: fnv1.RunFunctionResponse -) -> tuple[fnv1.RunFunctionRequest, fnv1.RunFunctionResponse]: - """Build the guard-and-namespaces case from a base request and response. - - Observes ModelReplicas, ModelRoutes and ModelCaches across several - namespaces, so the function composes a single reason-only ClusterUsage - blocking the InferenceCluster's deletion, whatever their count or namespace, - and mirrors the deduplicated union of their namespaces: team-a (replica), - team-b (replica and route), team-c (route), team-d (cache staging onto this - cluster). A cache staging only onto another cluster (team-e) is filtered out - by its status.clusters[], proving the namespaces track what actually lands - here. The guard's reason names every kind in use. - """ - req = fnv1.RunFunctionRequest() - req.CopyFrom(base_req) - req.required_resources["model-replicas"].items.append(_replica_item("deploy-test-cluster-0", "team-a")) - req.required_resources["model-replicas"].items.append(_replica_item("deploy-test-cluster-0", "team-b")) - req.required_resources["model-routes"].items.append(_route_item("svc-eu", "team-b")) - req.required_resources["model-routes"].items.append(_route_item("svc-eu", "team-c")) - req.required_resources["model-caches"].items.append(_cache_item("qwen", "team-d", ["test-cluster"])) - req.required_resources["model-caches"].items.append(_cache_item("kimi", "team-e", ["other-cluster"])) - - want = fnv1.RunFunctionResponse() - want.CopyFrom(base_want) - want.desired.resources["usage-replicas"].CopyFrom( - _guard_clusterusage("ModelReplicas, ModelRoutes and ModelCaches use this InferenceCluster") - ) - want.desired.resources["namespace-team-a"].CopyFrom(_namespace_object("team-a", "mp-team-a-bd964")) - want.desired.resources["namespace-team-b"].CopyFrom(_namespace_object("team-b", "mp-team-b-6bd62")) - want.desired.resources["namespace-team-c"].CopyFrom(_namespace_object("team-c", "mp-team-c-d79d9")) - want.desired.resources["namespace-team-d"].CopyFrom(_namespace_object("team-d", "mp-team-d-c2383")) - return req, want - - -def _route_guard_case( - base_req: fnv1.RunFunctionRequest, base_want: fnv1.RunFunctionResponse -) -> tuple[fnv1.RunFunctionRequest, fnv1.RunFunctionResponse]: - """A ModelRoute on the cluster composes the guard on its own. - - A route composes its routing Objects through the cluster's - ClusterProviderConfig, so it blocks deletion without any replica there, and - its team's namespace is mirrored. - """ - req = fnv1.RunFunctionRequest() - req.CopyFrom(base_req) - req.required_resources["model-routes"].items.append(_route_item("svc-eu", "team-c")) - - want = fnv1.RunFunctionResponse() - want.CopyFrom(base_want) - want.desired.resources["usage-replicas"].CopyFrom(_guard_clusterusage("ModelRoutes use this InferenceCluster")) - want.desired.resources["namespace-team-c"].CopyFrom(_namespace_object("team-c", "mp-team-c-d79d9")) - return req, want - - -def _cache_guard_case( - base_req: fnv1.RunFunctionRequest, base_want: fnv1.RunFunctionResponse -) -> tuple[fnv1.RunFunctionRequest, fnv1.RunFunctionResponse]: - """A ModelCache staging onto the cluster composes the guard on its own. - - A cache composes its PVC through the cluster's ClusterProviderConfig, so it - blocks deletion without any replica there, and its team's namespace is - mirrored. - """ - req = fnv1.RunFunctionRequest() - req.CopyFrom(base_req) - req.required_resources["model-caches"].items.append(_cache_item("qwen", "team-d", ["test-cluster"])) - - want = fnv1.RunFunctionResponse() - want.CopyFrom(base_want) - want.desired.resources["usage-replicas"].CopyFrom(_guard_clusterusage("ModelCaches use this InferenceCluster")) - want.desired.resources["namespace-team-d"].CopyFrom(_namespace_object("team-d", "mp-team-d-c2383")) - return req, want - - -def _gateway_guard_case( - base_req: fnv1.RunFunctionRequest, base_want: fnv1.RunFunctionResponse -) -> tuple[fnv1.RunFunctionRequest, fnv1.RunFunctionResponse]: - """An InferenceGateway running on the cluster composes the guard on its own. - - A gateway composes its Gateway and routing Objects through the cluster's - ClusterProviderConfig just as a replica does, so it blocks deletion too. It - is cluster scoped, so it mirrors no namespace. - """ - req = fnv1.RunFunctionRequest() - req.CopyFrom(base_req) - req.required_resources["gateways"].items.append(_gateway_item("public", "test-cluster")) - - want = fnv1.RunFunctionResponse() - want.CopyFrom(base_want) - want.desired.resources["usage-replicas"].CopyFrom( - _guard_clusterusage("InferenceGateways use this InferenceCluster") - ) - return req, want - - -def _unused_case( - base_req: fnv1.RunFunctionRequest, base_want: fnv1.RunFunctionResponse -) -> tuple[fnv1.RunFunctionRequest, fnv1.RunFunctionResponse]: - """Every guard requirement resolved to nothing on this cluster: no ClusterUsage. - - This is the teardown transition - the last user is gone, so the function - stops composing the guard and the cluster becomes deletable. A cache staging - onto another cluster and a gateway running on another cluster don't hold it. - base_want must not contain usage-replicas. - """ - req = fnv1.RunFunctionRequest() - req.CopyFrom(base_req) - # Empty-but-present requirements, as Crossplane returns when a selector - # matched nothing. - req.required_resources["model-replicas"].ClearField("items") - req.required_resources["model-routes"].ClearField("items") - req.required_resources["model-caches"].items.append(_cache_item("kimi", "team-e", ["other-cluster"])) - req.required_resources["gateways"].items.append(_gateway_item("elsewhere", "other-cluster")) - return req, base_want - - -def _early_return_guard_case() -> tuple[fnv1.RunFunctionRequest, fnv1.RunFunctionResponse]: - """The guard is composed even when compose() returns early. - - resolve_classes() returns False whenever a referenced InferenceClass isn't - observed yet - a routine transient. The function returns before composing - the cluster, but the guard runs first, so a referencing replica still blocks - deletion. This is the case that regresses if the guard is gated behind class - resolution or cluster source. - """ - xr = v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), - ), - nodePools=[v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4)], - ), - ) - # The class requirement is declared but not fulfilled, so resolve_classes - # gates and compose() returns early. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json"))) - ), - ) - req.required_resources["model-replicas"].items.append(_replica_item("deploy-test-cluster-0", "team-a")) - - want = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - # The guard and namespace are marked ready, so the XR is marked not - # ready while it waits for its classes. - composite=fnv1.Resource(ready=fnv1.READY_FALSE), - resources={ - "usage-replicas": _guard_clusterusage("ModelReplicas use this InferenceCluster"), - "namespace-team-a": _namespace_object("team-a", "mp-team-a-bd964"), - }, +def _namespace_object(*, team: str, name: str) -> fnv1.Resource: + """The composed Object that mirrors team's namespace onto the cluster as the Namespace name.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Namespace", + "metadata": {"name": name, "labels": {"modelplane.ai/namespace": team}}, + }, + }, + }, + } ), - context=structpb.Struct(), - ) - want.requirements.resources["gateways"].CopyFrom(_gateways_selector()) - want.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) - want.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) - want.requirements.resources["model-caches"].CopyFrom(_caches_selector()) - want.requirements.resources["class-gpu-l4"].CopyFrom( - fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4") - ) - want.conditions.append( - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForClasses", - message="Waiting for InferenceClasses: gpu-l4", - ) + ready=fnv1.READY_TRUE, ) - want.results.append(fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for InferenceClasses: gpu-l4")) - return req, want -def _compose_cases() -> list[Case]: # noqa: PLR0915 - """The RunFunction cases, built from shared bases. - - Many table entries, each exercising a distinct compose path across the - GKE, EKS, and Existing sources, push this over the statement limit. - """ - # Shared InferenceClass resource for required_resources. - inference_class_l4 = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-l4"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], - "provisioning": { - "provider": "GKE", - "gke": { - "machineType": "g2-standard-48", - "diskSizeGb": 100, - "accelerator": { - "type": "nvidia-l4", - "count": 1, - }, - }, - }, - }, - } +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - # Shared resource selector for class requirement. - class_selector = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-l4", - ) - # --- Case 1: Existing cluster with secrets composes backend and CPC. --- - req1 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing( - secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), - ), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - ), - ], - ), - ).model_dump(exclude_none=True, mode="json") +# Every want requires what the function reads, before it composes anything: the +# ModelReplicas and ModelRoutes labelled for test-cluster, across all +# namespaces; every ModelCache, since a cache fans out to many clusters and so +# can't be label-selected to one, leaving the function to filter by +# status.clusters[]; every InferenceGateway, since the cluster gateway accepts +# client certificates from each of their CAs, which is how an InferenceGateway +# proves itself; and the InferenceClass behind each node pool. +# +# A cloud cluster's case observes its ManagedResourceActivationPolicy with every +# kind in status.activated, so the function composes the cluster XR rather than +# waiting for activation, unless the case says otherwise. +# +# gateway-test-cluster-09532.modelplane-system.svc.cluster.local is the internal +# name Modelplane derives for test-cluster's gateway, which +# compose-inference-gateway resolves. +COMPOSE_CASES = [ + ComposeCase( + name="ExistingCluster", + reason="An Existing cluster composes a ClusterProviderConfig and a ServingStack from its kubeconfig Secret.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], ), ), - ), - ) - req1.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - "namespace": "modelplane-system", - "gpuPools": [ + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - } + ], + cache=None, + gateway=None, ), - ), - resources={ - "cluster-provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "my-kubeconfig", - "key": "kubeconfig", - }, - }, - }, - } + resources={ + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None ), - ready=fnv1.READY_TRUE, - ), - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Existing", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "my-kubeconfig", - "key": "kubeconfig", - }, - ], - }, - } + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + gpu=None, + ready=fnv1.READY_UNSPECIFIED, ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", + }, ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, ), - ], - context=structpb.Struct(), - ) - want1.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - - # --- Case 1b: Existing cluster with a non-GCP identity threads the - # declared identity type into the CPC and the ServingStack. --- - req1b = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing( - secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), - identitySecretRef=v1alpha1.IdentitySecretRef( - name="nebius-creds", - key="credentials.json", - type="NebiusServiceAccountCredentials", - ), - ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], + ), + ), + ComposeCase( + name="ExistingNonGCPIdentity", + reason="An Existing cluster threads its non-GCP identity's type into the ClusterProviderConfig and the ServingStack's secrets.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing( + secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), + identitySecretRef=v1alpha1.IdentitySecretRef( + name="nebius-creds", + key="credentials.json", + type="NebiusServiceAccountCredentials", ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - ), - ], ), - ).model_dump(exclude_none=True, mode="json") + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], ), ), - ), - ) - req1b.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - # want1b mirrors want1 but with the Nebius identity on the CPC and an - # extra ServingStack identity secret of the same type. - want1b = fnv1.RunFunctionResponse() - want1b.CopyFrom(want1) - cpc1b = want1b.desired.resources["cluster-provider-config-kubernetes"] - cpc1b_dict = resource.struct_to_dict(cpc1b.resource) - cpc1b_dict["spec"]["identity"] = { - "type": "NebiusServiceAccountCredentials", - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "nebius-creds", - "key": "credentials.json", - }, - } - cpc1b.resource.CopyFrom(resource.dict_to_struct(cpc1b_dict)) - backend1b = want1b.desired.resources["serving-stack"] - backend1b_dict = resource.struct_to_dict(backend1b.resource) - backend1b_dict["spec"]["secrets"].append( - { - "type": "NebiusServiceAccountCredentials", - "name": "nebius-creds", - "key": "credentials.json", - } - ) - backend1b.resource.CopyFrom(resource.dict_to_struct(backend1b_dict)) - # want1 gains the replica, route, cache and gateway requirements in place - # from the guard cases below, after this snapshot; add them here so want1b - # matches on its own. - want1b.requirements.resources["gateways"].CopyFrom(_gateways_selector()) - want1b.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) - want1b.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) - want1b.requirements.resources["model-caches"].CopyFrom(_caches_selector()) - - # --- Case 2: GKE cluster first pass - no observed GKE, classes resolved. --- - req2 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="GKE", - gke=v1alpha1.Gke( - region="us-central1", - ), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - zones=["us-central1-a"], - ), - ], + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, ), - ).model_dump(exclude_none=True, mode="json") + ], ), - ), + }, ), - ) - req2.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - } + ], + cache=None, + gateway=None, ), - ), - resources={ - "gke-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": { - "name": "test-cluster", + resources={ + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", + identity={ + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": { "namespace": "modelplane-system", + "name": "nebius-creds", + "key": "credentials.json", }, - "spec": { - "region": "us-central1", - "kubernetesVersion": "1.35", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "machineType": "g2-standard-48", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - "acceleratorCount": 1, - }, - "zones": ["us-central1-a"], - }, - ], + }, + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[ + {"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}, + { + "type": "NebiusServiceAccountCredentials", + "name": "nebius-creds", + "key": "credentials.json", }, - } + ], + client_cas=None, + gpu=None, + ready=fnv1.READY_UNSPECIFIED, ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", + }, ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, ), - ], - context=structpb.Struct(), - ) - want2.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - - # --- Case 3: Existing cluster second pass - backend observed ready. --- - req3 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing( - secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), - ), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - ), - ], + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], + ), + ), + ComposeCase( + name="GKEFirstPass", + reason="With no GKECluster observed yet, a GKE cluster composes only the activation policy and the GKECluster.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="GKE", gke=v1alpha1.Gke(region="us-central1")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], ), - ).model_dump(exclude_none=True, mode="json") + ], ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", + ] + ), + }, ), - resources={ - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": {"name": "test-cluster-serving-stack-fd00b"}, - "status": { - "conditions": [{"type": "Ready", "status": "True"}], - "gateway": {"address": "34.55.100.10"}, + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - } - ), + ), + ], ), }, ), - ) - req3.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], - "gateway": {"address": "34.55.100.10"}, }, - } + ], + cache=None, + gateway=None, ), + resources={ + "activation": _activation_policy( + activate=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "gke-cluster": _gke_cluster(credentials=None, ready=fnv1.READY_UNSPECIFIED), + }, ), - resources={ - "cluster-provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "my-kubeconfig", - "key": "kubeconfig", - }, - }, - }, - } + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), ), - ready=fnv1.READY_TRUE, - ), - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Existing", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "my-kubeconfig", - "key": "kubeconfig", - }, - ], - }, - } + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" ), - ready=fnv1.READY_TRUE, - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="BackendHealthy", - ), - ], - context=structpb.Struct(), - ) - want3.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - - # --- Case 4: EKS cluster first pass - no observed EKS, classes resolved. --- - inference_class_l4_eks = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-l4-eks"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), ], - "provisioning": { - "provider": "EKS", - "eks": { - "instanceType": "g6.xlarge", - "diskSizeGb": 100, - "accelerator": {"type": "nvidia-l4", "count": 1}, - }, - }, - }, - } - class_selector_eks = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-l4-eks", - ) - - req4 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="EKS", - eks=v1alpha1.Eks(region="us-west-2"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4-eks", - nodeCount=2, - maxNodeCount=4, - zones=["us-west-2a", "us-west-2b"], - ), - ], + ), + ), + ComposeCase( + name="GKECredentials", + reason="A GKE cluster's credentials pass through to the GKECluster's spec.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="GKE", + gke=v1alpha1.Gke( + region="us-central1", + credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-gcp-account"), ), - ).model_dump(exclude_none=True, mode="json"), + ), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], + ), + ], ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", + ] + ), + }, ), - ), - ) - req4.required_resources["class-gpu-l4-eks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), - ) - - want4 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - "namespace": "modelplane-system", - "gpuPools": [ + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - }, + ], + cache=None, + gateway=None, ), + resources={ + "activation": _activation_policy( + activate=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "gke-cluster": _gke_cluster( + credentials={"type": "ProviderConfig", "name": "my-gcp-account"}, ready=fnv1.READY_UNSPECIFIED + ), + }, ), - resources={ - "eks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-west-2", - "kubernetesVersion": "1.36", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - }, - "zones": ["us-west-2a", "us-west-2b"], - }, - ], - }, - }, + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], + ), + ), + ComposeCase( + name="ExistingBackendReady", + reason="Once its ServingStack is observed Ready with an address, an Existing cluster relays the address and reports its backend healthy.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], + ), + resources={ + "serving-stack": _observed_serving_stack(gateway={"address": "34.55.100.10"}), + }, ), - ], - context=structpb.Struct(), - ) - want4.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) - - # --- Case 8: EKS first pass with a node pool backed by a Capacity - # Block. The reservation ID flows through to the EKSCluster node pool's - # capacityBlock, which compose-eks-cluster turns into a CAPACITY_BLOCK - # node group. --- - req8 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="EKS", - eks=v1alpha1.Eks(region="us-west-2"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4-eks", - nodeCount=2, - maxNodeCount=4, - zones=["us-west-2a"], - capacityBlock=v1alpha1.CapacityBlock( - capacityReservationId="cr-0123456789abcdef0", - ), - ), - ], + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, ), - ).model_dump(exclude_none=True, mode="json"), + ], ), - ), + }, ), - ) - req8.required_resources["class-gpu-l4-eks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), - ) - - want8 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - }, + ], + cache=None, + gateway={"address": "34.55.100.10"}, ), - ), - resources={ - "eks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-west-2", - "kubernetesVersion": "1.36", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - }, - "zones": ["us-west-2a"], - "capacityBlock": { - "capacityReservationId": "cr-0123456789abcdef0", - }, - }, - ], - }, - }, + resources={ + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + gpu=None, + ready=fnv1.READY_TRUE, + ), + }, ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, ), - ], - context=structpb.Struct(), - ) - want8.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) - - # --- Case 9: EKS first pass with a node pool that opts into the EFA - # fabric. fabric.type flows through to the EKSCluster node pool, which - # compose-eks-cluster turns into EFA launch-template interfaces. --- - req9 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="EKS", - eks=v1alpha1.Eks(region="us-west-2"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4-eks", - nodeCount=2, - maxNodeCount=4, - zones=["us-west-2a"], - fabric=v1alpha1.Fabric(type="EFA"), - ), - ], + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_TRUE, reason="BackendHealthy"), + ], + ), + ), + ComposeCase( + name="EKSFirstPass", + reason="With no EKSCluster observed yet, an EKS cluster composes only the activation policy and the EKSCluster.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="EKS", eks=v1alpha1.Eks(region="us-west-2")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a", "us-west-2b"], ), - ).model_dump(exclude_none=True, mode="json"), + ], ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ] + ), + }, ), - ), - ) - req9.required_resources["class-gpu-l4-eks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), - ) - - want9 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + required_resources={ + "class-gpu-l4-eks": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4-eks", + count=1, + memory="24Gi", + provisioning={ + "provider": "EKS", + "eks": { + "instanceType": "g6.xlarge", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - "namespace": "modelplane-system", - "gpuPools": [ + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - }, + ], + cache=None, + gateway=None, ), - ), - resources={ - "eks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-west-2", - "kubernetesVersion": "1.36", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - }, - "zones": ["us-west-2a"], - "fabric": "EFA", - }, - ], - }, - }, + resources={ + "activation": _activation_policy( + activate=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "eks-cluster": _eks_cluster( + zones=["us-west-2a", "us-west-2b"], + capacity_block=None, + fabric=None, + ready=fnv1.READY_UNSPECIFIED, ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want9.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) - - # --- Case 5: EKS cluster not yet ready (no kubeconfig observed) but a - # ClusterProviderConfig already exists from a prior reconcile. The CPC - # is built only from the kubeconfig, so without one it's simply omitted - # from desired state this reconcile (and recreated once the kubeconfig - # is observed again) - it is never emitted with an empty secretRef. - observed_cpc = { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, - }, - } - req5 = fnv1.RunFunctionRequest() - req5.CopyFrom(req4) - req5.observed.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource(resource=resource.dict_to_struct(observed_cpc)), - ) - - # Desired state is identical to case 4: no ClusterProviderConfig. - want5 = fnv1.RunFunctionResponse() - want5.CopyFrom(want4) - - # --- Case 6: GKE cluster ready - composes CPC, backend, usage, and the - # VPC-pinned modelplane-rwx Filestore StorageClass on the workload - # cluster (default cache storage class). --- - observed_gke_ready = { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "region": "us-central1", - "nodePools": [{"name": "system", "role": "System", "machineType": "e2-standard-4"}], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2026-06-08T00:00:00Z", }, - ], - # The backing GKECluster reports its effective RWX StorageClass; - # the InferenceCluster relays it up to its own status.cache. - "cache": {"storageClassName": "modelplane-rwx"}, - "secrets": [ - {"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}, - { - "type": "GoogleApplicationCredentials", - "name": "test-cluster-sa-key-fghij", - "key": "credentials.json", + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4-eks": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4-eks" + ), }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), ], - }, - } - req6 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="GKE", - gke=v1alpha1.Gke( - region="us-central1", - ), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - zones=["us-central1-a"], - ), - ], + ), + ), + # The ClusterProviderConfig is built only from the kubeconfig, so it's + # recreated once the kubeconfig is observed again. It is never emitted with + # an empty secretRef. + ComposeCase( + name="EKSObservedCPC", + reason="With a ClusterProviderConfig observed but no EKSCluster yet, an EKS cluster leaves the ClusterProviderConfig out of desired state.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="EKS", eks=v1alpha1.Eks(region="us-west-2")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a", "us-west-2b"], ), - ).model_dump(exclude_none=True, mode="json") + ], ), + resources={ + "cluster-provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + }, + }, + } + ), + ), + "activation": _observed_activation_policy( + activated=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ] + ), + }, ), - resources={ - "gke-cluster": fnv1.Resource(resource=resource.dict_to_struct(observed_gke_ready)), + required_resources={ + "class-gpu-l4-eks": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4-eks", + count=1, + memory="24Gi", + provisioning={ + "provider": "EKS", + "eks": { + "instanceType": "g6.xlarge", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + ), + ], + ), }, ), - ) - req6.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - want6 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - } - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], - # Relayed from the backing GKECluster's status.cache. - "cache": {"storageClassName": "modelplane-rwx"}, }, - } + ], + cache=None, + gateway=None, ), + resources={ + "activation": _activation_policy( + activate=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "eks-cluster": _eks_cluster( + zones=["us-west-2a", "us-west-2b"], + capacity_block=None, + fabric=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, ), - resources={ - "gke-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-central1", - "kubernetesVersion": "1.35", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "machineType": "g2-standard-48", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - "acceleratorCount": 1, - }, - "zones": ["us-central1-a"], - }, - ], - }, - } + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), ), - ready=fnv1.READY_TRUE, - ), - "cluster-provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, - "identity": { - "type": "GoogleApplicationCredentials", - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-sa-key-fghij", - "key": "credentials.json", - }, - }, - }, - } + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), ), - ready=fnv1.READY_TRUE, - ), - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "GKE", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - { - "type": "GoogleApplicationCredentials", - "name": "test-cluster-sa-key-fghij", - "key": "credentials.json", - }, - ], - }, - } + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4-eks": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4-eks" ), - ), - "usage-gke-by-backend": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], + ), + ), + # The GKE kubeconfig has no embedded credentials, hence the service account + # identity on the ClusterProviderConfig. + ComposeCase( + name="GKEReady", + reason="Once its GKECluster is Ready, a GKE cluster composes the ClusterProviderConfig, ServingStack and Usage, and relays its RWX StorageClass.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="GKE", gke=v1alpha1.Gke(region="us-central1")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], + ), + ], + ), + resources={ + "gke-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "us-central1", + "nodePools": [{"name": "system", "role": "System", "machineType": "e2-standard-4"}], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + }, + ], + "cache": {"storageClassName": "modelplane-rwx"}, + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + { + "type": "GoogleApplicationCredentials", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", + }, + ], + }, + } + ), + ), + "activation": _observed_activation_policy( + activated=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", + ] ), - ready=fnv1.READY_TRUE, + }, + ), + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + ), + ], ), }, ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="GKE cluster ready, composing backend", - ), - ], - context=structpb.Struct(), - ) - want6.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - - # --- Case 7: EKS cluster ready - kubeconfig observed on the EKSCluster - # status. The function wires the ClusterProviderConfig, composes the - # ServingStack backend, and emits the Usage that blocks EKSCluster - # deletion until the ServingStack is gone. --- - req7 = fnv1.RunFunctionRequest() - req7.CopyFrom(req4) - req7.observed.resources["eks-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "region": "us-west-2", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - }, - ], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + cache={"storageClassName": "modelplane-rwx"}, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", ], - # The backing EKSCluster reports its effective RWX - # StorageClass; the InferenceCluster relays it up. - "cache": {"storageClassName": "modelplane-rwx-efs"}, - }, - } - ), - ), - ) - - want7 = fnv1.RunFunctionResponse() - want7.CopyFrom(want4) - # Mark the EKSCluster ready and relay its status.cache up to status.cache. - _eks_ready_extras(want7, "modelplane-rwx-efs") - want7.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { + ready=fnv1.READY_TRUE, + ), + "gke-cluster": _gke_cluster(credentials=None, ready=fnv1.READY_TRUE), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="test-cluster-kubeconfig-abcde", + identity={ + "type": "GoogleApplicationCredentials", "source": "Secret", "secretRef": { "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", }, }, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - ) - want7.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "EKS", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ + ), + "serving-stack": _serving_stack( + cloud="GKE", + secrets=[ + {"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}, { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + "type": "GoogleApplicationCredentials", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", }, ], - }, - } + client_cas=None, + gpu=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-gke-by-backend": _backend_usage(cluster_kind="GKECluster"), + }, ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="GKE cluster ready, composing backend")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], ), - ) - want7.desired.resources["usage-eks-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } + ), + ComposeCase( + name="EKSReady", + reason="Once its EKSCluster is Ready, an EKS cluster composes the ClusterProviderConfig, ServingStack and Usage, and relays its status.cache.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="EKS", eks=v1alpha1.Eks(region="us-west-2")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a", "us-west-2b"], + ), + ], + ), + resources={ + "eks-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "EKSCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "us-west-2", + "nodePools": [ + {"name": "l4-pool", "role": "GPU", "instanceType": "g6.xlarge", "nodeCount": 2}, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + "cache": {"storageClassName": "modelplane-rwx-efs"}, + }, + } + ), + ), + "activation": _observed_activation_policy( + activated=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ] + ), + }, ), - ready=fnv1.READY_TRUE, + required_resources={ + "class-gpu-l4-eks": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4-eks", + count=1, + memory="24Gi", + provisioning={ + "provider": "EKS", + "eks": { + "instanceType": "g6.xlarge", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + ), + ], + ), + }, ), - ) - del want7.conditions[:] - want7.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want7.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="EKS cluster ready, composing backend", - ) - ) - - # --- Case 10: Nebius first pass composes the NebiusCluster XR only. - # The pool's InfiniBand fabric flows through to the NebiusCluster - # pool's fabric, and minNodeCount stays unset so the pool's - # autoscaling floor defaults to its node count downstream. --- - inference_class_h100_nebius = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-h100-nebius"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + cache={"storageClassName": "modelplane-rwx-efs"}, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "eks-cluster": _eks_cluster( + zones=["us-west-2a", "us-west-2b"], capacity_block=None, fabric=None, ready=fnv1.READY_TRUE + ), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="test-cluster-kubeconfig-abcde", identity=None + ), + "serving-stack": _serving_stack( + cloud="EKS", + secrets=[{"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}], + client_cas=None, + gpu=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-eks-by-backend": _backend_usage(cluster_kind="EKSCluster"), }, - ], - "provisioning": { - "provider": "Nebius", - "nebius": { - "platform": "gpu-h100-sxm", - "preset": "8gpu-128vcpu-1600gb", - "diskSizeGb": 200, - "accelerator": {"type": "nvidia-h100", "count": 8}, + ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="EKS cluster ready, composing backend")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4-eks": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4-eks" + ), }, - }, - }, - } - class_selector_nebius = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-h100-nebius", - ) - - req10 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Nebius", - nebius=v1alpha1.Nebius(), + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], + ), + ), + # compose-eks-cluster turns the pool's capacityBlock into a CAPACITY_BLOCK + # node group. + ComposeCase( + name="EKSCapacityBlock", + reason="A node pool backed by a Capacity Block passes its reservation ID through to the EKSCluster pool's capacityBlock.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="EKS", eks=v1alpha1.Eks(region="us-west-2")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a"], + capacityBlock=v1alpha1.CapacityBlock( + capacityReservationId="cr-0123456789abcdef0", ), - nodePools=[ - v1alpha1.NodePool( - name="h100-pool", - className="gpu-h100-nebius", - nodeCount=2, - maxNodeCount=4, - fabric=v1alpha1.Fabric( - type="InfiniBand", - infiniband=v1alpha1.Infiniband(fabric="fabric-2"), - ), - ), - ], ), - ).model_dump(exclude_none=True, mode="json"), + ], ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ] + ), + }, ), - ), - ) - req10.required_resources["class-gpu-h100-nebius"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_nebius)), - ) - - want10 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + required_resources={ + "class-gpu-l4-eks": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4-eks", + count=1, + memory="24Gi", + provisioning={ + "provider": "EKS", + "eks": { + "instanceType": "g6.xlarge", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - "namespace": "modelplane-system", - "gpuPools": [ + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "h100-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - }, + ], + cache=None, + gateway=None, ), + resources={ + "activation": _activation_policy( + activate=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "eks-cluster": _eks_cluster( + zones=["us-west-2a"], + capacity_block={"capacityReservationId": "cr-0123456789abcdef0"}, + fabric=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, ), - resources={ - "nebius-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "NebiusCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "kubernetesVersion": "1.34", - "nodePools": [ - { - "name": "h100-pool", - "role": "GPU", - "platform": "gpu-h100-sxm", - "preset": "8gpu-128vcpu-1600gb", - "diskSizeGb": 200, - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-h100", - "driversPreset": "cuda13.0", - }, - "fabric": { - "type": "InfiniBand", - "infiniband": {"fabric": "fabric-2"}, - }, - }, - ], - }, - }, + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4-eks": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4-eks" ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], + ), + ), + # compose-eks-cluster turns the pool's fabric into EFA launch-template + # interfaces. + ComposeCase( + name="EKSFabricEFA", + reason="A node pool that opts into the EFA fabric passes fabric.type through to the EKSCluster pool.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="EKS", eks=v1alpha1.Eks(region="us-west-2")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a"], + fabric=v1alpha1.Fabric(type="EFA"), + ), + ], + ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ] + ), + }, + ), + required_resources={ + "class-gpu-l4-eks": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4-eks", + count=1, + memory="24Gi", + provisioning={ + "provider": "EKS", + "eks": { + "instanceType": "g6.xlarge", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + ), + ], ), }, ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "eks-cluster": _eks_cluster( + zones=["us-west-2a"], capacity_block=None, fabric="EFA", ready=fnv1.READY_UNSPECIFIED + ), + }, ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4-eks": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4-eks" + ), + }, ), - ], - context=structpb.Struct(), - ) - want10.requirements.resources["class-gpu-h100-nebius"].CopyFrom(class_selector_nebius) - - # --- Case 11: Nebius cluster ready - kubeconfig and service account - # credentials observed on the NebiusCluster status. The function wires - # the ClusterProviderConfig with the Nebius identity (the mk8s - # kubeconfig has no embedded credentials), composes the ServingStack - # backend with both secrets, and emits the Usage that blocks - # NebiusCluster deletion until the ServingStack is gone. --- - req11 = fnv1.RunFunctionRequest() - req11.CopyFrom(req10) - req11.observed.resources["nebius-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "NebiusCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "nodePools": [ - { - "name": "h100-pool", - "role": "GPU", - "platform": "gpu-h100-sxm", - "preset": "8gpu-128vcpu-1600gb", - "nodeCount": 2, - }, - ], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], + ), + ), + # minNodeCount stays unset so the pool's autoscaling floor defaults to its + # node count downstream. + ComposeCase( + name="NebiusFirstPass", + reason="With no NebiusCluster observed yet, a Nebius cluster composes only the activation policy and a NebiusCluster carrying its pool's InfiniBand fabric.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="Nebius", nebius=v1alpha1.Nebius()), + node_pools=[ + v1alpha1.NodePool( + name="h100-pool", + className="gpu-h100-nebius", + nodeCount=2, + maxNodeCount=4, + fabric=v1alpha1.Fabric( + type="InfiniBand", + infiniband=v1alpha1.Infiniband(fabric="fabric-2"), + ), + ), + ], + ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "filesystems.compute.nebius.m.upbound.io", + "gpuclusters.compute.nebius.m.upbound.io", + "clusters.mk8s.nebius.m.upbound.io", + "nodegroups.mk8s.nebius.m.upbound.io", + "networks.vpc.nebius.m.upbound.io", + "subnets.vpc.nebius.m.upbound.io", + ] + ), + }, + ), + required_resources={ + "class-gpu-h100-nebius": fnv1.Resources( + items=[ + _inference_class( + name="gpu-h100-nebius", + count=8, + memory="81559Mi", + provisioning={ + "provider": "Nebius", + "nebius": { + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "diskSizeGb": 200, + "accelerator": {"type": "nvidia-h100", "count": 8}, + }, }, + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "h100-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "filesystems.compute.nebius.m.upbound.io", + "gpuclusters.compute.nebius.m.upbound.io", + "clusters.mk8s.nebius.m.upbound.io", + "nodegroups.mk8s.nebius.m.upbound.io", + "networks.vpc.nebius.m.upbound.io", + "subnets.vpc.nebius.m.upbound.io", ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - # The credential entry carries a namespace: it - # is the Nebius ClusterProviderConfig's Secret, - # which lives outside modelplane-system. + ready=fnv1.READY_TRUE, + ), + "nebius-cluster": fnv1.Resource( + resource=resource.dict_to_struct( { - "type": "NebiusServiceAccountCredentials", - "name": "nebius-credentials", - "key": "credentials.json", - "namespace": "crossplane-system", - }, - ], - }, - } + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "NebiusCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "kubernetesVersion": "1.34", + "nodePools": [ + { + "name": "h100-pool", + "role": "GPU", + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "diskSizeGb": 200, + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": {"acceleratorType": "nvidia-h100", "driversPreset": "cuda13.0"}, + "fabric": {"type": "InfiniBand", "infiniband": {"fabric": "fabric-2"}}, + }, + ], + }, + } + ), + ), + }, ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-h100-nebius": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-h100-nebius" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], ), - ) - - want11 = fnv1.RunFunctionResponse() - want11.CopyFrom(want10) - want11.desired.resources["nebius-cluster"].ready = fnv1.READY_TRUE - want11.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + ), + # The mk8s kubeconfig has no embedded credentials, hence the identity. The + # credentials Secret carries a namespace: it is the Nebius + # ClusterProviderConfig's Secret, which lives outside modelplane-system. + ComposeCase( + name="NebiusReady", + reason="Once its NebiusCluster is Ready, a Nebius cluster composes a ClusterProviderConfig with the Nebius identity, the ServingStack and the Usage.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="Nebius", nebius=v1alpha1.Nebius()), + node_pools=[ + v1alpha1.NodePool( + name="h100-pool", + className="gpu-h100-nebius", + nodeCount=2, + maxNodeCount=4, + fabric=v1alpha1.Fabric( + type="InfiniBand", + infiniband=v1alpha1.Infiniband(fabric="fabric-2"), + ), + ), + ], + ), + resources={ + "nebius-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "NebiusCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "nodePools": [ + { + "name": "h100-pool", + "role": "GPU", + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "nodeCount": 2, + }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + { + "type": "NebiusServiceAccountCredentials", + "name": "nebius-credentials", + "key": "credentials.json", + "namespace": "crossplane-system", + }, + ], + }, + } + ), + ), + "activation": _observed_activation_policy( + activated=[ + "filesystems.compute.nebius.m.upbound.io", + "gpuclusters.compute.nebius.m.upbound.io", + "clusters.mk8s.nebius.m.upbound.io", + "nodegroups.mk8s.nebius.m.upbound.io", + "networks.vpc.nebius.m.upbound.io", + "subnets.vpc.nebius.m.upbound.io", + ] + ), + }, + ), + required_resources={ + "class-gpu-h100-nebius": fnv1.Resources( + items=[ + _inference_class( + name="gpu-h100-nebius", + count=8, + memory="81559Mi", + provisioning={ + "provider": "Nebius", + "nebius": { + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "diskSizeGb": 200, + "accelerator": {"type": "nvidia-h100", "count": 8}, + }, }, + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "h100-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], }, - "identity": { + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "filesystems.compute.nebius.m.upbound.io", + "gpuclusters.compute.nebius.m.upbound.io", + "clusters.mk8s.nebius.m.upbound.io", + "nodegroups.mk8s.nebius.m.upbound.io", + "networks.vpc.nebius.m.upbound.io", + "subnets.vpc.nebius.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "nebius-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "NebiusCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "kubernetesVersion": "1.34", + "nodePools": [ + { + "name": "h100-pool", + "role": "GPU", + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "diskSizeGb": 200, + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": {"acceleratorType": "nvidia-h100", "driversPreset": "cuda13.0"}, + "fabric": {"type": "InfiniBand", "infiniband": {"fabric": "fabric-2"}}, + }, + ], + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="test-cluster-kubeconfig-abcde", + identity={ "type": "NebiusServiceAccountCredentials", "source": "Secret", "secretRef": { @@ -1980,32 +2315,11 @@ def _compose_cases() -> list[Case]: # noqa: PLR0915 "key": "credentials.json", }, }, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - ) - want11.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Nebius", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, + ), + "serving-stack": _serving_stack( + cloud="Nebius", + secrets=[ + {"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}, { "type": "NebiusServiceAccountCredentials", "name": "nebius-credentials", @@ -2013,1758 +2327,2718 @@ def _compose_cases() -> list[Case]: # noqa: PLR0915 "namespace": "crossplane-system", }, ], - }, - } + client_cas=None, + gpu=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-nebius-by-backend": _backend_usage(cluster_kind="NebiusCluster"), + }, + ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Nebius cluster ready, composing backend")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-h100-nebius": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-h100-nebius" + ), + }, ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], ), - ) - want11.desired.resources["usage-nebius-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "NebiusCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } + ), + # The fabric is the plain string - Azure has no user-selectable fabric ID. + # The pool sets minNodeCount to 1, as an AKS GPU pool must, because the AKS + # autoscaler can't scale a DRA pool up from zero nodes. + ComposeCase( + name="AKSFirstPass", + reason="With no AKSCluster observed yet, an AKS cluster composes only the activation policy and an AKSCluster carrying its pool's fabric and minNodeCount.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="AKS", aks=v1alpha1.Aks(location="westeurope")), + node_pools=[ + v1alpha1.NodePool( + name="h100pool", + className="gpu-h100-aks", + nodeCount=2, + minNodeCount=1, + maxNodeCount=4, + fabric=v1alpha1.Fabric(type="InfiniBand"), + ), + ], + ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "kubernetesclusters.containerservice.azure.m.upbound.io", + "kubernetesclusternodepools.containerservice.azure.m.upbound.io", + "subnets.network.azure.m.upbound.io", + "virtualnetworks.network.azure.m.upbound.io", + "resourcegroups.azure.m.upbound.io", + ] + ), + }, ), - ready=fnv1.READY_TRUE, + required_resources={ + "class-gpu-h100-aks": fnv1.Resources( + items=[ + _inference_class( + name="gpu-h100-aks", + count=8, + memory="81559Mi", + provisioning={ + "provider": "AKS", + "aks": { + "vmSize": "Standard_ND96isr_H100_v5", + "diskSizeGb": 200, + "accelerator": {"type": "nvidia-h100", "count": 8}, + }, + }, + ), + ], + ), + }, ), - ) - del want11.conditions[:] - want11.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want11.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Nebius cluster ready, composing backend", - ) - ) - - # --- Case 12: AKS first pass composes the AKSCluster XR only. The - # pool's InfiniBand fabric flows through to the AKSCluster pool as the - # plain fabric string - Azure has no user-selectable fabric ID. --- - inference_class_h100_aks = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-h100-aks"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "h100pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "kubernetesclusters.containerservice.azure.m.upbound.io", + "kubernetesclusternodepools.containerservice.azure.m.upbound.io", + "subnets.network.azure.m.upbound.io", + "virtualnetworks.network.azure.m.upbound.io", + "resourcegroups.azure.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "aks-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "AKSCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "location": "westeurope", + "kubernetesVersion": "1.34", + "nodePools": [ + { + "name": "h100pool", + "role": "GPU", + "vmSize": "Standard_ND96isr_H100_v5", + "diskSizeGb": 200, + "nodeCount": 2, + "minNodeCount": 1, + "maxNodeCount": 4, + "gpu": {"acceleratorType": "nvidia-h100"}, + "fabric": "InfiniBand", + }, + ], + }, + } + ), + ), }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-h100-aks": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-h100-aks" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), ], - "provisioning": { - "provider": "AKS", - "aks": { - "vmSize": "Standard_ND96isr_H100_v5", - "diskSizeGb": 200, - "accelerator": {"type": "nvidia-h100", "count": 8}, + ), + ), + # The kubeconfig embeds a client certificate, so the ClusterProviderConfig + # carries no identity (unlike GKE and Nebius). + ComposeCase( + name="AKSReady", + reason="Once its AKSCluster is Ready, an AKS cluster composes a ClusterProviderConfig without identity, the ServingStack and the Usage, and relays its status.cache.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="AKS", aks=v1alpha1.Aks(location="westeurope")), + node_pools=[ + v1alpha1.NodePool( + name="h100pool", + className="gpu-h100-aks", + nodeCount=2, + minNodeCount=1, + maxNodeCount=4, + fabric=v1alpha1.Fabric(type="InfiniBand"), + ), + ], + ), + resources={ + "aks-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "AKSCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "location": "westeurope", + "nodePools": [ + { + "name": "h100pool", + "role": "GPU", + "vmSize": "Standard_ND96isr_H100_v5", + "nodeCount": 2, + }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + "cache": {"storageClassName": "modelplane-rwx-fs"}, + }, + } + ), + ), + "activation": _observed_activation_policy( + activated=[ + "kubernetesclusters.containerservice.azure.m.upbound.io", + "kubernetesclusternodepools.containerservice.azure.m.upbound.io", + "subnets.network.azure.m.upbound.io", + "virtualnetworks.network.azure.m.upbound.io", + "resourcegroups.azure.m.upbound.io", + ] + ), }, + ), + required_resources={ + "class-gpu-h100-aks": fnv1.Resources( + items=[ + _inference_class( + name="gpu-h100-aks", + count=8, + memory="81559Mi", + provisioning={ + "provider": "AKS", + "aks": { + "vmSize": "Standard_ND96isr_H100_v5", + "diskSizeGb": 200, + "accelerator": {"type": "nvidia-h100", "count": 8}, + }, + }, + ), + ], + ), }, - }, - } - class_selector_aks = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-h100-aks", - ) - - req12 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="AKS", - aks=v1alpha1.Aks(location="westeurope"), - ), - nodePools=[ - v1alpha1.NodePool( - name="h100pool", - className="gpu-h100-aks", - nodeCount=2, - # AKS GPU pools keep minNodeCount at - # 1: the AKS autoscaler can't scale a - # DRA pool up from zero nodes. - minNodeCount=1, - maxNodeCount=4, - fabric=v1alpha1.Fabric(type="InfiniBand"), - ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "h100pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, ], - ), - ).model_dump(exclude_none=True, mode="json"), + }, + ], + cache={"storageClassName": "modelplane-rwx-fs"}, + gateway=None, ), + resources={ + "activation": _activation_policy( + activate=[ + "kubernetesclusters.containerservice.azure.m.upbound.io", + "kubernetesclusternodepools.containerservice.azure.m.upbound.io", + "subnets.network.azure.m.upbound.io", + "virtualnetworks.network.azure.m.upbound.io", + "resourcegroups.azure.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "aks-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "AKSCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "location": "westeurope", + "kubernetesVersion": "1.34", + "nodePools": [ + { + "name": "h100pool", + "role": "GPU", + "vmSize": "Standard_ND96isr_H100_v5", + "diskSizeGb": 200, + "nodeCount": 2, + "minNodeCount": 1, + "maxNodeCount": 4, + "gpu": {"acceleratorType": "nvidia-h100"}, + "fabric": "InfiniBand", + }, + ], + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="test-cluster-kubeconfig-abcde", identity=None + ), + "serving-stack": _serving_stack( + cloud="AKS", + secrets=[{"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}], + client_cas=None, + gpu=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-aks-by-backend": _backend_usage(cluster_kind="AKSCluster"), + }, ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="AKS cluster ready, composing backend")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-h100-aks": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-h100-aks" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], ), - ) - req12.required_resources["class-gpu-h100-aks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_aks)), - ) - - want12 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + ), + # A kind is missing while, for example, a provider is still installing. + # Leaving the policy not ready keeps the composite from reporting ready. + # Observing the policy with one kind missing, here + # nodepools.container.gcp.m.upbound.io, exercises the all-kinds check rather + # than the policy-absent branch. + ComposeCase( + name="KindNotActivated", + reason="While its activation policy is missing one kind and no GKECluster is observed, a GKE cluster composes only the policy, not ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="GKE", gke=v1alpha1.Gke(region="us-central1")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], + ), + ], + ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + ] + ), + }, + ), + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - "namespace": "modelplane-system", - "gpuPools": [ + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "h100pool", - "nodes": 4, - "devices": [ + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", + ], + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], + ), + ), + # Once the cluster is observed, the function keeps composing it even when + # its activation policy isn't observed, so an activation blip never drops a + # provisioned cluster from desired state. + ComposeCase( + name="ActivationBlip", + reason="With its GKECluster observed but no activation policy, a GKE cluster keeps composing the GKECluster and everything built on it.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="GKE", gke=v1alpha1.Gke(region="us-central1")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], + ), + ], + ), + resources={ + "gke-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "us-central1", + "nodePools": [{"name": "system", "role": "System", "machineType": "e2-standard-4"}], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + }, + ], + "cache": {"storageClassName": "modelplane-rwx"}, + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, + "type": "GoogleApplicationCredentials", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", }, ], }, - ], - }, - }, - ), + } + ), + ), + }, ), - resources={ - "aks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "AKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "location": "westeurope", - "kubernetesVersion": "1.34", - "nodePools": [ - { - "name": "h100pool", - "role": "GPU", - "vmSize": "Standard_ND96isr_H100_v5", - "diskSizeGb": 200, - "nodeCount": 2, - "minNodeCount": 1, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-h100", - }, - "fabric": "InfiniBand", - }, - ], + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - }, - ), + ), + ], ), }, ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want12.requirements.resources["class-gpu-h100-aks"].CopyFrom(class_selector_aks) - - # --- Case 13: AKS cluster ready - kubeconfig observed on the - # AKSCluster status. The kubeconfig embeds a client certificate, so - # the ClusterProviderConfig carries no identity (unlike GKE/Nebius). - # The function composes the ServingStack backend and the Usage that - # blocks AKSCluster deletion until the ServingStack is gone, and - # relays the AKSCluster's status.cache up to status.cache. --- - req13 = fnv1.RunFunctionRequest() - req13.CopyFrom(req12) - req13.observed.resources["aks-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "AKSCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "location": "westeurope", - "nodePools": [ - { - "name": "h100pool", - "role": "GPU", - "vmSize": "Standard_ND96isr_H100_v5", - "nodeCount": 2, - }, - ], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + cache={"storageClassName": "modelplane-rwx"}, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", ], - # The backing AKSCluster reports its effective RWX - # StorageClass; the InferenceCluster relays it up. - "cache": {"storageClassName": "modelplane-rwx-fs"}, - }, - } - ), - ), - ) - - want13 = fnv1.RunFunctionResponse() - want13.CopyFrom(want12) - want13.desired.resources["aks-cluster"].ready = fnv1.READY_TRUE - status13 = want13.desired.composite.resource.fields["status"].struct_value - status13.fields["cache"].struct_value.fields["storageClassName"].string_value = "modelplane-rwx-fs" - want13.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { + ready=fnv1.READY_TRUE, + ), + "gke-cluster": _gke_cluster(credentials=None, ready=fnv1.READY_TRUE), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="test-cluster-kubeconfig-abcde", + identity={ + "type": "GoogleApplicationCredentials", "source": "Secret", "secretRef": { "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", }, }, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - ) - want13.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "AKS", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ + ), + "serving-stack": _serving_stack( + cloud="GKE", + secrets=[ + {"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}, { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + "type": "GoogleApplicationCredentials", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", }, ], - }, - } - ), - ), - ) - want13.desired.resources["usage-aks-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "AKSCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } + client_cas=None, + gpu=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-gke-by-backend": _backend_usage(cluster_kind="GKECluster"), + }, ), - ready=fnv1.READY_TRUE, - ), - ) - del want13.conditions[:] - want13.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want13.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="AKS cluster ready, composing backend", - ) - ) - - # --- Case 14: Vultr first pass composes the VultrCluster XR only. - # minNodeCount stays unset so the pool's autoscaling floor defaults - # to its node count downstream. --- - inference_class_l40s_vultr = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-l40s-vultr"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="GKE cluster ready, composing backend")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), ], - "provisioning": { - "provider": "Vultr", - "vultr": { - "plan": "vcg-l40s-16c-180g-48vram", - "accelerator": {"type": "nvidia-l40s", "count": 1}, + ), + ), + # minNodeCount stays unset so the pool's autoscaling floor defaults to its + # node count downstream. + ComposeCase( + name="VultrFirstPass", + reason="With no VultrCluster observed yet, a Vultr cluster composes only the activation policy and the VultrCluster.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="Vultr", vultr=v1alpha1.Vultr(region="ewr")), + node_pools=[ + v1alpha1.NodePool(name="l40s-pool", className="gpu-l40s-vultr", nodeCount=2, maxNodeCount=4), + ], + ), + resources={ + "activation": _observed_activation_policy( + activated=["kubernetes.vke.vultr.m.upbound.io", "kubernetesnodepools.vke.vultr.m.upbound.io"] + ), }, - }, - }, - } - class_selector_vultr = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-l40s-vultr", - ) - - req14 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Vultr", - vultr=v1alpha1.Vultr(region="ewr"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l40s-pool", - className="gpu-l40s-vultr", - nodeCount=2, - maxNodeCount=4, - ), - ], + ), + required_resources={ + "class-gpu-l40s-vultr": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l40s-vultr", + count=1, + memory="46068Mi", + provisioning={ + "provider": "Vultr", + "vultr": { + "plan": "vcg-l40s-16c-180g-48vram", + "accelerator": {"type": "nvidia-l40s", "count": 1}, + }, + }, ), - ).model_dump(exclude_none=True, mode="json"), + ], ), - ), + }, ), - ) - req14.required_resources["class-gpu-l40s-vultr"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l40s_vultr)), - ) - - want14 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l40s-pool", + "nodes": 4, + "devices": [ { - "name": "l40s-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, }, ], }, - }, + ], + cache=None, + gateway=None, ), - ), - resources={ - "vultr-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "ewr", - "kubernetesVersion": "v1.36.2+1", - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "plan": "vcg-l40s-16c-180g-48vram", - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-l40s", - }, - }, - ], - }, - }, + resources={ + "activation": _activation_policy( + activate=["kubernetes.vke.vultr.m.upbound.io", "kubernetesnodepools.vke.vultr.m.upbound.io"], + ready=fnv1.READY_TRUE, ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", + "vultr-cluster": _vultr_cluster(credentials=None, ready=fnv1.READY_UNSPECIFIED), + }, ), - ], - context=structpb.Struct(), - ) - want14.requirements.resources["class-gpu-l40s-vultr"].CopyFrom(class_selector_vultr) - - # --- Case 14b: Vultr credentials pass through to the VultrCluster - # spec, mirroring the GKE/EKS/AKS passthrough. --- - req_creds_vultr = fnv1.RunFunctionRequest() - req_creds_vultr.CopyFrom(req14) - req_creds_vultr.observed.composite.CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Vultr", - vultr=v1alpha1.Vultr( - region="ewr", - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="my-vultr-account", - ), - ), - ), - nodePools=[ - v1alpha1.NodePool( - name="l40s-pool", - className="gpu-l40s-vultr", - nodeCount=2, - maxNodeCount=4, - ), - ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), ), - ).model_dump(exclude_none=True, mode="json"), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l40s-vultr": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l40s-vultr" + ), + }, ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], ), - ) - - want_creds_vultr = fnv1.RunFunctionResponse() - want_creds_vultr.CopyFrom(want14) - want_creds_vultr.desired.resources["vultr-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "ewr", - "kubernetesVersion": "v1.36.2+1", - "credentials": { - "type": "ProviderConfig", - "name": "my-vultr-account", - }, - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "plan": "vcg-l40s-16c-180g-48vram", - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-l40s", - }, - }, - ], - }, + ), + ComposeCase( + name="VultrCredentials", + reason="A Vultr cluster's credentials pass through to the VultrCluster's spec.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Vultr", + vultr=v1alpha1.Vultr( + region="ewr", + credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-vultr-account"), + ), + ), + node_pools=[ + v1alpha1.NodePool(name="l40s-pool", className="gpu-l40s-vultr", nodeCount=2, maxNodeCount=4), + ], + ), + resources={ + "activation": _observed_activation_policy( + activated=["kubernetes.vke.vultr.m.upbound.io", "kubernetesnodepools.vke.vultr.m.upbound.io"] + ), }, ), - ), - ) - - # --- Case 15: Vultr cluster ready - kubeconfig observed on the - # VultrCluster status. The VKE kubeconfig embeds static client - # certificates, so the ClusterProviderConfig carries no identity - # (unlike Nebius). The function composes the ServingStack backend - # with the kubeconfig and emits the Usage that blocks VultrCluster - # deletion until the ServingStack is gone. VultrCluster reports no - # cache StorageClass, so status.cache stays unset. --- - req15 = fnv1.RunFunctionRequest() - req15.CopyFrom(req14) - req15.observed.resources["vultr-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "region": "ewr", - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "plan": "vcg-l40s-16c-180g-48vram", - "nodeCount": 2, - }, - ], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + required_resources={ + "class-gpu-l40s-vultr": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l40s-vultr", + count=1, + memory="46068Mi", + provisioning={ + "provider": "Vultr", + "vultr": { + "plan": "vcg-l40s-16c-180g-48vram", + "accelerator": {"type": "nvidia-l40s", "count": 1}, + }, }, - ], - }, - } - ), + ), + ], + ), + }, ), - ) - - want15 = fnv1.RunFunctionResponse() - want15.CopyFrom(want14) - want15.desired.resources["vultr-cluster"].ready = fnv1.READY_TRUE - want15.desired.composite.CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l40s-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, + }, + ], }, - "namespace": "modelplane-system", - "gpuPools": [ - { - "name": "l40s-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, - }, - ], - }, - ], - }, + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=["kubernetes.vke.vultr.m.upbound.io", "kubernetesnodepools.vke.vultr.m.upbound.io"], + ready=fnv1.READY_TRUE, + ), + "vultr-cluster": _vultr_cluster( + credentials={"type": "ProviderConfig", "name": "my-vultr-account"}, ready=fnv1.READY_UNSPECIFIED + ), }, ), - ), - ) - want15.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, - }, - } + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l40s-vultr": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l40s-vultr" + ), + }, ), - ready=fnv1.READY_TRUE, + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], ), - ) - want15.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Vultr", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ + ), + # The VKE kubeconfig embeds static client certificates, so the + # ClusterProviderConfig carries no identity (unlike Nebius). VultrCluster + # reports no cache StorageClass, so status.cache stays unset. + ComposeCase( + name="VultrReady", + reason="Once its VultrCluster is Ready, a Vultr cluster composes a ClusterProviderConfig without identity, the ServingStack and the Usage.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="Vultr", vultr=v1alpha1.Vultr(region="ewr")), + node_pools=[ + v1alpha1.NodePool(name="l40s-pool", className="gpu-l40s-vultr", nodeCount=2, maxNodeCount=4), + ], + ), + resources={ + "vultr-cluster": fnv1.Resource( + resource=resource.dict_to_struct( { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - ], - }, - } + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "VultrCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "ewr", + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "plan": "vcg-l40s-16c-180g-48vram", + "nodeCount": 2, + }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + }, + } + ), + ), + "activation": _observed_activation_policy( + activated=["kubernetes.vke.vultr.m.upbound.io", "kubernetesnodepools.vke.vultr.m.upbound.io"] + ), + }, ), + required_resources={ + "class-gpu-l40s-vultr": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l40s-vultr", + count=1, + memory="46068Mi", + provisioning={ + "provider": "Vultr", + "vultr": { + "plan": "vcg-l40s-16c-180g-48vram", + "accelerator": {"type": "nvidia-l40s", "count": 1}, + }, + }, + ), + ], + ), + }, ), - ) - want15.desired.resources["usage-vultr-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l40s-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, + }, + ], }, - "replayDeletion": True, - }, - } + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=["kubernetes.vke.vultr.m.upbound.io", "kubernetesnodepools.vke.vultr.m.upbound.io"], + ready=fnv1.READY_TRUE, + ), + "vultr-cluster": _vultr_cluster(credentials=None, ready=fnv1.READY_TRUE), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="test-cluster-kubeconfig-abcde", identity=None + ), + "serving-stack": _serving_stack( + cloud="Vultr", + secrets=[{"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}], + client_cas=None, + gpu=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-vultr-by-backend": _backend_usage(cluster_kind="VultrCluster"), + }, ), - ready=fnv1.READY_TRUE, - ), - ) - del want15.conditions[:] - want15.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want15.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Vultr cluster ready, composing backend", - ) - ) - - # --- Case 16: Civo first pass composes the CivoCluster XR only. - # minNodeCount stays unset so the pool's autoscaling floor defaults - # to its node count downstream. --- - inference_class_l40s_civo = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-l40s-civo"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Vultr cluster ready, composing backend")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l40s-vultr": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l40s-vultr" + ), }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), ], - "provisioning": { - "provider": "Civo", - "civo": { - "size": "an.g1.l40s.kube.x1", - "accelerator": {"type": "nvidia-l40s", "count": 1}, - }, - }, - }, - } - class_selector_civo = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-l40s-civo", - ) - - req16 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Civo", - civo=v1alpha1.Civo(region="LON1"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l40s-pool", - className="gpu-l40s-civo", - nodeCount=2, - maxNodeCount=4, - ), - ], - ), - ).model_dump(exclude_none=True, mode="json"), + ), + ), + # minNodeCount stays unset so the pool's autoscaling floor defaults to its + # node count downstream. + ComposeCase( + name="CivoFirstPass", + reason="With no CivoCluster observed yet, a Civo cluster composes only the activation policy and the CivoCluster.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="Civo", civo=v1alpha1.Civo(region="LON1")), + node_pools=[ + v1alpha1.NodePool(name="l40s-pool", className="gpu-l40s-civo", nodeCount=2, maxNodeCount=4), + ], ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "clusters.kubernetes.civo.m.upbound.io", + "nodepools.kubernetes.civo.m.upbound.io", + "networks.vpc.civo.m.upbound.io", + "firewalls.vpc.civo.m.upbound.io", + ] + ), + }, ), - ), - ) - req16.required_resources["class-gpu-l40s-civo"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l40s_civo)), - ) - - want16 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + required_resources={ + "class-gpu-l40s-civo": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l40s-civo", + count=1, + memory="46068Mi", + provisioning={ + "provider": "Civo", + "civo": { + "size": "an.g1.l40s.kube.x1", + "accelerator": {"type": "nvidia-l40s", "count": 1}, + }, }, - "namespace": "modelplane-system", - "gpuPools": [ + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l40s-pool", + "nodes": 4, + "devices": [ { - "name": "l40s-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, }, ], }, - }, + ], + cache=None, + gateway=None, ), - ), - resources={ - "civo-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "CivoCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "LON1", - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "size": "an.g1.l40s.kube.x1", - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-l40s", - }, - }, - ], + resources={ + "activation": _activation_policy( + activate=[ + "clusters.kubernetes.civo.m.upbound.io", + "nodepools.kubernetes.civo.m.upbound.io", + "networks.vpc.civo.m.upbound.io", + "firewalls.vpc.civo.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "civo-cluster": _civo_cluster( + node_pools=[ + { + "name": "l40s-pool", + "role": "GPU", + "size": "an.g1.l40s.kube.x1", + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": {"acceleratorType": "nvidia-l40s"}, }, - }, + ], + credentials=None, + ready=fnv1.READY_UNSPECIFIED, ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", + }, ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l40s-civo": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l40s-civo" + ), + }, ), - ], - context=structpb.Struct(), - ) - want16.requirements.resources["class-gpu-l40s-civo"].CopyFrom(class_selector_civo) - - # --- Case 16b: Civo credentials pass through to the CivoCluster - # spec, mirroring the Vultr passthrough. --- - req_creds_civo = fnv1.RunFunctionRequest() - req_creds_civo.CopyFrom(req16) - req_creds_civo.observed.composite.CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Civo", - civo=v1alpha1.Civo( - region="LON1", - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="my-civo-account", - ), - ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], + ), + ), + ComposeCase( + name="CivoCredentials", + reason="A Civo cluster's credentials pass through to the CivoCluster's spec.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Civo", + civo=v1alpha1.Civo( + region="LON1", + credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-civo-account"), ), - nodePools=[ - v1alpha1.NodePool( - name="l40s-pool", - className="gpu-l40s-civo", - nodeCount=2, - maxNodeCount=4, - ), - ], ), - ).model_dump(exclude_none=True, mode="json"), + node_pools=[ + v1alpha1.NodePool(name="l40s-pool", className="gpu-l40s-civo", nodeCount=2, maxNodeCount=4), + ], + ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "clusters.kubernetes.civo.m.upbound.io", + "nodepools.kubernetes.civo.m.upbound.io", + "networks.vpc.civo.m.upbound.io", + "firewalls.vpc.civo.m.upbound.io", + ] + ), + }, ), + required_resources={ + "class-gpu-l40s-civo": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l40s-civo", + count=1, + memory="46068Mi", + provisioning={ + "provider": "Civo", + "civo": { + "size": "an.g1.l40s.kube.x1", + "accelerator": {"type": "nvidia-l40s", "count": 1}, + }, + }, + ), + ], + ), + }, ), - ) - - want_creds_civo = fnv1.RunFunctionResponse() - want_creds_civo.CopyFrom(want16) - want_creds_civo.desired.resources["civo-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "CivoCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "LON1", - "credentials": { - "type": "ProviderConfig", - "name": "my-civo-account", + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l40s-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, + }, + ], }, - "nodePools": [ + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "clusters.kubernetes.civo.m.upbound.io", + "nodepools.kubernetes.civo.m.upbound.io", + "networks.vpc.civo.m.upbound.io", + "firewalls.vpc.civo.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "civo-cluster": _civo_cluster( + node_pools=[ { "name": "l40s-pool", "role": "GPU", "size": "an.g1.l40s.kube.x1", "nodeCount": 2, "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-l40s", - }, + "gpu": {"acceleratorType": "nvidia-l40s"}, }, ], - }, + credentials={"type": "ProviderConfig", "name": "my-civo-account"}, + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l40s-civo": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l40s-civo" + ), }, ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], ), - ) - - # --- Case 17: Civo cluster ready - kubeconfig observed on the - # CivoCluster status. The Civo kubeconfig embeds static client - # certificates, so the ClusterProviderConfig carries no identity - # (unlike Nebius). The function composes the ServingStack backend - # with the kubeconfig and emits the Usage that blocks CivoCluster - # deletion until the ServingStack is gone. CivoCluster reports no - # cache StorageClass, so status.cache stays unset. --- - req17 = fnv1.RunFunctionRequest() - req17.CopyFrom(req16) - req17.observed.resources["civo-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "CivoCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "region": "LON1", - "nodePools": [ + ), + # The Civo kubeconfig embeds static client certificates, so the + # ClusterProviderConfig carries no identity (unlike Nebius). CivoCluster + # reports no cache StorageClass, so status.cache stays unset. + ComposeCase( + name="CivoReady", + reason="Once its CivoCluster is Ready, a Civo cluster composes a ClusterProviderConfig without identity, the ServingStack and the Usage.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="Civo", civo=v1alpha1.Civo(region="LON1")), + node_pools=[ + v1alpha1.NodePool(name="l40s-pool", className="gpu-l40s-civo", nodeCount=2, maxNodeCount=4), + ], + ), + resources={ + "civo-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "CivoCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "LON1", + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "size": "an.g1.l40s.kube.x1", + "nodeCount": 2, + }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + }, + } + ), + ), + "activation": _observed_activation_policy( + activated=[ + "clusters.kubernetes.civo.m.upbound.io", + "nodepools.kubernetes.civo.m.upbound.io", + "networks.vpc.civo.m.upbound.io", + "firewalls.vpc.civo.m.upbound.io", + ] + ), + }, + ), + required_resources={ + "class-gpu-l40s-civo": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l40s-civo", + count=1, + memory="46068Mi", + provisioning={ + "provider": "Civo", + "civo": { + "size": "an.g1.l40s.kube.x1", + "accelerator": {"type": "nvidia-l40s", "count": 1}, + }, + }, + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l40s-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "clusters.kubernetes.civo.m.upbound.io", + "nodepools.kubernetes.civo.m.upbound.io", + "networks.vpc.civo.m.upbound.io", + "firewalls.vpc.civo.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "civo-cluster": _civo_cluster( + node_pools=[ { "name": "l40s-pool", "role": "GPU", "size": "an.g1.l40s.kube.x1", "nodeCount": 2, + "maxNodeCount": 4, + "gpu": {"acceleratorType": "nvidia-l40s"}, }, ], - }, - "status": { - "conditions": [ + credentials=None, + ready=fnv1.READY_TRUE, + ), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="test-cluster-kubeconfig-abcde", identity=None + ), + "serving-stack": _serving_stack( + cloud="Civo", + secrets=[{"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}], + client_cas=None, + gpu=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-civo-by-backend": _backend_usage(cluster_kind="CivoCluster"), + }, + ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Civo cluster ready, composing backend")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l40s-civo": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l40s-civo" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], + ), + ), + # A lone H100 SXM has NVLink links but no peer, so the serving stack must + # load that pool's driver with NVreg_NvLinkDisable=1. CivoReady's L40S pool + # gets no spec.gpu. + # + # The request still returns the gpu-l40s-civo class, which no node pool + # references, and the observed CivoCluster still reports l40s-pool in its + # spec. The function reads neither. + ComposeCase( + name="CivoSingleH100", + reason="Once its CivoCluster is Ready, a Civo cluster whose pool has one H100 per node disables NVLink for that pool in the ServingStack's spec.gpu.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="Civo", civo=v1alpha1.Civo(region="LON1")), + node_pools=[ + v1alpha1.NodePool(name="h100-pool", className="gpu-h100-civo", nodeCount=1), + ], + ), + resources={ + "civo-cluster": fnv1.Resource( + resource=resource.dict_to_struct( { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "CivoCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "LON1", + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "size": "an.g1.l40s.kube.x1", + "nodeCount": 2, + }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + }, + } + ), + ), + "activation": _observed_activation_policy( + activated=[ + "clusters.kubernetes.civo.m.upbound.io", + "nodepools.kubernetes.civo.m.upbound.io", + "networks.vpc.civo.m.upbound.io", + "firewalls.vpc.civo.m.upbound.io", + ] + ), + }, + ), + required_resources={ + "class-gpu-l40s-civo": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l40s-civo", + count=1, + memory="46068Mi", + provisioning={ + "provider": "Civo", + "civo": { + "size": "an.g1.l40s.kube.x1", + "accelerator": {"type": "nvidia-l40s", "count": 1}, + }, }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + ), + ], + ), + "class-gpu-h100-civo": fnv1.Resources( + items=[ + _inference_class( + name="gpu-h100-civo", + count=1, + memory="81559Mi", + provisioning={ + "provider": "Civo", + "civo": { + "size": "an.g1.h100.kube.x1", + "accelerator": {"type": "nvidia-h100", "count": 1}, + }, }, - ], - }, - } - ), + ), + ], + ), + }, ), - ) - - want17 = fnv1.RunFunctionResponse() - want17.CopyFrom(want16) - want17.desired.resources["civo-cluster"].ready = fnv1.READY_TRUE - want17.desired.composite.CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "h100-pool", + "nodes": 1, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], }, - "namespace": "modelplane-system", - "gpuPools": [ + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "clusters.kubernetes.civo.m.upbound.io", + "nodepools.kubernetes.civo.m.upbound.io", + "networks.vpc.civo.m.upbound.io", + "firewalls.vpc.civo.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "civo-cluster": _civo_cluster( + node_pools=[ { - "name": "l40s-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, - }, - ], + "name": "h100-pool", + "role": "GPU", + "size": "an.g1.h100.kube.x1", + "nodeCount": 1, + "gpu": {"acceleratorType": "nvidia-h100"}, }, ], - }, + credentials=None, + ready=fnv1.READY_TRUE, + ), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="test-cluster-kubeconfig-abcde", identity=None + ), + "serving-stack": _serving_stack( + cloud="Civo", + secrets=[{"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}], + client_cas=None, + gpu={"pools": [{"name": "h100-pool", "disableNvLink": True}]}, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-civo-by-backend": _backend_usage(cluster_kind="CivoCluster"), }, ), - ), - ) - want17.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, - }, - } + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Civo cluster ready, composing backend")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-h100-civo": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-h100-civo" + ), + }, ), - ready=fnv1.READY_TRUE, + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], ), - ) - want17.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Civo", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - ], - }, - } + ), + # The deletion guard cases, ReplicasRoutesAndCaches through ClassUnresolved, + # observe an existing cluster along with whatever uses it. + # + # One reason-only ClusterUsage blocks deletion whatever the count or + # namespace of its users. The mirrored namespaces are the deduplicated union + # of team-a (replica), team-b (replica and route), team-c (route) and team-d + # (cache staging onto this cluster). A cache staging only onto another + # cluster (team-e) is filtered out by its status.clusters[], proving the + # namespaces track what actually lands here. + ComposeCase( + name="ReplicasRoutesAndCaches", + reason="ModelReplicas, ModelRoutes and ModelCaches across namespaces compose one ClusterUsage naming each kind, and mirror the namespaces that land here.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], + ), ), + required_resources={ + "model-replicas": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "deploy-test-cluster-0", + "namespace": "team-a", + "labels": {"modelplane.ai/cluster": "test-cluster"}, + }, + } + ) + ), + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "deploy-test-cluster-0", + "namespace": "team-b", + "labels": {"modelplane.ai/cluster": "test-cluster"}, + }, + } + ) + ), + ], + ), + "model-routes": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelRoute", + "metadata": { + "name": "svc-eu", + "namespace": "team-b", + "labels": {"modelplane.ai/cluster": "test-cluster"}, + }, + } + ) + ), + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelRoute", + "metadata": { + "name": "svc-eu", + "namespace": "team-c", + "labels": {"modelplane.ai/cluster": "test-cluster"}, + }, + } + ) + ), + ], + ), + "model-caches": fnv1.Resources( + items=[ + _model_cache(name="qwen", namespace="team-d", cluster="test-cluster"), + _model_cache(name="kimi", namespace="team-e", cluster="other-cluster"), + ], + ), + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + ), + ], + ), + }, ), - ) - want17.desired.resources["usage-civo-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "CivoCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], }, - "replayDeletion": True, - }, - } + ], + cache=None, + gateway=None, + ), + resources={ + "usage-replicas": _guard_cluster_usage( + reason="ModelReplicas, ModelRoutes and ModelCaches use this InferenceCluster" + ), + "namespace-team-a": _namespace_object(team="team-a", name="mp-team-a-bd964"), + "namespace-team-b": _namespace_object(team="team-b", name="mp-team-b-6bd62"), + "namespace-team-c": _namespace_object(team="team-c", name="mp-team-c-d79d9"), + "namespace-team-d": _namespace_object(team="team-d", name="mp-team-d-c2383"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + gpu=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, ), - ready=fnv1.READY_TRUE, - ), - ) - del want17.conditions[:] - want17.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want17.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Civo cluster ready, composing backend", - ) - ) - - # --- Case 17b: a Civo pool of single-H100 nodes (accelerator - # nvidia-h100, count 1) projects per-pool NVLink disable into the - # backend's spec.gpu: a lone H100 SXM has NVLink links but no peer, - # so the serving stack must load that pool's driver with - # NVreg_NvLinkDisable=1. The L40S case above projects no spec.gpu. --- - inference_class_h100_civo = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-h100-civo"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "81559Mi"}}, + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), ], - "provisioning": { - "provider": "Civo", - "civo": { - "size": "an.g1.h100.kube.x1", - "accelerator": {"type": "nvidia-h100", "count": 1}, - }, + ), + ), + # A ModelRoute composes its routing Objects through the cluster's + # ClusterProviderConfig, so it needs the cluster as much as a replica does. + ComposeCase( + name="RouteOnCluster", + reason="A ModelRoute on the cluster, with no replica there, composes the guard and mirrors its team's namespace.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], + ), + ), + required_resources={ + "model-routes": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelRoute", + "metadata": { + "name": "svc-eu", + "namespace": "team-c", + "labels": {"modelplane.ai/cluster": "test-cluster"}, + }, + } + ) + ) + ] + ), + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + ), + ], + ), }, - }, - } - - req17b = fnv1.RunFunctionRequest() - req17b.CopyFrom(req17) - req17b.observed.composite.CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Civo", - civo=v1alpha1.Civo(region="LON1"), - ), - nodePools=[ - v1alpha1.NodePool( - name="h100-pool", - className="gpu-h100-civo", - nodeCount=1, - ), - ], + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "usage-replicas": _guard_cluster_usage(reason="ModelRoutes use this InferenceCluster"), + "namespace-team-c": _namespace_object(team="team-c", name="mp-team-c-d79d9"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + gpu=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), ), - ).model_dump(exclude_none=True, mode="json"), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], ), - ) - req17b.required_resources["class-gpu-h100-civo"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_civo)), - ) - - want17b = fnv1.RunFunctionResponse() - want17b.CopyFrom(want17) - del want17b.requirements.resources["class-gpu-l40s-civo"] - want17b.requirements.resources["class-gpu-h100-civo"].CopyFrom( - fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-h100-civo", + ), + # A ModelCache composes its PVC through the cluster's ClusterProviderConfig, + # so it needs the cluster as much as a replica does. + ComposeCase( + name="CacheOnCluster", + reason="A ModelCache staging onto the cluster, with no replica there, composes the guard and mirrors its team's namespace.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], + ), + ), + required_resources={ + "model-caches": fnv1.Resources( + items=[_model_cache(name="qwen", namespace="team-d", cluster="test-cluster")] + ), + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + ), + ], + ), + }, ), - ) - want17b.desired.composite.CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], }, - "namespace": "modelplane-system", - "gpuPools": [ - { - "name": "h100-pool", - "nodes": 1, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "81559Mi"}}, - }, - ], - }, - ], - }, + ], + cache=None, + gateway=None, + ), + resources={ + "usage-replicas": _guard_cluster_usage(reason="ModelCaches use this InferenceCluster"), + "namespace-team-d": _namespace_object(team="team-d", name="mp-team-d-c2383"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + gpu=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), }, ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], ), - ) - want17b.desired.resources["civo-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "CivoCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "LON1", - "nodePools": [ - { - "name": "h100-pool", - "role": "GPU", - "size": "an.g1.h100.kube.x1", - "nodeCount": 1, - "gpu": { - "acceleratorType": "nvidia-h100", + ), + # An InferenceGateway composes its Gateway and routing Objects through the + # cluster's ClusterProviderConfig just as a replica does. It is cluster + # scoped, so it has no namespace to mirror. + ComposeCase( + name="GatewayOnCluster", + reason="An InferenceGateway on the cluster composes the guard but mirrors no namespace.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], + ), + ), + required_resources={ + "gateways": fnv1.Resources( + items=[_inference_gateway(name="public", cluster="test-cluster", status=None)] + ), + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, }, }, - ], - }, - }, - ), - ready=fnv1.READY_TRUE, + ), + ], + ), + }, ), - ) - want17b.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Civo", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "gpu": { - "pools": [ - {"name": "h100-pool", "disableNvLink": True}, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, ], }, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - ], - }, - } + ], + cache=None, + gateway=None, + ), + resources={ + "usage-replicas": _guard_cluster_usage(reason="InferenceGateways use this InferenceCluster"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + gpu=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], ), - ) - - # Every compose path emits the ModelReplica guard requirement. - for want in ( - want1, - want2, - want3, - want4, - want5, - want6, - want7, - want8, - want9, - want10, - want11, - want12, - want13, - want14, - want_creds_vultr, - want15, - want16, - want_creds_civo, - want17, - want17b, - ): - want.requirements.resources["gateways"].CopyFrom(_gateways_selector()) - want.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) - want.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) - want.requirements.resources["model-caches"].CopyFrom(_caches_selector()) - - # The guard cases reuse case 1's request and response. - guard_cases = [ - Case( - "ModelReplicas, ModelRoutes and ModelCaches compose the guard and the mirrored namespaces", - *_guard_case(req1, want1), - ), - Case("a ModelRoute on the cluster composes the guard", *_route_guard_case(req1, want1)), - Case("a ModelCache staging onto the cluster composes the guard", *_cache_guard_case(req1, want1)), - Case("an InferenceGateway on the cluster composes the guard", *_gateway_guard_case(req1, want1)), - Case("nothing on the cluster leaves it deletable", *_unused_case(req1, want1)), - Case("guard is composed even when compose returns early", *_early_return_guard_case()), - ] - - # --- Case credentials: GKE with custom credentials passes them through to GKECluster. --- - req_creds = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="GKE", - gke=v1alpha1.Gke( - region="us-central1", - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="my-gcp-account", - ), - ), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - zones=["us-central1-a"], - ), - ], - ), - ).model_dump(exclude_none=True, mode="json") + ), + # This is the teardown transition - the last user is gone, so the function + # stops composing the guard. The replica and route requirements are empty + # but present, as Crossplane returns them when a selector matches nothing. + ComposeCase( + name="NothingOnCluster", + reason="With only a ModelCache and an InferenceGateway on another cluster, the cluster composes no ClusterUsage and is deletable.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], ), ), - ), - ) - req_creds.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - want_creds = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + required_resources={ + "model-replicas": fnv1.Resources(), + "model-routes": fnv1.Resources(), + "model-caches": fnv1.Resources( + items=[_model_cache(name="kimi", namespace="team-e", cluster="other-cluster")] + ), + "gateways": fnv1.Resources( + items=[_inference_gateway(name="elsewhere", cluster="other-cluster", status=None)] + ), + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - "namespace": "modelplane-system", - "gpuPools": [ + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - } + ], + cache=None, + gateway=None, + ), + resources={ + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + gpu=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], + ), + ), + # resolve_classes() returns False whenever a referenced InferenceClass isn't + # observed yet - a routine transient. The guard runs first, so a + # referencing replica still blocks deletion. This is the case that + # regresses if the guard is gated behind class resolution or cluster source. + # + # Only the guard and namespace are composed, both ready, so the function + # marks the XR not ready itself. Otherwise it would read ready while it + # waits for its classes. + ComposeCase( + name="ClassUnresolved", + reason="With its class unresolved and a replica on the cluster, the function composes the guard and the replica's namespace before returning early.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], ), ), - resources={ - "gke-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-central1", - "kubernetesVersion": "1.35", - "credentials": { - "type": "ProviderConfig", - "name": "my-gcp-account", - }, - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "machineType": "g2-standard-48", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - "acceleratorCount": 1, - }, - "zones": ["us-central1-a"], + required_resources={ + "model-replicas": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "deploy-test-cluster-0", + "namespace": "team-a", + "labels": {"modelplane.ai/cluster": "test-cluster"}, }, - ], - }, - } - ), + } + ) + ) + ] ), }, ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(ready=fnv1.READY_FALSE), + resources={ + "usage-replicas": _guard_cluster_usage(reason="ModelReplicas use this InferenceCluster"), + "namespace-team-a": _namespace_object(team="team-a", name="mp-team-a-bd964"), + }, ), - ], - context=structpb.Struct(), - ) - want_creds.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - want_creds.requirements.resources["gateways"].CopyFrom(_gateways_selector()) - want_creds.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) - want_creds.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) - want_creds.requirements.resources["model-caches"].CopyFrom(_caches_selector()) - - # Every cloud cluster composes an activation policy; with the policy - # observed Healthy the cluster XR is composed. - for req, want, kinds in [ - (req2, want2, fn._ACTIVATE_GCP), - (req_creds, want_creds, fn._ACTIVATE_GCP), - (req4, want4, fn._ACTIVATE_AWS), - (req5, want5, fn._ACTIVATE_AWS), - (req6, want6, fn._ACTIVATE_GCP), - (req7, want7, fn._ACTIVATE_AWS), - (req8, want8, fn._ACTIVATE_AWS), - (req9, want9, fn._ACTIVATE_AWS), - (req10, want10, fn._ACTIVATE_NEBIUS), - (req11, want11, fn._ACTIVATE_NEBIUS), - (req12, want12, fn._ACTIVATE_AZURE), - (req13, want13, fn._ACTIVATE_AZURE), - (req14, want14, fn._ACTIVATE_VULTR), - (req_creds_vultr, want_creds_vultr, fn._ACTIVATE_VULTR), - (req15, want15, fn._ACTIVATE_VULTR), - (req16, want16, fn._ACTIVATE_CIVO), - (req_creds_civo, want_creds_civo, fn._ACTIVATE_CIVO), - (req17, want17, fn._ACTIVATE_CIVO), - (req17b, want17b, fn._ACTIVATE_CIVO), - ]: - _observe_activated(req, kinds) - _want_activation(want, kinds) - - # While the policy is missing even one of the kinds from status.activated - # (e.g. a provider still installing), and with no cluster observed, the - # function composes only the activation policy (not marked ready, so the - # composite doesn't report ready), not the cluster XR. Observing the - # policy with a kind missing exercises the all-kinds check rather than - # the policy-absent branch. - req_unactivated = copy.deepcopy(req2) - _observe_activated(req_unactivated, fn._ACTIVATE_GCP[:-1]) - want_unactivated = copy.deepcopy(want2) - del want_unactivated.desired.resources["gke-cluster"] - want_unactivated.desired.resources["activation"].ClearField("ready") - - # Once the cluster is observed, the function keeps composing it even - # when the policy momentarily stops reporting the kinds active, so an - # activation blip never drops a provisioned cluster from desired state. - req_blip = copy.deepcopy(req6) - del req_blip.observed.resources["activation"] - want_blip = copy.deepcopy(want6) - - return [ - Case(name="existing cluster with secrets composes backend and CPC", req=req1, want=want1), - Case(name="existing cluster with a non-GCP identity threads the identity type", req=req1b, want=want1b), - Case(name="GKE cluster first pass composes GKECluster XR only", req=req2, want=want2), - Case(name="GKE credentials pass through to GKECluster spec", req=req_creds, want=want_creds), - Case(name="existing cluster second pass with backend ready", req=req3, want=want3), - Case(name="EKS cluster first pass composes EKSCluster XR only", req=req4, want=want4), - Case(name="EKS cluster not ready re-emits existing CPC unchanged", req=req5, want=want5), - Case(name="GKE cluster ready composes CPC, backend, usage, and RWX StorageClass", req=req6, want=want6), - Case(name="EKS cluster ready composes ServingStack and Usage", req=req7, want=want7), - Case( - name="EKS node pool with a Capacity Block sets capacityBlock on the EKSCluster pool", - req=req8, - want=want8, - ), - Case( - name="EKS node pool with fabric EFA sets fabric on the EKSCluster pool", - req=req9, - want=want9, - ), - Case(name="Nebius cluster first pass composes NebiusCluster XR only", req=req10, want=want10), - Case( - name="Nebius cluster ready composes CPC with Nebius identity, ServingStack, and Usage", - req=req11, - want=want11, - ), - Case(name="AKS cluster first pass composes AKSCluster XR only", req=req12, want=want12), - Case( - name="AKS cluster ready composes CPC without identity, ServingStack, and Usage", - req=req13, - want=want13, - ), - Case(name="cloud cluster not activated composes only the policy", req=req_unactivated, want=want_unactivated), - Case(name="observed cluster keeps composing through an activation blip", req=req_blip, want=want_blip), - Case(name="Vultr cluster first pass composes VultrCluster XR only", req=req14, want=want14), - Case( - name="Vultr credentials pass through to VultrCluster spec", - req=req_creds_vultr, - want=want_creds_vultr, - ), - Case( - name="Vultr cluster ready composes CPC without identity, ServingStack, and Usage", - req=req15, - want=want15, - ), - Case(name="Civo cluster first pass composes CivoCluster XR only", req=req16, want=want16), - Case( - name="Civo credentials pass through to CivoCluster spec", - req=req_creds_civo, - want=want_creds_civo, - ), - Case( - name="Civo cluster ready composes CPC without identity, ServingStack, and Usage", - req=req17, - want=want17, - ), - Case( - name="Civo single-H100 pool projects per-pool NVLink disable to the backend", - req=req17b, - want=want17b, - ), - *guard_cases, - ] - - -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) -def test_compose(case: Case) -> None: - """RunFunction composes the resources an InferenceCluster needs.""" - got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) - - -# The hostname gate, which is what keeps a cluster off the schedule until -# traffic to it is mutually authenticated in both directions. - - -def _gateway_status_request(*, address: str | None, ca: str | None, gateway_cas: list[str]) -> fnv1.RunFunctionRequest: - """A cluster and whatever its serving stack and the fleet's gateways have - published so far. The gateway name is Modelplane's own, so nothing - configures it.""" - xr = v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for InferenceClasses: gpu-l4")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForClasses", + message="Waiting for InferenceClasses: gpu-l4", + ), + ], ), - ) - stack_status: dict = {"conditions": [{"type": "Ready", "status": "True"}]} - gateway: dict = {} - if address: - gateway["address"] = address - if ca: - gateway["caCertificate"] = ca - if gateway: - stack_status["gateway"] = gateway - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json"))), - resources={ - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": {"name": "test-cluster-serving-stack-fd00b"}, - "status": stack_status, - } + ), + # The hostname gate, which is what keeps a cluster off the schedule until + # traffic to it is mutually authenticated in both directions. These cases + # observe an existing cluster's ServingStack ready, with whatever it and the + # fleet's InferenceGateways have published so far. The gateway name is + # Modelplane's own, so nothing configures it. An InferenceGateway running on + # test-cluster also composes the deletion guard. + # + # This cluster's CA lets an InferenceGateway tell it reached the right + # cluster, and an InferenceGateway CA makes the cluster gateway demand a + # client certificate. + ComposeCase( + name="MutuallyAuthenticated", + reason="With an address, this cluster's CA and an InferenceGateway CA all published, the cluster publishes its gateway hostname.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=None, + ), + resources={ + "serving-stack": _observed_serving_stack( + gateway={"address": "34.55.100.10", "caCertificate": "cluster-ca"} ), + }, + ), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="fleet-0", cluster="test-cluster", status={"clientCACertificate": "fleet-ca"} + ) + ], ), }, ), - ) - for i, cert in enumerate(gateway_cas): - req.required_resources["gateways"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceGateway", - "metadata": {"name": f"fleet-{i}"}, - "spec": {"clusterName": "test-cluster"}, - "status": {"clientCACertificate": cert}, - } + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[], + cache=None, + gateway={ + "address": "34.55.100.10", + "caCertificate": "cluster-ca", + "hostname": "gateway-test-cluster-09532.modelplane-system.svc.cluster.local", + }, ), - ) - ) - return req - - -def _gateway_status(req: fnv1.RunFunctionRequest) -> dict: - """The status.gateway RunFunction writes to the InferenceCluster for req.""" - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - return resource.struct_to_dict(got.desired.composite.resource).get("status", {}).get("gateway", {}) - - -def test_gateway_status_hostname_published_once_both_directions_are_authenticated() -> None: - """The hostname is published once the address and both directions' CAs are.""" - # An address to reach, this cluster's CA so an InferenceGateway can tell it - # reached the right cluster, and an InferenceGateway CA so the cluster - # gateway demands a client certificate. - status = _gateway_status(_gateway_status_request(address="34.55.100.10", ca="cluster-ca", gateway_cas=["fleet-ca"])) - assert status == { - "address": "34.55.100.10", - "caCertificate": "cluster-ca", - "hostname": _GATEWAY_HOSTNAME, - } - - -def test_gateway_status_no_hostname_without_an_inference_gateway_ca() -> None: - """No hostname is published until an InferenceGateway has published a CA.""" + resources={ + "usage-replicas": _guard_cluster_usage(reason="InferenceGateways use this InferenceCluster"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=[{"name": "fleet-0", "certificate": "fleet-ca"}], + gpu=None, + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_TRUE, reason="BackendHealthy"), + ], + ), + ), # The case that matters: the cluster gateway only demands a client # certificate when it has a CA to check against, and with none it serves no # Gateway at all. Publishing the hostname anyway would make the cluster - # schedulable when nothing is listening on it, so every request routed - # there would be stranded. - status = _gateway_status(_gateway_status_request(address="34.55.100.10", ca="cluster-ca", gateway_cas=[])) - assert status == {"address": "34.55.100.10", "caCertificate": "cluster-ca"} - - -def test_gateway_status_no_hostname_without_this_clusters_ca() -> None: - """No hostname is published until the cluster's own gateway CA is.""" - # Without it an InferenceGateway can't validate the cluster gateway it - # reaches, so it would have to fall back to the public trust store. - status = _gateway_status(_gateway_status_request(address="34.55.100.10", ca=None, gateway_cas=["fleet-ca"])) - assert status == {"address": "34.55.100.10"} - - -def test_gateway_status_no_gateway_status_before_an_address() -> None: - """No gateway status at all is published before the gateway has an address.""" + # schedulable when nothing is listening on it, so every request routed there + # would be stranded. + ComposeCase( + name="NoGatewayCA", + reason="With no InferenceGateway CA published, the cluster publishes its address and CA but no hostname.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=None, + ), + resources={ + "serving-stack": _observed_serving_stack( + gateway={"address": "34.55.100.10", "caCertificate": "cluster-ca"} + ), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[], cache=None, gateway={"address": "34.55.100.10", "caCertificate": "cluster-ca"} + ), + resources={ + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + gpu=None, + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_TRUE, reason="BackendHealthy"), + ], + ), + ), + # Without this cluster's CA an InferenceGateway can't validate the cluster + # gateway it reaches, so it would have to fall back to the public trust + # store. + ComposeCase( + name="NoClusterCA", + reason="Without this cluster's CA, the cluster publishes its address but no hostname.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=None, + ), + resources={ + "serving-stack": _observed_serving_stack(gateway={"address": "34.55.100.10"}), + }, + ), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="fleet-0", cluster="test-cluster", status={"clientCACertificate": "fleet-ca"} + ) + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster(gpu_pools=[], cache=None, gateway={"address": "34.55.100.10"}), + resources={ + "usage-replicas": _guard_cluster_usage(reason="InferenceGateways use this InferenceCluster"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=[{"name": "fleet-0", "certificate": "fleet-ca"}], + gpu=None, + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_TRUE, reason="BackendHealthy"), + ], + ), + ), # A hostname that resolves to nothing strands every request routed to it, # and the CA is republished from the same status. - status = _gateway_status(_gateway_status_request(address=None, ca="cluster-ca", gateway_cas=["fleet-ca"])) - assert status == {} - - -def test_gateway_status_serving_stack_accepts_every_inference_gateway_ca() -> None: - """The ServingStack's gateway accepts every published InferenceGateway CA, sorted.""" - # Any InferenceGateway may forward to this cluster, so its gateway accepts - # every published CA, whichever cluster the InferenceGateway runs on. These - # CAs are what switches the cluster gateway's mTLS listener on. One that - # hasn't published a CA yet is left out rather than holding the others - # back, and the list is sorted so it doesn't churn. - req = _gateway_status_request(address="34.55.100.10", ca="cluster-ca", gateway_cas=["fleet-0-ca", "fleet-1-ca"]) - for name, status in (("aaa", {"clientCACertificate": "aaa-ca"}), ("pending", {})): - req.required_resources["gateways"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceGateway", - "metadata": {"name": name}, - "spec": {"clusterName": "elsewhere"}, - "status": status, - } + ComposeCase( + name="NoAddress", + reason="Before its ServingStack publishes an address, the cluster publishes no gateway status, not even its CA.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=None, ), - ) - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - stack = resource.struct_to_dict(got.desired.resources[fn.BACKEND_RESOURCE_KEY].resource) - assert stack["spec"]["gateway"] == { - "hostname": _GATEWAY_HOSTNAME, - "clientCAs": [ - {"name": "aaa", "certificate": "aaa-ca"}, - {"name": "fleet-0", "certificate": "fleet-0-ca"}, - {"name": "fleet-1", "certificate": "fleet-1-ca"}, - ], - } + resources={ + "serving-stack": _observed_serving_stack(gateway={"caCertificate": "cluster-ca"}), + }, + ), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="fleet-0", cluster="test-cluster", status={"clientCACertificate": "fleet-ca"} + ) + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster(gpu_pools=[], cache=None, gateway=None), + resources={ + "usage-replicas": _guard_cluster_usage(reason="InferenceGateways use this InferenceCluster"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=[{"name": "fleet-0", "certificate": "fleet-ca"}], + gpu=None, + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_TRUE, reason="BackendHealthy"), + ], + ), + ), + # Any InferenceGateway may forward to this cluster, and these CAs are what + # switches the cluster gateway's mTLS listener on. One that hasn't published + # a CA yet is left out rather than holding the others back, and the list is + # sorted so it doesn't churn. + ComposeCase( + name="ManyGatewayCAs", + reason="The ServingStack accepts the CA of every InferenceGateway that has published one, on any cluster, sorted by name.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=None, + ), + resources={ + "serving-stack": _observed_serving_stack( + gateway={"address": "34.55.100.10", "caCertificate": "cluster-ca"} + ), + }, + ), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="fleet-0", cluster="test-cluster", status={"clientCACertificate": "fleet-0-ca"} + ), + _inference_gateway( + name="fleet-1", cluster="test-cluster", status={"clientCACertificate": "fleet-1-ca"} + ), + _inference_gateway(name="aaa", cluster="elsewhere", status={"clientCACertificate": "aaa-ca"}), + _inference_gateway(name="pending", cluster="elsewhere", status={}), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[], + cache=None, + gateway={ + "address": "34.55.100.10", + "caCertificate": "cluster-ca", + "hostname": "gateway-test-cluster-09532.modelplane-system.svc.cluster.local", + }, + ), + resources={ + "usage-replicas": _guard_cluster_usage(reason="InferenceGateways use this InferenceCluster"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=[ + {"name": "aaa", "certificate": "aaa-ca"}, + {"name": "fleet-0", "certificate": "fleet-0-ca"}, + {"name": "fleet-1", "certificate": "fleet-1-ca"}, + ], + gpu=None, + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_TRUE, reason="BackendHealthy"), + ], + ), + ), +] -# The derived gateway hostname doubles as an SNI and a certificate SAN, so two -# clusters must never derive the same one. +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: ComposeCase) -> None: + """RunFunction composes the resources an InferenceCluster needs.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want), case.reason -def test_gateway_hostname_dots_become_a_single_dns_label() -> None: - """A dotted cluster name still gives a hostname whose first label has no dots.""" +# The derived gateway hostname doubles as an SNI and a certificate SAN, so two +# clusters must never derive the same one. These cases call _gateway_hostname +# directly, because through RunFunction each cluster name would need a whole +# compose case, renaming every resource the function names after the cluster. +GATEWAY_HOSTNAME_CASES = [ # A dotted cluster name is a DNS-1123 subdomain, but the first segment of # the hostname has to be one DNS-1035 label. - hostname = fn._gateway_hostname("eu.example") - label = hostname.split(".")[0] - assert "." not in label - assert label.startswith("gateway-") - - -def test_gateway_hostname_names_differing_only_in_dots_do_not_collide() -> None: - """Cluster names that differ only in dots get different hostnames.""" - # The hash covers the raw cluster name, so 'eu.example' and 'eu-example' - # get different hostnames. Sharing one, a cluster's Service would shadow - # the other's under a certificate it accepts. - assert fn._gateway_hostname("eu.example") != fn._gateway_hostname("eu-example") + # + # With DashedTwin, this pins that eu.example and eu-example hash + # differently: the hash covers the raw cluster name, before dots become + # dashes. Sharing one hostname, a cluster's Service would shadow the other's + # under a certificate it accepts. + GatewayHostnameCase( + name="DottedName", + reason="A dotted cluster name becomes a single DNS label, hashed before its dots become dashes.", + cluster_name="eu.example", + want="gateway-eu-example-ad4f6.modelplane-system.svc.cluster.local", + ), + GatewayHostnameCase( + name="DashedTwin", + reason="A dashed cluster name hashes to a gateway hostname distinct from its dotted twin's.", + cluster_name="eu-example", + want="gateway-eu-example-1a2b0.modelplane-system.svc.cluster.local", + ), +] + + +@pytest.mark.parametrize("case", GATEWAY_HOSTNAME_CASES, ids=lambda case: case.name) +def test_gateway_hostname(case: GatewayHostnameCase) -> None: + """_gateway_hostname derives a cluster's gateway hostname.""" + assert fn._gateway_hostname(case.cluster_name) == case.want, case.reason diff --git a/functions/compose-inference-gateway/tests/test_fn.py b/functions/compose-inference-gateway/tests/test_fn.py index 22e76ec57..131ef4c76 100644 --- a/functions/compose-inference-gateway/tests/test_fn.py +++ b/functions/compose-inference-gateway/tests/test_fn.py @@ -15,7 +15,6 @@ """Tests for the compose-inference-gateway function.""" import asyncio -import base64 import dataclasses import json @@ -27,10 +26,7 @@ from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.inferencegateway import v1alpha1 - -_PC = "gw-eu-cluster-kubeconfig" -_CLUSTER = "gw-eu" -_ADDRESS = "34.56.129.3" +from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @dataclasses.dataclass @@ -38,134 +34,98 @@ class Case: """A test case for compose-inference-gateway.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse -def _xr(*, name: str = "eu", **spec) -> dict: # noqa: ANN003 - """The InferenceGateway XR, built from the generated model so a field the - XRD doesn't define can't creep into a test.""" - xr = v1alpha1.InferenceGateway( - apiVersion="modelplane.ai/v1alpha1", - kind="InferenceGateway", - metadata={"name": name}, - spec=v1alpha1.Spec(clusterName=_CLUSTER, **spec), - ) - return xr.model_dump(exclude_none=True, mode="json", by_alias=True) - - -def _api_key_auth() -> v1alpha1.Auth: - """Caller auth by API key, against the Secrets _requirements(auth=True) selects.""" - return v1alpha1.Auth( - method="APIKey", - apiKey=v1alpha1.ApiKey( - secretSelector=v1alpha1.SecretSelector(matchLabels={"modelplane.ai/inference-keys": "true"}) - ), +def _xr(*, name: str, tls: bool, caller_secret_labels: dict[str, str] | None) -> fnv1.Resource: + """The XR on gw-eu, with TLS from eu-tls-0 if tls, and API-key auth by Secrets matching caller_secret_labels unless None.""" + tls_spec = v1alpha1.Tls(certificateRefs=[v1alpha1.CertificateRef(name="eu-tls-0")]) if tls else None + auth = None + if caller_secret_labels is not None: + auth = v1alpha1.Auth( + method="APIKey", + apiKey=v1alpha1.ApiKey(secretSelector=v1alpha1.SecretSelector(matchLabels=caller_secret_labels)), + ) + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceGateway( + apiVersion="modelplane.ai/v1alpha1", + kind="InferenceGateway", + metadata=metav1.ObjectMeta(name=name), + spec=v1alpha1.Spec(clusterName="gw-eu", tls=tls_spec, auth=auth), + ).model_dump(exclude_none=True, mode="json", by_alias=True) + ) ) -def _cluster(*, provider_config: str | None = _PC) -> dict: - """An observed InferenceCluster, optionally without a providerConfigRef. - - A registered cluster with no GPU pools, which is what a region with callers - but no accelerators looks like, and the least a gateway needs. - """ - status: dict = {} - if provider_config: - status["providerConfigRef"] = {"name": provider_config} - return { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceCluster", - "metadata": {"name": _CLUSTER}, - "spec": { - "cluster": { - "source": "Existing", - "existing": {"secretRef": {"name": f"{_CLUSTER}-kubeconfig", "key": "kubeconfig"}}, - } - }, - "status": status, - } +def _desired_xr(*, status: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The desired XR, carrying status, or only its readiness if status is None.""" + if status is None: + return fnv1.Resource(ready=ready) + return fnv1.Resource(resource=resource.dict_to_struct({"status": status}), ready=ready) -def _cluster_with_gateway(name: str, *, address: str, hostname: str) -> dict: - """An observed InferenceCluster whose gateway has published an address and - the internal name Modelplane derived for it.""" - return { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceCluster", - "metadata": {"name": name}, - "spec": { - "cluster": { - "source": "Existing", - "existing": {"secretRef": {"name": f"{name}-kubeconfig", "key": "kubeconfig"}}, +def _cluster(*, provider_config_ref: bool) -> fnv1.Resource: + """The gateway's InferenceCluster, gw-eu, as the clusters requirement returns it.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "gw-eu"}, + "spec": { + "cluster": { + "source": "Existing", + "existing": {"secretRef": {"name": "gw-eu-kubeconfig", "key": "kubeconfig"}}, + } + }, + "status": {"providerConfigRef": {"name": "gw-eu-cluster-kubeconfig"}} if provider_config_ref else {}, } - }, - "status": {"gateway": {"address": address, "hostname": hostname}}, - } + ) + ) -def _gateway_xr(name: str, cluster: str) -> dict: - """Another InferenceGateway, for the one-per-cluster contest.""" - return { +def _inference_gateway(*, name: str, address: str | None) -> fnv1.Resource: + """An InferenceGateway on gw-eu, as the gateways requirement returns it.""" + gateway: dict = { "apiVersion": "modelplane.ai/v1alpha1", "kind": "InferenceGateway", "metadata": {"name": name}, - "spec": {"clusterName": cluster}, - } - - -def _secret(name: str, data: dict[str, str]) -> dict: - """A control-plane Secret, with values base64 encoded as the API server - stores them, since the function copies data verbatim.""" - return { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": name, "namespace": fn.CONTROL_PLANE_NAMESPACE}, - "data": {k: base64.b64encode(v.encode()).decode() for k, v in data.items()}, + "spec": {"clusterName": "gw-eu"}, } + if address is not None: + gateway["status"] = {"address": address} + return fnv1.Resource(resource=resource.dict_to_struct(gateway)) -def _required(**resources) -> dict: # noqa: ANN003 - """Build the request's required_resources map.""" - return { - name: fnv1.Resources(items=[fnv1.Resource(resource=resource.dict_to_struct(r)) for r in items]) - for name, items in resources.items() - } - - -def _requirements(*, auth: bool = False, tls: int = 0) -> fnv1.Requirements: - """The requirements the function always emits, in the order it emits them.""" - reqs = { - "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), - "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), - } - if auth: - reqs["caller-secrets"] = fnv1.ResourceSelector( - api_version="v1", - kind="Secret", - namespace=fn.CONTROL_PLANE_NAMESPACE, - match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), - ) - for i in range(tls): - reqs[f"tls-secret-{i}"] = fnv1.ResourceSelector( - api_version="v1", kind="Secret", namespace=fn.CONTROL_PLANE_NAMESPACE, match_name=f"eu-tls-{i}" +def _caller_key_secret(*, name: str, data: dict[str, str]) -> fnv1.Resource: + """A caller-key Secret, as the caller-secrets requirement returns it.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": name, "namespace": "modelplane-system"}, + "data": data, + } ) - return fnv1.Requirements(resources=reqs) - + ) -def _observed_gateway(address: str | None, *, ready: bool) -> fnv1.Resource: - """The composed Gateway Object as observed, optionally with an address. - lastTransitionTime is fixed so the input is deterministic. - """ - manifest: dict = { - "apiVersion": "gateway.networking.k8s.io/v1", - "kind": "Gateway", - "metadata": {"name": fn._GATEWAY_NAME, "namespace": fn.REMOTE_NAMESPACE}, +def _observed_gateway(*, address: str, ready: bool) -> fnv1.Resource: + """The composed Gateway Object, observed once the Gateway has an address, with a Ready condition only if ready.""" + status: dict = { + "atProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "Gateway", + "metadata": {"name": "inference-gateway", "namespace": "modelplane-system"}, + "status": {"addresses": [{"type": "IPAddress", "value": address}]}, + } + } } - if address: - manifest["status"] = {"addresses": [{"type": "IPAddress", "value": address}]} - status: dict = {"atProvider": {"manifest": manifest}} if ready: status["conditions"] = [ { @@ -186,12 +146,8 @@ def _observed_gateway(address: str | None, *, ready: bool) -> fnv1.Resource: ) -def _observed_accepted() -> fnv1.Resource: - """A composed policy Object as observed once accepted. - - Its readiness comes from a CEL query on the policy's own Accepted condition, - so an Object that merely applied isn't enough. - """ +def _observed_caller_auth() -> fnv1.Resource: + """The composed caller-auth Object, observed Ready once its policy is accepted.""" return fnv1.Resource( resource=resource.dict_to_struct( { @@ -212,353 +168,156 @@ def _observed_accepted() -> fnv1.Resource: ) -def _not_ready(reason: str, message: str, requirements: fnv1.Requirements) -> fnv1.RunFunctionResponse: - """The whole response for a pass that composes nothing: no desired - resources, one GatewayReady=False condition, and the reason as a result.""" - return fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - context=structpb.Struct(), - requirements=requirements, - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=reason, - message=message, - ) - ], - results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message=message)], - ) - - -GATES_CASES = [ - Case( - name="unresolved requirements compose nothing", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - ), - want=_not_ready( - fn.CONDITION_REASON_WAITING_FOR_CLUSTER, - "Waiting for the gateway's cluster and the other gateways to resolve", - _requirements(), - ), - ), - Case( - name="a named cluster that does not exist", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required(clusters=[], gateways=[_gateway_xr("eu", _CLUSTER)]), - ), - want=_not_ready( - fn.CONDITION_REASON_WAITING_FOR_CLUSTER, - f"InferenceCluster {_CLUSTER} does not exist", - _requirements(), - ), - ), - Case( - name="a cluster that already hosts a lower-named gateway", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER), _gateway_xr("aaa", _CLUSTER)], - ), - ), - want=_not_ready( - fn.CONDITION_REASON_CLUSTER_TAKEN, - f"InferenceCluster {_CLUSTER} already hosts InferenceGateway aaa", - _requirements(), - ), - ), - Case( - name="a cluster with no providerConfigRef yet", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required( - clusters=[_cluster(provider_config=None)], gateways=[_gateway_xr("eu", _CLUSTER)] - ), - ), - want=_not_ready( - fn.CONDITION_REASON_WAITING_FOR_CLUSTER, - f"InferenceCluster {_CLUSTER} has not published a providerConfigRef", - _requirements(), - ), - ), -] - - -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - -@pytest.mark.parametrize("case", GATES_CASES, ids=lambda case: case.name) -def test_gates(case: Case) -> None: - """Passes where the gateway can't be composed compose nothing, and say - why. Asserting the whole response proves nothing is composed against a - cluster we can't reach, rather than a subset being applied.""" - got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) - - -def test_minimal_gateway() -> None: - """A gateway with no TLS or auth: the getting-started shape. +# The helpers below build the resources the function composes. Each is a +# provider-kubernetes Object that applies one manifest to the gateway's cluster +# through its ClusterProviderConfig, and every namespaced manifest lands in +# modelplane-system there. +# +# Each Object sets its own namespace too. An InferenceGateway is cluster-scoped, +# and Crossplane only defaults a composed namespaced resource's namespace from a +# namespaced composite. Without it every reconcile fails with "an empty +# namespace may not be set when a resource name is provided" and nothing is +# composed at all. - Composes the gateway objects and no auth policies, and reports no - endpoints until the Gateway has an address. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert sorted(got.desired.resources) == sorted( - [ - # The CA whose client certificates a cluster gateway trusts, - # published as a ClusterIssuer for compose-model-route to issue - # per-namespace client certificates from. - "client-ca-certificate", - "client-ca-issuer", - "client-ca-bundle", - "client-ca-configmap", - "client-selfsigned-issuer", - "client-traffic-policy", - "envoy-proxy", - "failover-policy", - "gateway", - "healthz-filter", - "healthz-route", - ] - ), "composes the gateway objects and its client PKI, and no caller auth" - for key, res in got.desired.resources.items(): - d = resource.struct_to_dict(res.resource) - assert d["kind"] == "Object", f"{key} targets the gateway's cluster" - assert d["spec"]["providerConfigRef"] == {"kind": "ClusterProviderConfig", "name": _PC}, ( - f"{key} uses the cluster's ClusterProviderConfig" - ) - # An InferenceGateway is cluster-scoped, and Crossplane only - # defaults a composed namespaced resource's namespace from a - # namespaced composite. Without this every reconcile fails with - # "an empty namespace may not be set when a resource name is - # provided" and nothing is composed at all. - assert d["metadata"]["namespace"] == fn.CONTROL_PLANE_NAMESPACE, ( - f"{key} sets its own namespace, which a cluster-scoped XR must" +def _envoy_proxy() -> fnv1.Resource: + """The composed EnvoyProxy, configuring the gateway's proxy pods and its usage-record access log.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "EnvoyProxy", + "metadata": {"name": "inference-gateway", "namespace": "modelplane-system"}, + "spec": { + # Two proxy pods spread softly across nodes and + # zones, a disruption budget so a drain can't + # evict both, and ndots:1. Without ndots:1 every + # backend hostname is resolved against each of + # the pod's search domains first, since they all + # have fewer than five dots. A cluster whose + # upstream resolver is slow then stalls + # resolution, and Envoy answers 503 with nothing + # but DNS timeouts to show for it. + "provider": { + "type": "Kubernetes", + "kubernetes": { + "envoyService": {"externalTrafficPolicy": "Cluster"}, + "envoyDeployment": { + "replicas": 2, + "patch": { + "type": "StrategicMerge", + "value": { + "spec": { + "template": { + "spec": { + "dnsConfig": { + "options": [{"name": "ndots", "value": "1"}] + } + } + } + } + }, + }, + "pod": { + "topologySpreadConstraints": [ + { + "maxSkew": 1, + "topologyKey": "kubernetes.io/hostname", + "whenUnsatisfiable": "ScheduleAnyway", + "labelSelector": { + "matchLabels": { + "gateway.envoyproxy.io/owning-gateway-name": "inference-gateway", + "gateway.envoyproxy.io/owning-gateway-namespace": "modelplane-system", + } + }, + }, + { + "maxSkew": 1, + "topologyKey": "topology.kubernetes.io/zone", + "whenUnsatisfiable": "ScheduleAnyway", + "labelSelector": { + "matchLabels": { + "gateway.envoyproxy.io/owning-gateway-name": "inference-gateway", + "gateway.envoyproxy.io/owning-gateway-namespace": "modelplane-system", + } + }, + }, + ] + }, + }, + "envoyPDB": {"maxUnavailable": 1}, + }, + }, + # A stopping pod drains for as long as a request + # may run by default, so a restart doesn't cut + # off streams in flight. + "shutdown": {"drainTimeout": "300s"}, + "telemetry": { + "accessLog": { + "settings": [ + { + "format": { + "type": "JSON", + # The caller and token fields + # read request metadata, not + # the response body or a + # header. The caller header + # is stripped before a + # third-party backend sees + # it, so a log reading the + # header loses the caller on + # exactly the records that + # attribute provider spend. + "json": { + "caller": "%DYNAMIC_METADATA(io.envoy.ai_gateway:caller)%", + "service": "%REQ(X-AI-EG-MODEL)%", + "endpoint": "%DYNAMIC_METADATA(io.envoy.ai_gateway:ai_service_backend_name)%", + "served_model": "%DYNAMIC_METADATA(io.envoy.ai_gateway:model_name_override)%", + "response_model": "%DYNAMIC_METADATA(io.envoy.ai_gateway:response_model)%", + "input_tokens": "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_input_token)%", + "output_tokens": "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_output_token)%", + "total_tokens": "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_total_token)%", + "status": "%RESPONSE_CODE%", + "duration_ms": "%DURATION%", + "start_time": "%START_TIME%", + }, + }, + "sinks": [{"type": "File", "file": {"path": "/dev/stdout"}}], + } + ] + } + }, + }, + } + }, + }, + } ) - manifest = d["spec"]["forProvider"]["manifest"] - if manifest["kind"] == "ClusterIssuer": - # Cluster-scoped: compose-model-route issues client certs from it - # into team namespaces, so it has no namespace of its own. - assert "namespace" not in manifest["metadata"], f"{key} is cluster-scoped, so it sets no namespace" - continue - if manifest["kind"] == "Bundle": - # A Bundle is cluster-scoped, so it has no namespace of its own. - # It picks the namespace it syncs its ConfigMap to by selector. - assert "namespace" not in manifest["metadata"], f"{key} is cluster-scoped, so it sets no namespace" - assert manifest["spec"]["target"]["namespaceSelector"] == { - "matchLabels": {"kubernetes.io/metadata.name": fn.REMOTE_NAMESPACE} - }, f"{key} syncs only to the remote namespace" - continue - assert manifest["metadata"]["namespace"] == fn.REMOTE_NAMESPACE, f"{key} lands in the remote namespace" - - # Two attempts per priority, so a retry tries another endpoint at the - # same priority before moving down. At one, a single transient failure - # on one replica would send the request to the next priority, which may - # be a paid provider. - failover = resource.struct_to_dict(got.desired.resources["failover-policy"].resource) - assert failover["spec"]["forProvider"]["manifest"]["spec"]["retry"] == { - "numAttemptsPerPriority": 2, - "numRetries": 3, - "retryOn": { - # retriable-status-codes has to be present for the status - # codes below to do anything: Envoy Gateway replaces retry_on - # wholesale with this list, and Envoy only consults - # retriable_status_codes when retry_on names it. Without it a - # provider answering 503 or 429 is never retried, which is - # the case failover exists for. - "triggers": [ - "connect-failure", - "refused-stream", - "reset", - "retriable-status-codes", - ], - # 429 so a rate-limited provider's traffic overflows to - # another endpoint rather than failing back to the caller. - "httpStatusCodes": [429, 503], - }, - } - # Panic mode defaults to 50%, above which Envoy ignores health and - # spreads traffic over every endpoint including the ejected ones. Every - # endpoint of a ModelService shares one cluster, so ejecting a whole - # priority tier usually crosses it and failover stops working. - # - # Asserted on the whole healthCheck, because panicThreshold is a sibling - # of passive rather than a field inside it, and nested wrongly the API - # server prunes it while the policy still applies. - assert failover["spec"]["forProvider"]["manifest"]["spec"]["healthCheck"] == { - "passive": { - "baseEjectionTime": "30s", - "consecutive5XxErrors": 5, - "interval": "5s", - "maxEjectionPercent": 100, - }, - "panicThreshold": 0, - } - assert failover["spec"]["forProvider"]["manifest"]["spec"]["targetRefs"] == [ - {"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": fn._GATEWAY_NAME} - ], "targets the Gateway, so it covers every ModelService's route" + ) - # AI Gateway buffers whole bodies, and Envoy Gateway's 32KiB default - # buffer limit answers 413 to a long prompt or non-streamed completion. - assert resource.struct_to_dict(got.desired.resources["client-traffic-policy"].resource)["spec"]["forProvider"][ - "manifest" - ] == { - "apiVersion": "gateway.envoyproxy.io/v1alpha1", - "kind": "ClientTrafficPolicy", - "metadata": {"name": "inference-gateway-client-traffic", "namespace": "modelplane-system"}, - "spec": { - "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": "inference-gateway"}], - "connection": {"bufferLimit": "50Mi"}, - "http2": {"initialStreamWindowSize": "16Mi", "initialConnectionWindowSize": "24Mi"}, - }, - } - # The token fields must read request metadata, not the response body or - # a header. The caller header is stripped before a third-party backend - # sees it, so a log reading the header loses the caller on exactly the - # records that attribute provider spend. - log = resource.struct_to_dict(got.desired.resources["envoy-proxy"].resource) - fields = log["spec"]["forProvider"]["manifest"]["spec"]["telemetry"]["accessLog"]["settings"][0]["format"]["json"] - # Two proxy pods spread softly across nodes and zones, a disruption - # budget so a drain can't evict both, and ndots:1. Without ndots:1 every - # backend hostname is resolved against each of the pod's search domains - # first, since they all have fewer than five dots. A cluster whose - # upstream resolver is slow then stalls resolution, and Envoy answers 503 - # with nothing but DNS timeouts to show for it. - proxy_labels = { - "gateway.envoyproxy.io/owning-gateway-name": "inference-gateway", - "gateway.envoyproxy.io/owning-gateway-namespace": "modelplane-system", - } - assert log["spec"]["forProvider"]["manifest"]["spec"]["provider"] == { - "type": "Kubernetes", - "kubernetes": { - "envoyService": {"externalTrafficPolicy": "Cluster"}, - "envoyDeployment": { - "replicas": 2, - "patch": {"type": "StrategicMerge", "value": fn._NDOTS_PATCH}, - "pod": { - "topologySpreadConstraints": [ - { - "maxSkew": 1, - "topologyKey": "kubernetes.io/hostname", - "whenUnsatisfiable": "ScheduleAnyway", - "labelSelector": {"matchLabels": proxy_labels}, - }, - { - "maxSkew": 1, - "topologyKey": "topology.kubernetes.io/zone", - "whenUnsatisfiable": "ScheduleAnyway", - "labelSelector": {"matchLabels": proxy_labels}, - }, - ] - }, - }, - "envoyPDB": {"maxUnavailable": 1}, +def _gateway(*, https: bool, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Gateway: an HTTP listener, joined by an HTTPS one terminating eu-tls-0 when https is set.""" + http_listener = { + "name": "http", + "protocol": "HTTP", + "port": 80, + "allowedRoutes": { + "namespaces": { + "from": "Selector", + "selector": {"matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}]}, + } }, } - # A stopping pod drains for as long as a request may run by default, so - # a restart doesn't cut off streams in flight. - assert log["spec"]["forProvider"]["manifest"]["spec"]["shutdown"] == {"drainTimeout": "300s"} - assert fn._NDOTS_PATCH["spec"]["template"]["spec"]["dnsConfig"]["options"] == [{"name": "ndots", "value": "1"}] - - assert fields["caller"] == "%DYNAMIC_METADATA(io.envoy.ai_gateway:caller)%" - assert fields["input_tokens"] == "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_input_token)%" - assert fields["output_tokens"] == "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_output_token)%" - - gw = resource.struct_to_dict(got.desired.resources["gateway"].resource) - manifest = gw["spec"]["forProvider"]["manifest"] - assert manifest["spec"]["listeners"] == [ - { - "name": "http", - "protocol": "HTTP", - "port": 80, - "allowedRoutes": { - "namespaces": { - "from": "Selector", - "selector": {"matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}]}, - } - }, - } - ], "one HTTP listener, no hostname, accepting routes from the mirrored namespaces" - assert manifest["spec"]["infrastructure"]["parametersRef"] == { - "group": "gateway.envoyproxy.io", - "kind": "EnvoyProxy", - "name": fn._GATEWAY_NAME, - }, "its own EnvoyProxy, not the GatewayClass's" - assert resource.struct_to_dict(got.desired.composite.resource).get("status") == {}, ( - "nothing to report until the Gateway has an address" - ) - - -def test_full_gateway() -> None: - """A gateway with TLS and auth, whose Gateway has an address. - - Checks the things a caller depends on: the HTTPS listener, the Secrets - copied to the cluster, the caller policy naming them, /healthz exempted - from that policy, and a status publishing no URLs, since a caller - reaches a TLS gateway on a DNS name only its owner knows. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr( - tls=v1alpha1.Tls(certificateRefs=[v1alpha1.CertificateRef(name="eu-tls-0")]), - auth=_api_key_auth(), - ) - ) - ), - resources={ - "gateway": _observed_gateway(_ADDRESS, ready=True), - "caller-auth": _observed_accepted(), - }, - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{ - "caller-secrets": [_secret("ml-team-keys", {"ml-team-assistant": "sk-mp-a1b2c3"})], - "tls-secret-0": [_secret("eu-tls-0", {"tls.crt": "cert", "tls.key": "key"})], - }, - ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - assert sorted(got.desired.resources) == [ - "caller-auth", - "caller-secret-ml-team-keys", - "client-ca-bundle", - "client-ca-certificate", - "client-ca-configmap", - "client-ca-issuer", - "client-selfsigned-issuer", - "client-traffic-policy", - "envoy-proxy", - "failover-policy", - "gateway", - "healthz-auth", - "healthz-filter", - "healthz-route", - "redirect-auth", - "redirect-route", - "tls-secret-eu-tls-0", - ] - - def manifest(key: str) -> dict: - return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] - - assert manifest("gateway")["spec"]["listeners"][1] == { + https_listener = { "name": "https", "protocol": "HTTPS", "port": 443, @@ -570,622 +329,2612 @@ def manifest(key: str) -> dict: } }, } - assert _to_dict(got.requirements) == _to_dict(_requirements(auth=True, tls=1)) - assert manifest("tls-secret-eu-tls-0") == { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": "eu-tls-0", "namespace": fn.REMOTE_NAMESPACE}, - "type": "kubernetes.io/tls", - "data": { - "tls.crt": base64.b64encode(b"cert").decode(), - "tls.key": base64.b64encode(b"key").decode(), - }, - }, "the certificate is copied verbatim, keeping the name the Gateway refers to it by" - assert manifest("caller-auth")["spec"]["apiKeyAuth"] == { - "credentialRefs": [{"name": "callers-ml-team-keys"}], - # Authorization for OpenAI clients, x-api-key for Anthropic ones. - "extractFrom": [{"headers": ["Authorization", "x-api-key"]}], - "forwardClientIDHeader": fn._CALLER_HEADER, - "sanitize": True, - } - assert manifest("healthz-auth")["spec"] == { - "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "HTTPRoute", "name": fn._HEALTHZ_NAME}], - "authorization": {"defaultAction": "Allow"}, - }, "/healthz overrides the Gateway-level policy so a health check needs no credential" - # Inference binds to the HTTPS listener alone, so :80 carries only - # /healthz and this catch-all redirect to it. /healthz is an Exact match, - # so it still answers a plain-HTTP health check. - assert manifest("healthz-route")["spec"]["parentRefs"] == [ - { - "group": "gateway.networking.k8s.io", - "kind": "Gateway", - "name": fn._GATEWAY_NAME, - "sectionName": "http", - } - ] - assert manifest("redirect-route")["spec"] == { - "parentRefs": [ - { - "group": "gateway.networking.k8s.io", - "kind": "Gateway", - "name": fn._GATEWAY_NAME, - "sectionName": "http", - } - ], - "rules": [ + return fnv1.Resource( + resource=resource.dict_to_struct( { - "matches": [{"path": {"type": "PathPrefix", "value": "/"}}], - "filters": [{"type": "RequestRedirect", "requestRedirect": {"scheme": "https", "statusCode": 301}}], + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status) && has(object.status.addresses) && object.status.addresses.size() > 0", + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "Gateway", + "metadata": {"name": "inference-gateway", "namespace": "modelplane-system"}, + "spec": { + "gatewayClassName": "envoy", + "infrastructure": { + "parametersRef": { + "group": "gateway.envoyproxy.io", + "kind": "EnvoyProxy", + "name": "inference-gateway", + } + }, + "listeners": [http_listener, https_listener] if https else [http_listener], + }, + } + }, + }, } - ], - } - assert manifest("redirect-auth")["spec"] == { - "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "HTTPRoute", "name": fn._REDIRECT_NAME}], - "authorization": {"defaultAction": "Allow"}, - }, "the redirect must happen before auth, or an unauthenticated caller gets 401 instead of being sent to HTTPS" - assert resource.struct_to_dict(got.desired.composite.resource)["status"] == {"address": _ADDRESS} - assert [_to_dict(c) for c in got.conditions] == [ - _to_dict( - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_TRUE, - reason=fn.CONDITION_REASON_GATEWAY_PROGRAMMED, - ) - ) - ] - - -def test_endpoints_are_built_from_the_address() -> None: - """A gateway serving plain HTTP publishes URLs on its address, which is - something a caller can actually put in an SDK's base_url.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={"gateway": _observed_gateway(_ADDRESS, ready=False)}, ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert resource.struct_to_dict(got.desired.composite.resource)["status"] == { - "address": _ADDRESS, - "endpoints": { - "openAI": f"http://{_ADDRESS}/v1", - "anthropic": f"http://{_ADDRESS}/anthropic/v1", - }, - } - assert next(iter(got.conditions)).reason == fn.CONDITION_REASON_WAITING_FOR_GATEWAY, ( - "an address alone isn't readiness; the Gateway must be programmed" + ready=ready, ) -def test_an_ipv6_address_is_bracketed_in_the_endpoints() -> None: - """A bare IPv6 literal collides with the port separator in a URL, so an - SDK given http://2001:db8::1/v1 as a base_url can't use it.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={"gateway": _observed_gateway("2001:db8::1", ready=False)}, - ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), +def _client_selfsigned_issuer() -> fnv1.Resource: + """The composed self-signed Issuer that signs the client CA.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "Issuer", + "metadata": {"name": "inference-gateway-selfsigned", "namespace": "modelplane-system"}, + "spec": {"selfSigned": {}}, + } + }, + }, + } + ) ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert resource.struct_to_dict(got.desired.composite.resource)["status"]["endpoints"] == { - "openAI": "http://[2001:db8::1]/v1", - "anthropic": "http://[2001:db8::1]/anthropic/v1", - } - -def test_resolves_each_cluster_gateway_name() -> None: - """A Service per cluster gateway, resolving its name to its address here. - A ModelService's backends address a cluster gateway by the name - compose-inference-cluster derived, and this gateway's Envoy resolves it, - so its cluster needs a Service of that name. An IP is served by a - headless Service and an EndpointSlice; a hostname, which is how a cloud - load balancer names itself, by an ExternalName Service. An IP also gets - the Endpoints the slice supersedes, because kube-dns reads only that and - is what GKE runs. A cluster that hasn't published both an address and a - name gets neither. - """ - ipv4 = "prod-ipv4-gateway-aaaaa.modelplane-system.svc.cluster.local" - ipv6 = "prod-ipv6-gateway-bbbbb.modelplane-system.svc.cluster.local" - dns = "prod-dns-gateway-ccccc.modelplane-system.svc.cluster.local" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required( - gateways=[_gateway_xr("eu", _CLUSTER)], - clusters=[ - _cluster(), # this gateway's own cluster, no gateway published yet - _cluster_with_gateway("prod-ipv4", address="203.0.113.7", hostname=ipv4), - _cluster_with_gateway("prod-ipv6", address="2001:db8::1", hostname=ipv6), - _cluster_with_gateway("prod-dns", address="lb-x.elb.amazonaws.com", hostname=dns), - ], - ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - resolvers = { - key: resource.struct_to_dict(res.resource) - for key, res in got.desired.resources.items() - if key.startswith("cluster-name") - } - for key, obj in resolvers.items(): - assert obj["spec"]["providerConfigRef"] == {"kind": "ClusterProviderConfig", "name": _PC}, ( - f"{key} is composed against this gateway's own cluster" +def _client_ca_certificate(*, common_name: str) -> fnv1.Resource: + """The composed Certificate for the client CA, whose client certificates a cluster gateway trusts.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.conditions) && " + "object.status.conditions.exists(c, c.type == 'Ready' && c.status == 'True')" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "Certificate", + "metadata": {"name": "inference-gateway-ca", "namespace": "modelplane-system"}, + "spec": { + "isCA": True, + "commonName": common_name, + "secretName": "inference-gateway-ca", + "duration": "87600h", + "renewBefore": "8760h", + "privateKey": {"algorithm": "ECDSA", "size": 256}, + "issuerRef": { + "name": "inference-gateway-selfsigned", + "kind": "Issuer", + "group": "cert-manager.io", + }, + }, + } + }, + }, + } ) - manifests = {key: obj["spec"]["forProvider"]["manifest"] for key, obj in resolvers.items()} - assert manifests == { - "cluster-name-prod-ipv4-gateway-aaaaa": { - "apiVersion": "v1", - "kind": "Service", - "metadata": {"name": "prod-ipv4-gateway-aaaaa", "namespace": fn.REMOTE_NAMESPACE}, - "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, - }, - "cluster-name-endpoints-prod-ipv4-gateway-aaaaa": { - "apiVersion": "v1", - "kind": "Endpoints", - "metadata": { - "name": "prod-ipv4-gateway-aaaaa", - "namespace": fn.REMOTE_NAMESPACE, - # Off, or the mirroring controller writes a second - # EndpointSlice over the one composed beside this. - "labels": {"endpointslice.kubernetes.io/skip-mirror": "true"}, - }, - "subsets": [{"addresses": [{"ip": "203.0.113.7"}], "ports": [{"name": "https", "port": 443}]}], - }, - "cluster-name-slice-prod-ipv4-gateway-aaaaa": { - "apiVersion": "discovery.k8s.io/v1", - "kind": "EndpointSlice", - "metadata": { - "name": "prod-ipv4-gateway-aaaaa", - "namespace": fn.REMOTE_NAMESPACE, - "labels": {"kubernetes.io/service-name": "prod-ipv4-gateway-aaaaa"}, - }, - "addressType": "IPv4", - "ports": [{"name": "https", "port": 443}], - "endpoints": [{"addresses": ["203.0.113.7"], "conditions": {"ready": True}}], - }, - "cluster-name-prod-ipv6-gateway-bbbbb": { - "apiVersion": "v1", - "kind": "Service", - "metadata": {"name": "prod-ipv6-gateway-bbbbb", "namespace": fn.REMOTE_NAMESPACE}, - "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, - }, - "cluster-name-endpoints-prod-ipv6-gateway-bbbbb": { - "apiVersion": "v1", - "kind": "Endpoints", - "metadata": { - "name": "prod-ipv6-gateway-bbbbb", - "namespace": fn.REMOTE_NAMESPACE, - # Off, or the mirroring controller writes a second - # EndpointSlice over the one composed beside this. - "labels": {"endpointslice.kubernetes.io/skip-mirror": "true"}, - }, - "subsets": [{"addresses": [{"ip": "2001:db8::1"}], "ports": [{"name": "https", "port": 443}]}], - }, - "cluster-name-slice-prod-ipv6-gateway-bbbbb": { - "apiVersion": "discovery.k8s.io/v1", - "kind": "EndpointSlice", - "metadata": { - "name": "prod-ipv6-gateway-bbbbb", - "namespace": fn.REMOTE_NAMESPACE, - "labels": {"kubernetes.io/service-name": "prod-ipv6-gateway-bbbbb"}, - }, - "addressType": "IPv6", - "ports": [{"name": "https", "port": 443}], - "endpoints": [{"addresses": ["2001:db8::1"], "conditions": {"ready": True}}], - }, - "cluster-name-prod-dns-gateway-ccccc": { - "apiVersion": "v1", - "kind": "Service", - "metadata": {"name": "prod-dns-gateway-ccccc", "namespace": fn.REMOTE_NAMESPACE}, - "spec": {"type": "ExternalName", "externalName": "lb-x.elb.amazonaws.com"}, - }, - }, ( - "IP clusters get a headless Service + EndpointSlice, the hostname cluster an ExternalName, " - "and the own cluster with nothing published gets neither" ) -def test_certificate_common_names_fit_the_x509_limit() -> None: - """A long gateway name must not push a certificate commonName past the - 64-byte X.509 limit, which cert-manager's webhook rejects. A gateway name - is a cluster-scoped resource name, so it can be up to 253 characters.""" - long_name = "g" + "a" * 62 - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(name=long_name)))), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr(long_name, _CLUSTER)]), +def _client_ca_issuer() -> fnv1.Resource: + """The composed ClusterIssuer, backed by the client CA, that compose-model-route issues client certificates from.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "ClusterIssuer", + "metadata": {"name": "inference-gateway-ca"}, + "spec": {"ca": {"secretName": "inference-gateway-ca"}}, + } + }, + }, + } + ) ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - manifest = resource.struct_to_dict(got.desired.resources["client-ca-certificate"].resource)["spec"]["forProvider"][ - "manifest" - ] - cn = manifest["spec"]["commonName"] - assert len(cn.encode()) <= 64, "client-ca-certificate commonName exceeds the 64-byte X.509 limit" -def test_a_rejected_caller_policy_is_not_ready() -> None: - """A gateway whose caller policy was rejected refuses every request with - a 500 while its Gateway still has an address. Envoy Gateway rejects the - policy when two selected Secrets share a key value, so this is reachable - by writing two Secrets.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth()))), - # The Gateway is programmed; the policy is not accepted. - resources={"gateway": _observed_gateway(_ADDRESS, ready=True)}, - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{"caller-secrets": [_secret("ml-team-keys", {"a": "sk-1"})]}, - ), +def _client_ca_bundle() -> fnv1.Resource: + """The composed trust-manager Bundle copying the client CA's certificate, without its key, into a ConfigMap.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.conditions) && " + "object.status.conditions.exists(c, c.type == 'Synced' && c.status == 'True')" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "trust.cert-manager.io/v1alpha1", + "kind": "Bundle", + "metadata": {"name": "inference-gateway-ca"}, + "spec": { + "sources": [{"secret": {"name": "inference-gateway-ca", "key": "ca.crt"}}], + "target": { + "configMap": {"key": "ca.crt"}, + "namespaceSelector": { + "matchLabels": {"kubernetes.io/metadata.name": "modelplane-system"} + }, + }, + }, + } + }, + }, + } + ) ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - cond = next(iter(got.conditions)) - assert cond.status == fnv1.STATUS_CONDITION_FALSE - assert cond.reason == fn.CONDITION_REASON_AUTH_NOT_ACCEPTED -SHARED_CALLER_KEY_CASES = [ - ( - "two Secrets share a key", - [_secret("team-a-keys", {"a": "sk-1"}), _secret("team-b-keys", {"b": "sk-1"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, - message="Caller team-b-keys/b has the same key as team-a-keys/a, so Envoy Gateway rejects the " - "caller authentication policy and every request is refused", - ), - ), - ( - "one Secret shares a key", - [_secret("team-a-keys", {"a": "sk-1", "z": "sk-1"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, - message="Caller team-a-keys/z has the same key as team-a-keys/a, so Envoy Gateway rejects the " - "caller authentication policy and every request is refused", - ), - ), - ( - # Listed out of order. Walked in name order, team-a's x is seen - # first, so team-b's x is skipped and its y repeats x's key. - # Walked as listed, team-a's x would be the skipped one. - "Secrets are walked in name order", - [_secret("team-b-keys", {"x": "sk-2", "y": "sk-1"}), _secret("team-a-keys", {"x": "sk-1"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, - message="Caller team-b-keys/y has the same key as team-a-keys/x, so Envoy Gateway rejects the " - "caller authentication policy and every request is refused", - ), - ), - ( - "a repeated caller name is skipped, whatever its key", - [_secret("team-a-keys", {"a": "sk-1"}), _secret("team-b-keys", {"a": "sk-1", "b": "sk-2"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_TRUE, - reason=fn.CONDITION_REASON_GATEWAY_PROGRAMMED, - ), - ), -] +def _client_ca_configmap() -> fnv1.Resource: + """The composed Object observing, without writing, the ConfigMap the client CA Bundle syncs.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "managementPolicies": ["Observe"], + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "inference-gateway-ca", "namespace": "modelplane-system"}, + } + }, + }, + } + ) + ) -@pytest.mark.parametrize("case", SHARED_CALLER_KEY_CASES, ids=lambda case: case[0]) -def test_a_shared_caller_key_is_not_ready_before_the_policy_is_observed( - case: tuple[str, list[dict], fnv1.Condition], -) -> None: - """Envoy Gateway rejects the caller policy when two callers share a key, - but the policy's Object still reads as accepted until provider-kubernetes - next observes it. The gateway reports the outage from the Secrets - themselves, and skips a repeated caller name before comparing its key, - as Envoy Gateway does.""" - _, secrets, want = case - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth()))), - resources={ - "gateway": _observed_gateway(_ADDRESS, ready=True), - "caller-auth": _observed_accepted(), - }, - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{"caller-secrets": secrets}, - ), +def _caller_secret(*, name: str, data: dict[str, str]) -> fnv1.Resource: + """A composed copy of a caller-key Secret, with its data copied verbatim.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": name, "namespace": "modelplane-system"}, + "type": "Opaque", + "data": data, + } + }, + }, + } + ) ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert [_to_dict(c) for c in got.conditions] == [_to_dict(want)] -def test_caller_secrets_are_listed_in_name_order() -> None: - """Envoy Gateway keeps the first Secret listed when two hold the same - caller name, so the policy lists them by name rather than in the order - they resolved in, and the winner doesn't change between reconciles.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth())))), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{ - "caller-secrets": [ - _secret("team-b-keys", {"b": "sk-2"}), - _secret("team-a-keys", {"a": "sk-1"}), - ] - }, +def _caller_auth(*, credential_refs: list[dict], ready: fnv1.Ready) -> fnv1.Resource: + """The composed SecurityPolicy authenticating callers by API key, against the Secrets credential_refs names.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "SecurityPolicy", + "metadata": {"name": "inference-gateway-callers", "namespace": "modelplane-system"}, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + } + ], + "apiKeyAuth": { + "credentialRefs": credential_refs, + # Authorization for OpenAI clients, x-api-key + # for Anthropic ones. + "extractFrom": [{"headers": ["Authorization", "x-api-key"]}], + "forwardClientIDHeader": "x-modelplane-caller", + "sanitize": True, + }, + }, + } + }, + }, + } ), + ready=ready, ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - policy = resource.struct_to_dict(got.desired.resources["caller-auth"].resource) - assert policy["spec"]["forProvider"]["manifest"]["spec"]["apiKeyAuth"]["credentialRefs"] == [ - {"name": "callers-team-a-keys"}, - {"name": "callers-team-b-keys"}, - ] -MISSING_CALLER_SECRET_CASES = [ - ( - "the selector matches no Secret", - {"caller-secrets": []}, - "spec.auth.apiKey.secretSelector matches no Secret, so no caller could authenticate", - ), - ( - "the caller Secrets have not resolved yet", - {}, - "Waiting for caller key Secrets to resolve", - ), -] +def _failover_policy() -> fnv1.Resource: + """The composed BackendTrafficPolicy that retries a failed request and ejects an endpoint that keeps failing.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "BackendTrafficPolicy", + "metadata": {"name": "inference-gateway-failover", "namespace": "modelplane-system"}, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + } + ], + "retry": { + # Two attempts per priority, so a retry tries + # another endpoint at the same priority + # before moving down. At one, a single + # transient failure on one replica would + # send the request to the next priority, + # which may be a paid provider. + "numAttemptsPerPriority": 2, + "numRetries": 3, + "retryOn": { + # retriable-status-codes has to be + # present for the status codes below to + # do anything: Envoy Gateway replaces + # retry_on wholesale with this list, and + # Envoy only consults + # retriable_status_codes when retry_on + # names it. Without it a provider + # answering 503 or 429 is never retried, + # which is the case failover exists for. + "triggers": [ + "connect-failure", + "refused-stream", + "reset", + "retriable-status-codes", + ], + # 429 so a rate-limited provider's + # traffic overflows to another endpoint + # rather than failing back to the caller. + "httpStatusCodes": [429, 503], + }, + }, + # Panic mode defaults to 50%, above which Envoy + # ignores health and spreads traffic over every + # endpoint including the ejected ones. Every + # endpoint of a ModelService shares one cluster, + # so ejecting a whole priority tier usually + # crosses it and failover stops working. + # panicThreshold is a sibling of passive rather + # than a field inside it: nested wrongly, the API + # server prunes it while the policy still + # applies. + "healthCheck": { + "passive": { + "baseEjectionTime": "30s", + "consecutive5XxErrors": 5, + "interval": "5s", + "maxEjectionPercent": 100, + }, + "panicThreshold": 0, + }, + }, + } + }, + }, + } + ) + ) -@pytest.mark.parametrize("case", MISSING_CALLER_SECRET_CASES, ids=lambda case: case[0]) -def test_a_missing_caller_secret_denies_but_keeps_the_gateway(case: tuple[str, dict, str]) -> None: - """Auth is asked for but no caller Secret has resolved. The Gateway is - still composed, so its load balancer and address survive, and its caller - policy denies every request rather than leaving the door open. Two states - reach this, the selector matching no Secret and the requirement not having - resolved yet, differing only in the reason reported.""" - _, extra, want_message = case - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth())))), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)], **extra), +def _client_traffic_policy() -> fnv1.Resource: + """The composed ClientTrafficPolicy that lets a whole request or response body fit in the proxy's buffer.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "ClientTrafficPolicy", + "metadata": {"name": "inference-gateway-client-traffic", "namespace": "modelplane-system"}, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + } + ], + "connection": {"bufferLimit": "50Mi"}, + "http2": {"initialStreamWindowSize": "16Mi", "initialConnectionWindowSize": "24Mi"}, + }, + } + }, + }, + } + ) ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert "gateway" in got.desired.resources, "the Gateway is kept, so its address survives" - assert not any(key.startswith("caller-secret-") for key in got.desired.resources), ( - "no caller Secret resolved, so none is copied to the cluster" + +def _healthz_filter() -> fnv1.Resource: + """The composed HTTPRouteFilter answering /healthz with a 200.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "HTTPRouteFilter", + "metadata": {"name": "inference-gateway-healthz", "namespace": "modelplane-system"}, + "spec": { + "directResponse": { + "statusCode": 200, + "contentType": "application/json", + "body": {"type": "Inline", "inline": '{"status":"ok"}'}, + } + }, + } + }, + }, + } + ) ) - spec = resource.struct_to_dict(got.desired.resources["caller-auth"].resource)["spec"]["forProvider"]["manifest"][ - "spec" - ] - assert spec == { - "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": fn._GATEWAY_NAME}], - "authorization": {"defaultAction": "Deny"}, - }, "with no caller key the policy denies every request rather than authenticating nobody by omission" - cond = next(iter(got.conditions)) - assert cond.status == fnv1.STATUS_CONDITION_FALSE - assert cond.reason == fn.CONDITION_REASON_SECRETS_MISSING - assert cond.message == want_message -def test_a_missing_tls_secret_keeps_the_gateway() -> None: - """A referenced TLS Secret hasn't resolved. The Gateway is still composed, - so its address survives; the HTTPS listener is left without a certificate - on the cluster until the Secret appears, rather than the whole Gateway - withdrawn and its load balancer moved.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr(tls=v1alpha1.Tls(certificateRefs=[v1alpha1.CertificateRef(name="eu-tls-0")])) - ) - ) - ), - required_resources=_required( - clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)], **{"tls-secret-0": []} - ), +def _healthz_route() -> fnv1.Resource: + """The composed HTTPRoute serving /healthz on the HTTP listener, as an Exact match that outranks any redirect.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "HTTPRoute", + "metadata": {"name": "inference-gateway-healthz", "namespace": "modelplane-system"}, + "spec": { + "parentRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + "sectionName": "http", + } + ], + "rules": [ + { + "matches": [{"path": {"type": "Exact", "value": "/healthz"}}], + "filters": [ + { + "type": "ExtensionRef", + "extensionRef": { + "group": "gateway.envoyproxy.io", + "kind": "HTTPRouteFilter", + "name": "inference-gateway-healthz", + }, + } + ], + } + ], + }, + } + }, + }, + } + ) ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - assert "gateway" in got.desired.resources, "the Gateway is kept, so its address survives" - assert "tls-secret-eu-tls-0" not in got.desired.resources, "the missing Secret isn't copied to the cluster" - listeners = resource.struct_to_dict(got.desired.resources["gateway"].resource)["spec"]["forProvider"]["manifest"][ - "spec" - ]["listeners"] - assert [ln["name"] for ln in listeners] == ["http", "https"], "the HTTPS listener is still declared" - cond = next(iter(got.conditions)) - assert cond.status == fnv1.STATUS_CONDITION_FALSE - assert cond.reason == fn.CONDITION_REASON_SECRETS_MISSING - assert cond.message == "Waiting for TLS Secrets: eu-tls-0" -def test_the_incumbent_keeps_its_cluster() -> None: - """A gateway created later must not take a cluster off one already - serving traffic. Doing so would delete the incumbent's Gateway and bring - its load balancer back on a different address.""" - # "aaa" sorts before "zzz" but "zzz" already has an address. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceGateway", - "metadata": {"name": "aaa"}, - "spec": {"clusterName": _CLUSTER}, - } - ) - ) - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[ - _gateway_xr("aaa", _CLUSTER), - {**_gateway_xr("zzz", _CLUSTER), "status": {"address": _ADDRESS}}, - ], - ), +def _healthz_auth() -> fnv1.Resource: + """The composed SecurityPolicy letting /healthz past caller auth, so a health check needs no credential.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "SecurityPolicy", + "metadata": {"name": "inference-gateway-healthz-open", "namespace": "modelplane-system"}, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "HTTPRoute", + "name": "inference-gateway-healthz", + } + ], + "authorization": {"defaultAction": "Allow"}, + }, + } + }, + }, + } + ) ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert len(got.desired.resources) == 0, "the newcomer composes nothing" - cond = next(iter(got.conditions)) - assert cond.reason == fn.CONDITION_REASON_CLUSTER_TAKEN - assert "zzz" in cond.message -def test_no_composed_object_observes_a_secret() -> None: - """No composed Object reads a Secret, which is what keeps this gateway's - client CA private key off the control plane. +def _redirect_route() -> fnv1.Resource: + """The composed HTTPRoute redirecting everything on :80 but /healthz to HTTPS.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "HTTPRoute", + "metadata": {"name": "inference-gateway-redirect", "namespace": "modelplane-system"}, + "spec": { + "parentRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + "sectionName": "http", + } + ], + "rules": [ + { + "matches": [{"path": {"type": "PathPrefix", "value": "/"}}], + "filters": [ + { + "type": "RequestRedirect", + "requestRedirect": {"scheme": "https", "statusCode": 301}, + } + ], + } + ], + }, + } + }, + }, + } + ) + ) + - provider-kubernetes copies an observed object's whole manifest into the - Object's status, and its --sanitize-secrets flag defaults to false, so - observing a Secret publishes every key in it to anyone who can get - objects. This CA signs the certificate every cluster gateway in the fleet - accepts, so leaking its key means anyone can reach any engine. +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - Asserted over everything composed rather than over the PKI, because the - cost of reintroducing this anywhere is the same. - Observing is the case that matters here. The Secrets this function - *writes* also end up in status, because provider-kubernetes reports what - it observes of what it manages, so this alone doesn't keep their contents - off the control plane. Those hold caller keys and serving certificates - that came from control-plane Secrets to begin with, so the exposure is a - wider audience for data already present rather than data that would - otherwise never be there, and prerequisites.yaml runs - provider-kubernetes with --sanitize-secrets to redact it. A CA private - key is different in kind: it is generated on the workload cluster and - observing it is the only way it could ever reach the control plane. - """ - # Auth and TLS both on, so the Secret-copying path is exercised: without - # them this function composes no Secret at all and the assertion holds - # vacuously. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr( - tls={"certificateRefs": [{"name": "eu-tls-0"}]}, - auth={"method": "APIKey", "apiKey": {"secretSelector": {"matchLabels": {"team": "ml"}}}}, - ) +# Secret data is base64 encoded, as the API server stores it: c2stMQ== is +# "sk-1", c2stMg== "sk-2", c2stbXAtYTFiMmMz "sk-mp-a1b2c3", Y2VydA== "cert" and +# a2V5 "key". +COMPOSE_CASES = [ + # Passes where the gateway can't be composed compose nothing, and say why. + # The whole response shows nothing is composed against a cluster the + # gateway can't reach or doesn't own, rather than a subset being applied. + Case( + name="RequirementsUnresolved", + reason="Before its cluster and the other gateways resolve, a gateway composes nothing and waits for them.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=False, caller_secret_labels=None)), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_xr(status=None, ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for the gateway's cluster and the other gateways to resolve", ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + message="Waiting for the gateway's cluster and the other gateways to resolve", + ) + ], ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{ - "caller-secrets": [_secret("ml-team-keys", {"alice": "key"})], - "tls-secret-0": [_secret("eu-tls-0", {"tls.crt": "cert", "tls.key": "key"})], + ), + Case( + name="ClusterNotFound", + reason="A gateway whose named InferenceCluster doesn't exist composes nothing and says so.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=False, caller_secret_labels=None)), + required_resources={ + "clusters": fnv1.Resources(), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), }, ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - composed_secrets = [] - observed_secrets = [] - for key, res in got.desired.resources.items(): - d = resource.struct_to_dict(res.resource) - manifest = d["spec"]["forProvider"]["manifest"] - if manifest["kind"] != "Secret": - continue - composed_secrets.append(key) - if "Observe" in d["spec"].get("managementPolicies", []): - observed_secrets.append(key) - assert observed_secrets == [], "these observe a Secret, so its private keys reach the control plane" - assert composed_secrets != [], "no Secret composed, so the assertion above proves nothing" - - -def test_client_pki_publishes_the_ca_without_its_key() -> None: - """The client CA's certificate reaches the control plane through a - trust-manager Bundle, which copies one named key into a ConfigMap, rather - than through the Secret that also holds the private key.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - def manifest(key: str) -> dict: - return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] - - assert manifest("client-ca-bundle") == { - "apiVersion": "trust.cert-manager.io/v1alpha1", - "kind": "Bundle", - "metadata": {"name": "inference-gateway-ca"}, - "spec": { - "sources": [{"secret": {"name": "inference-gateway-ca", "key": "ca.crt"}}], - "target": { - "configMap": {"key": "ca.crt"}, - "namespaceSelector": {"matchLabels": {"kubernetes.io/metadata.name": "modelplane-system"}}, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_xr(status=None, ready=fnv1.READY_FALSE)), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="InferenceCluster gw-eu does not exist")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + message="InferenceCluster gw-eu does not exist", + ) + ], + ), + ), + Case( + name="LowerNamedGateway", + reason=( + "A gateway whose cluster a lower-named gateway also targets, with neither yet holding an address " + "or a creation timestamp, composes nothing and reports the cluster taken." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=False, caller_secret_labels=None)), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources( + items=[_inference_gateway(name="eu", address=None), _inference_gateway(name="aaa", address=None)] + ), }, - }, - } - # Named after the Bundle, because that's the ConfigMap a Bundle syncs. - assert manifest("client-ca-configmap") == { - "apiVersion": "v1", - "kind": "ConfigMap", - "metadata": {"name": "inference-gateway-ca", "namespace": "modelplane-system"}, - } - assert resource.struct_to_dict(got.desired.resources["client-ca-configmap"].resource)["spec"][ - "managementPolicies" - ] == ["Observe"], "trust-manager owns this ConfigMap; Crossplane must not write it" - - -def test_client_ca_published_from_the_observed_configmap() -> None: - """status.clientCACertificate comes from the ConfigMap trust-manager - syncs, as plain text rather than base64. A cluster only trusts this - gateway once it has it, so nothing reaches an engine before it appears. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={ - "gateway": _observed_gateway("gw.example.org", ready=True), - "client-ca-configmap": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "status": { - "atProvider": { - "manifest": { - "apiVersion": "v1", - "kind": "ConfigMap", - "data": {"ca.crt": "-----BEGIN CERTIFICATE-----\nclient\n"}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_xr(status=None, ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="InferenceCluster gw-eu already hosts InferenceGateway aaa" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ClusterAlreadyHasGateway", + message="InferenceCluster gw-eu already hosts InferenceGateway aaa", + ) + ], + ), + ), + # Taking the cluster would delete the incumbent's Gateway and bring its load + # balancer back on a different address. + Case( + name="IncumbentHasAddress", + reason=( + "A gateway named aaa composes nothing on a cluster where zzz already has an address, " + "though aaa sorts first, so the incumbent keeps its cluster." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="aaa", tls=False, caller_secret_labels=None)), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources( + items=[ + _inference_gateway(name="aaa", address=None), + _inference_gateway(name="zzz", address="34.56.129.3"), + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_xr(status=None, ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="InferenceCluster gw-eu already hosts InferenceGateway zzz" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ClusterAlreadyHasGateway", + message="InferenceCluster gw-eu already hosts InferenceGateway zzz", + ) + ], + ), + ), + Case( + name="NoProviderConfigRef", + reason="A gateway whose cluster hasn't published a providerConfigRef yet composes nothing and waits for it.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=False, caller_secret_labels=None)), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=False)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_xr(status=None, ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="InferenceCluster gw-eu has not published a providerConfigRef", + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + message="InferenceCluster gw-eu has not published a providerConfigRef", + ) + ], + ), + ), + # The getting-started shape. + Case( + name="NoTLSOrAuth", + reason=( + "With no TLS or auth, a gateway composes its Gateway objects and client PKI but no caller auth or " + "redirect, and reports nothing in status until the Gateway has an address." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=False, caller_secret_labels=None)), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="Waiting for the Gateway on cluster gw-eu to be programmed" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="Waiting for the Gateway on cluster gw-eu to be programmed", + ) + ], + ), + ), + # The URLs are what a caller can actually put in an SDK's base_url. + Case( + name="GatewayHasAddress", + reason=( + "A plain HTTP gateway whose Gateway has an address but isn't programmed publishes the address and the " + "URLs built on it, and isn't Ready." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels=None), + resources={"gateway": _observed_gateway(address="34.56.129.3", ready=False)}, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "34.56.129.3", + "endpoints": { + "openAI": "http://34.56.129.3/v1", + "anthropic": "http://34.56.129.3/anthropic/v1", + }, + }, + ready=fnv1.READY_FALSE, + ), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="Waiting for the Gateway on cluster gw-eu to be programmed" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="Waiting for the Gateway on cluster gw-eu to be programmed", + ) + ], + ), + ), + # A bare IPv6 literal collides with the port separator in a URL, so an SDK + # given http://2001:db8::1/v1 as a base_url can't use it. + Case( + name="IPv6Address", + reason="A Gateway with an IPv6 address has it bracketed in the URLs the gateway publishes.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels=None), + resources={"gateway": _observed_gateway(address="2001:db8::1", ready=False)}, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "2001:db8::1", + "endpoints": { + "openAI": "http://[2001:db8::1]/v1", + "anthropic": "http://[2001:db8::1]/anthropic/v1", + }, + }, + ready=fnv1.READY_FALSE, + ), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="Waiting for the Gateway on cluster gw-eu to be programmed" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="Waiting for the Gateway on cluster gw-eu to be programmed", + ) + ], + ), + ), + # cert-manager's webhook rejects a commonName past X.509's 64-byte limit, + # and a gateway name is a cluster-scoped resource name, so it can be up to + # 253 characters. + Case( + name="LongGatewayName", + reason="A gateway with a 63-character name has its client CA's commonName cut to 64 bytes.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + name="gaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + tls=False, + caller_secret_labels=None, + ) + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="gaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", address=None + ) + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate( + common_name="Modelplane InferenceGateway CA gaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + ), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="Waiting for the Gateway on cluster gw-eu to be programmed" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="Waiting for the Gateway on cluster gw-eu to be programmed", + ) + ], + ), + ), + # A cluster only trusts this gateway once it has the CA, so nothing reaches + # an engine before it appears. + Case( + name="ClientCAObserved", + reason=( + "A programmed gateway that observes trust-manager's ConfigMap publishes the client CA from it in " + "status, as plain text." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels=None), + resources={ + "gateway": _observed_gateway(address="gw.example.org", ready=True), + "client-ca-configmap": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "status": { + "atProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "data": {"ca.crt": "-----BEGIN CERTIFICATE-----\nclient\n"}, + } } - } - }, - } + }, + } + ), ), - ), + }, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), }, ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - assert ( - resource.struct_to_dict(got.desired.composite.resource)["status"]["clientCACertificate"] - == "-----BEGIN CERTIFICATE-----\nclient\n" - ) - - -def test_no_client_ca_before_the_bundle_syncs() -> None: - """With no observed ConfigMap the gateway publishes no CA, so no cluster - trusts it yet and no cluster publishes a hostname on its account.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={"gateway": _observed_gateway("gw.example.org", ready=True)}, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "gw.example.org", + "clientCACertificate": "-----BEGIN CERTIFICATE-----\nclient\n", + "endpoints": { + "openAI": "http://gw.example.org/v1", + "anthropic": "http://gw.example.org/anthropic/v1", + }, + }, + ready=fnv1.READY_UNSPECIFIED, + ), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition(type="GatewayReady", status=fnv1.STATUS_CONDITION_TRUE, reason="GatewayProgrammed"), + ], ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - assert "clientCACertificate" not in resource.struct_to_dict(got.desired.composite.resource)["status"] + ), + # With no CA published no cluster trusts the gateway yet, so no cluster + # publishes a hostname on its account. + Case( + name="ClientCANotObserved", + reason="A programmed gateway that hasn't observed trust-manager's ConfigMap publishes no client CA.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels=None), + resources={"gateway": _observed_gateway(address="gw.example.org", ready=True)}, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "gw.example.org", + "endpoints": { + "openAI": "http://gw.example.org/v1", + "anthropic": "http://gw.example.org/anthropic/v1", + }, + }, + ready=fnv1.READY_UNSPECIFIED, + ), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition(type="GatewayReady", status=fnv1.STATUS_CONDITION_TRUE, reason="GatewayProgrammed"), + ], + ), + ), + # A ModelService's backends address a cluster gateway by the name + # compose-inference-cluster derived, and this gateway's Envoy resolves it, + # so its cluster needs a Service of that name. A hostname is how a cloud + # load balancer names itself. An IP gets the Endpoints the EndpointSlice + # supersedes as well, because kube-dns reads only that and is what GKE runs. + Case( + name="ClusterGatewaysPublished", + reason=( + "Each cluster publishing a gateway address and name gets a Service of that name on this gateway's " + "cluster, headless with Endpoints and an EndpointSlice for an IP or ExternalName for a hostname, " + "while this gateway's own cluster, publishing neither, gets none." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=False, caller_secret_labels=None)), + required_resources={ + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "clusters": fnv1.Resources( + items=[ + _cluster(provider_config_ref=True), + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "prod-ipv4"}, + "spec": { + "cluster": { + "source": "Existing", + "existing": { + "secretRef": {"name": "prod-ipv4-kubeconfig", "key": "kubeconfig"} + }, + } + }, + "status": { + "gateway": { + "address": "203.0.113.7", + "hostname": "prod-ipv4-gateway-aaaaa.modelplane-system.svc.cluster.local", + } + }, + } + ) + ), + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "prod-ipv6"}, + "spec": { + "cluster": { + "source": "Existing", + "existing": { + "secretRef": {"name": "prod-ipv6-kubeconfig", "key": "kubeconfig"} + }, + } + }, + "status": { + "gateway": { + "address": "2001:db8::1", + "hostname": "prod-ipv6-gateway-bbbbb.modelplane-system.svc.cluster.local", + } + }, + } + ) + ), + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "prod-dns"}, + "spec": { + "cluster": { + "source": "Existing", + "existing": { + "secretRef": {"name": "prod-dns-kubeconfig", "key": "kubeconfig"} + }, + } + }, + "status": { + "gateway": { + "address": "lb-x.elb.amazonaws.com", + "hostname": "prod-dns-gateway-ccccc.modelplane-system.svc.cluster.local", + } + }, + } + ) + ), + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "cluster-name-prod-ipv4-gateway-aaaaa": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Service", + "metadata": { + "name": "prod-ipv4-gateway-aaaaa", + "namespace": "modelplane-system", + }, + "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, + } + }, + }, + } + ) + ), + "cluster-name-endpoints-prod-ipv4-gateway-aaaaa": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Endpoints", + "metadata": { + "name": "prod-ipv4-gateway-aaaaa", + "namespace": "modelplane-system", + # Off, or the mirroring controller writes a second + # EndpointSlice over the one composed beside this. + "labels": {"endpointslice.kubernetes.io/skip-mirror": "true"}, + }, + "subsets": [ + { + "addresses": [{"ip": "203.0.113.7"}], + "ports": [{"name": "https", "port": 443}], + } + ], + } + }, + }, + } + ) + ), + "cluster-name-slice-prod-ipv4-gateway-aaaaa": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "discovery.k8s.io/v1", + "kind": "EndpointSlice", + "metadata": { + "name": "prod-ipv4-gateway-aaaaa", + "namespace": "modelplane-system", + "labels": {"kubernetes.io/service-name": "prod-ipv4-gateway-aaaaa"}, + }, + "addressType": "IPv4", + "ports": [{"name": "https", "port": 443}], + "endpoints": [ + {"addresses": ["203.0.113.7"], "conditions": {"ready": True}} + ], + } + }, + }, + } + ) + ), + "cluster-name-prod-ipv6-gateway-bbbbb": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Service", + "metadata": { + "name": "prod-ipv6-gateway-bbbbb", + "namespace": "modelplane-system", + }, + "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, + } + }, + }, + } + ) + ), + "cluster-name-endpoints-prod-ipv6-gateway-bbbbb": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Endpoints", + "metadata": { + "name": "prod-ipv6-gateway-bbbbb", + "namespace": "modelplane-system", + # Off, or the mirroring controller writes a second + # EndpointSlice over the one composed beside this. + "labels": {"endpointslice.kubernetes.io/skip-mirror": "true"}, + }, + "subsets": [ + { + "addresses": [{"ip": "2001:db8::1"}], + "ports": [{"name": "https", "port": 443}], + } + ], + } + }, + }, + } + ) + ), + "cluster-name-slice-prod-ipv6-gateway-bbbbb": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "discovery.k8s.io/v1", + "kind": "EndpointSlice", + "metadata": { + "name": "prod-ipv6-gateway-bbbbb", + "namespace": "modelplane-system", + "labels": {"kubernetes.io/service-name": "prod-ipv6-gateway-bbbbb"}, + }, + "addressType": "IPv6", + "ports": [{"name": "https", "port": 443}], + "endpoints": [ + {"addresses": ["2001:db8::1"], "conditions": {"ready": True}} + ], + } + }, + }, + } + ) + ), + "cluster-name-prod-dns-gateway-ccccc": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Service", + "metadata": { + "name": "prod-dns-gateway-ccccc", + "namespace": "modelplane-system", + }, + "spec": {"type": "ExternalName", "externalName": "lb-x.elb.amazonaws.com"}, + } + }, + }, + } + ) + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="Waiting for the Gateway on cluster gw-eu to be programmed" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="Waiting for the Gateway on cluster gw-eu to be programmed", + ) + ], + ), + ), + # Keeping the Gateway keeps its address. The listener is left without a + # certificate on the cluster until the Secret appears, rather than the whole + # Gateway withdrawn and its load balancer moved. + Case( + name="TLSSecretMissing", + reason=( + "A gateway whose TLS Secret doesn't exist still composes its HTTPS Gateway and redirect, and reports " + "the Secret missing." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=True, caller_secret_labels=None)), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "tls-secret-0": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=True, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "redirect-route": _redirect_route(), + }, + ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for TLS Secrets: eu-tls-0")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "tls-secret-0": fnv1.ResourceSelector( + api_version="v1", kind="Secret", namespace="modelplane-system", match_name="eu-tls-0" + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="SecretsMissing", + message="Waiting for TLS Secrets: eu-tls-0", + ) + ], + ), + ), + # A caller depends on the HTTPS listener, the Secrets copied to the + # cluster, the caller policy naming them, and /healthz and the :80 redirect + # exempted from that policy. Status publishes no URLs because a caller + # reaches a TLS gateway on a DNS name only its owner knows. The certificate + # is copied verbatim, keeping the name the Gateway refers to it by. The + # redirect is exempted because it must happen before auth, or an + # unauthenticated caller would get a 401 instead of being sent to HTTPS. + Case( + name="TLSAndAuth", + reason=( + "A programmed gateway with TLS and auth serves authenticated HTTPS, copies its Secrets to the " + "cluster, and publishes its address but no URLs." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=True, caller_secret_labels={"modelplane.ai/inference-keys": "true"}), + resources={ + "gateway": _observed_gateway(address="34.56.129.3", ready=True), + "caller-auth": _observed_caller_auth(), + }, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[_caller_key_secret(name="ml-team-keys", data={"ml-team-assistant": "c2stbXAtYTFiMmMz"})] + ), + "tls-secret-0": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "eu-tls-0", "namespace": "modelplane-system"}, + "data": {"tls.crt": "Y2VydA==", "tls.key": "a2V5"}, + } + ) + ) + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={"address": "34.56.129.3"}, ready=fnv1.READY_UNSPECIFIED), + resources={ + "caller-secret-ml-team-keys": _caller_secret( + name="callers-ml-team-keys", data={"ml-team-assistant": "c2stbXAtYTFiMmMz"} + ), + "tls-secret-eu-tls-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "eu-tls-0", "namespace": "modelplane-system"}, + "type": "kubernetes.io/tls", + "data": {"tls.crt": "Y2VydA==", "tls.key": "a2V5"}, + } + }, + }, + } + ) + ), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=True, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-ml-team-keys"}], ready=fnv1.READY_TRUE + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + "redirect-route": _redirect_route(), + "redirect-auth": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "SecurityPolicy", + "metadata": { + "name": "inference-gateway-redirect-open", + "namespace": "modelplane-system", + }, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "HTTPRoute", + "name": "inference-gateway-redirect", + } + ], + "authorization": {"defaultAction": "Allow"}, + }, + } + }, + }, + } + ) + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + "tls-secret-0": fnv1.ResourceSelector( + api_version="v1", kind="Secret", namespace="modelplane-system", match_name="eu-tls-0" + ), + } + ), + conditions=[ + fnv1.Condition(type="GatewayReady", status=fnv1.STATUS_CONDITION_TRUE, reason="GatewayProgrammed"), + ], + ), + ), + # A gateway whose caller policy was rejected refuses every request with a + # 500 while its Gateway still has an address. Envoy Gateway rejects the + # policy when two selected Secrets share a key value, so this is reachable + # by writing two Secrets. The function treats a policy whose Object hasn't + # been observed the same as a rejected one. + Case( + name="CallerPolicyNotObserved", + reason=( + "A programmed gateway whose caller policy's Object hasn't been observed publishes its URLs but " + "reports the policy not accepted and isn't Ready." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}), + resources={"gateway": _observed_gateway(address="34.56.129.3", ready=True)}, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[_caller_key_secret(name="ml-team-keys", data={"a": "c2stMQ=="})] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "34.56.129.3", + "endpoints": { + "openAI": "http://34.56.129.3/v1", + "anthropic": "http://34.56.129.3/anthropic/v1", + }, + }, + ready=fnv1.READY_FALSE, + ), + resources={ + "caller-secret-ml-team-keys": _caller_secret(name="callers-ml-team-keys", data={"a": "c2stMQ=="}), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-ml-team-keys"}], ready=fnv1.READY_UNSPECIFIED + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="The gateway's caller authentication policy has not been accepted, so every request is refused", + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="CallerAuthNotAccepted", + message="The gateway's caller authentication policy has not been accepted, so every request is refused", + ) + ], + ), + ), + # Envoy Gateway rejects the caller policy when two callers share a key, but + # the policy's Object still reads as accepted until provider-kubernetes + # next observes it. The gateway reports the outage from the Secrets + # themselves, and skips a repeated caller name before comparing its key, as + # Envoy Gateway does. These four cases observe the policy as accepted. + Case( + name="KeySharedAcrossSecrets", + reason="Two caller Secrets holding the same key leave the gateway not Ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}), + resources={ + "gateway": _observed_gateway(address="34.56.129.3", ready=True), + "caller-auth": _observed_caller_auth(), + }, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[ + _caller_key_secret(name="team-a-keys", data={"a": "c2stMQ=="}), + _caller_key_secret(name="team-b-keys", data={"b": "c2stMQ=="}), + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "34.56.129.3", + "endpoints": { + "openAI": "http://34.56.129.3/v1", + "anthropic": "http://34.56.129.3/anthropic/v1", + }, + }, + ready=fnv1.READY_FALSE, + ), + resources={ + "caller-secret-team-a-keys": _caller_secret(name="callers-team-a-keys", data={"a": "c2stMQ=="}), + "caller-secret-team-b-keys": _caller_secret(name="callers-team-b-keys", data={"b": "c2stMQ=="}), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-team-a-keys"}, {"name": "callers-team-b-keys"}], + ready=fnv1.READY_TRUE, + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Caller team-b-keys/b has the same key as team-a-keys/a, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="CallerAuthNotAccepted", + message="Caller team-b-keys/b has the same key as team-a-keys/a, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ) + ], + ), + ), + Case( + name="KeyRepeatedInSecret", + reason="One caller Secret holding the same key for two callers leaves the gateway not Ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}), + resources={ + "gateway": _observed_gateway(address="34.56.129.3", ready=True), + "caller-auth": _observed_caller_auth(), + }, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[_caller_key_secret(name="team-a-keys", data={"a": "c2stMQ==", "z": "c2stMQ=="})] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "34.56.129.3", + "endpoints": { + "openAI": "http://34.56.129.3/v1", + "anthropic": "http://34.56.129.3/anthropic/v1", + }, + }, + ready=fnv1.READY_FALSE, + ), + resources={ + "caller-secret-team-a-keys": _caller_secret( + name="callers-team-a-keys", data={"a": "c2stMQ==", "z": "c2stMQ=="} + ), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-team-a-keys"}], ready=fnv1.READY_TRUE + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Caller team-a-keys/z has the same key as team-a-keys/a, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="CallerAuthNotAccepted", + message="Caller team-a-keys/z has the same key as team-a-keys/a, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ) + ], + ), + ), + # Walked as listed, team-a's x would be the skipped one, and no key would + # repeat. + Case( + name="NameOrderRepeatsKey", + reason=( + "Caller Secrets listed out of name order are walked in name order, so team-b's repeated caller x is " + "skipped, its y repeats team-a's x key, and the gateway isn't Ready." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}), + resources={ + "gateway": _observed_gateway(address="34.56.129.3", ready=True), + "caller-auth": _observed_caller_auth(), + }, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[ + _caller_key_secret(name="team-b-keys", data={"x": "c2stMg==", "y": "c2stMQ=="}), + _caller_key_secret(name="team-a-keys", data={"x": "c2stMQ=="}), + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "34.56.129.3", + "endpoints": { + "openAI": "http://34.56.129.3/v1", + "anthropic": "http://34.56.129.3/anthropic/v1", + }, + }, + ready=fnv1.READY_FALSE, + ), + resources={ + "caller-secret-team-a-keys": _caller_secret(name="callers-team-a-keys", data={"x": "c2stMQ=="}), + "caller-secret-team-b-keys": _caller_secret( + name="callers-team-b-keys", data={"x": "c2stMg==", "y": "c2stMQ=="} + ), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-team-a-keys"}, {"name": "callers-team-b-keys"}], + ready=fnv1.READY_TRUE, + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Caller team-b-keys/y has the same key as team-a-keys/x, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="CallerAuthNotAccepted", + message="Caller team-b-keys/y has the same key as team-a-keys/x, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ) + ], + ), + ), + Case( + name="CallerNameRepeated", + reason=( + "A caller name repeated in a later Secret is skipped before its key is compared, so the gateway is Ready." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}), + resources={ + "gateway": _observed_gateway(address="34.56.129.3", ready=True), + "caller-auth": _observed_caller_auth(), + }, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[ + _caller_key_secret(name="team-a-keys", data={"a": "c2stMQ=="}), + _caller_key_secret(name="team-b-keys", data={"a": "c2stMQ==", "b": "c2stMg=="}), + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "34.56.129.3", + "endpoints": { + "openAI": "http://34.56.129.3/v1", + "anthropic": "http://34.56.129.3/anthropic/v1", + }, + }, + ready=fnv1.READY_UNSPECIFIED, + ), + resources={ + "caller-secret-team-a-keys": _caller_secret(name="callers-team-a-keys", data={"a": "c2stMQ=="}), + "caller-secret-team-b-keys": _caller_secret( + name="callers-team-b-keys", data={"a": "c2stMQ==", "b": "c2stMg=="} + ), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-team-a-keys"}, {"name": "callers-team-b-keys"}], + ready=fnv1.READY_TRUE, + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition(type="GatewayReady", status=fnv1.STATUS_CONDITION_TRUE, reason="GatewayProgrammed"), + ], + ), + ), + # Envoy Gateway keeps the first Secret listed when two hold the same caller + # name, so listing them by name keeps the winner from changing between + # reconciles. + Case( + name="CallerSecretsUnordered", + reason="Caller Secrets that resolve out of name order are listed in name order in the caller policy.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}) + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[ + _caller_key_secret(name="team-b-keys", data={"b": "c2stMg=="}), + _caller_key_secret(name="team-a-keys", data={"a": "c2stMQ=="}), + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "caller-secret-team-a-keys": _caller_secret(name="callers-team-a-keys", data={"a": "c2stMQ=="}), + "caller-secret-team-b-keys": _caller_secret(name="callers-team-b-keys", data={"b": "c2stMg=="}), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-team-a-keys"}, {"name": "callers-team-b-keys"}], + ready=fnv1.READY_UNSPECIFIED, + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="Waiting for the Gateway on cluster gw-eu to be programmed" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="Waiting for the Gateway on cluster gw-eu to be programmed", + ) + ], + ), + ), + # Keeping the Gateway keeps its load balancer and address. The caller policy + # denies every request rather than authenticating nobody by omission. This + # case and the next differ only in the message reported. + # + # The deny-all caller policy is a different role from _caller_auth's API-key + # policy, and only these two cases compose it, so it's written inline. + Case( + name="NoCallerSecretMatched", + reason=( + "A caller selector that matches no Secret still composes the Gateway, copies no caller Secret, " + "composes a caller policy denying every request, and reports the Secrets missing." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}) + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "SecurityPolicy", + "metadata": { + "name": "inference-gateway-callers", + "namespace": "modelplane-system", + }, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + } + ], + "authorization": {"defaultAction": "Deny"}, + }, + } + }, + }, + } + ) + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="spec.auth.apiKey.secretSelector matches no Secret, so no caller could authenticate", + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="SecretsMissing", + message="spec.auth.apiKey.secretSelector matches no Secret, so no caller could authenticate", + ) + ], + ), + ), + Case( + name="CallerSecretsUnresolved", + reason=( + "Caller Secrets that haven't resolved yet still compose the Gateway, copy no caller Secret, compose " + "a caller policy denying every request, and report the Secrets missing." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}) + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "SecurityPolicy", + "metadata": { + "name": "inference-gateway-callers", + "namespace": "modelplane-system", + }, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + } + ], + "authorization": {"defaultAction": "Deny"}, + }, + } + }, + }, + } + ) + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for caller key Secrets to resolve")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="SecretsMissing", + message="Waiting for caller key Secrets to resolve", + ) + ], + ), + ), + # No composed Object observes a Secret, which is what keeps this gateway's + # client CA private key off the control plane. provider-kubernetes copies + # an observed object's whole manifest into the Object's status, and its + # --sanitize-secrets flag defaults to false, so observing a Secret publishes + # every key in it to anyone who can get objects. This CA signs the + # certificate every cluster gateway in the fleet accepts, so leaking its + # key means anyone can reach any engine. + # + # Without auth or TLS this function composes no Secret at all. + # + # Observing is what matters here. The Secrets this function writes also end + # up in status, because provider-kubernetes reports what it observes of what + # it manages, so this alone doesn't keep their contents off the control + # plane. Those hold caller keys and serving certificates that came from + # control-plane Secrets to begin with, so the exposure is a wider audience + # for data already present rather than data that would otherwise never be + # there, and prerequisites.yaml runs provider-kubernetes with + # --sanitize-secrets to redact it. A CA private key is different in kind: it + # is generated on the workload cluster and observing it is the only way it + # could ever reach the control plane. + Case( + name="TLSAndAuthUnobserved", + reason=( + "With TLS and auth on and nothing observed yet, a gateway copies its TLS and caller Secrets to the " + "cluster, and no composed Object observes a Secret." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=True, caller_secret_labels={"team": "ml"})), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[_caller_key_secret(name="ml-team-keys", data={"alice": "a2V5"})] + ), + "tls-secret-0": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "eu-tls-0", "namespace": "modelplane-system"}, + "data": {"tls.crt": "Y2VydA==", "tls.key": "a2V5"}, + } + ) + ) + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "caller-secret-ml-team-keys": _caller_secret(name="callers-ml-team-keys", data={"alice": "a2V5"}), + "tls-secret-eu-tls-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "eu-tls-0", "namespace": "modelplane-system"}, + "type": "kubernetes.io/tls", + "data": {"tls.crt": "Y2VydA==", "tls.key": "a2V5"}, + } + }, + }, + } + ) + ), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=True, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-ml-team-keys"}], ready=fnv1.READY_UNSPECIFIED + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + "redirect-route": _redirect_route(), + "redirect-auth": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "SecurityPolicy", + "metadata": { + "name": "inference-gateway-redirect-open", + "namespace": "modelplane-system", + }, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "HTTPRoute", + "name": "inference-gateway-redirect", + } + ], + "authorization": {"defaultAction": "Allow"}, + }, + } + }, + }, + } + ) + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="Waiting for the Gateway on cluster gw-eu to be programmed" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"team": "ml"}), + ), + "tls-secret-0": fnv1.ResourceSelector( + api_version="v1", kind="Secret", namespace="modelplane-system", match_name="eu-tls-0" + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="Waiting for the Gateway on cluster gw-eu to be programmed", + ) + ], + ), + ), +] + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes an InferenceGateway and reports its readiness.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-metric-mapping/tests/test_fn.py b/functions/compose-metric-mapping/tests/test_fn.py index bf5acbb72..358a6dcb0 100644 --- a/functions/compose-metric-mapping/tests/test_fn.py +++ b/functions/compose-metric-mapping/tests/test_fn.py @@ -25,6 +25,8 @@ from google.protobuf import duration_pb2 as durationpb from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb +from models.ai.modelplane.metricmapping import v1alpha1 +from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @dataclasses.dataclass @@ -32,100 +34,153 @@ class Case: """A test case for compose-metric-mapping.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse +def _metric_mapping() -> fnv1.Resource: + """The my-engine MetricMapping XR, renaming my_engine_queued to modelplane_requests_waiting.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.MetricMapping( + metadata=metav1.ObjectMeta(name="my-engine"), + spec=v1alpha1.Spec( + # The model accepts its from_ field only by its alias, + # from, which is a Python keyword, so the Metric is + # validated from its wire form. + metrics=[ + v1alpha1.Metric.model_validate( + {"from": "my_engine_queued", "to": "modelplane_requests_waiting"} + ), + ], + ), + ).model_dump(exclude_none=True, mode="json", by_alias=True) + ), + ) + + +def _desired_metric_mapping(*, status: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The desired MetricMapping XR, carrying status, or only its readiness if status is None.""" + if status is None: + return fnv1.Resource(ready=ready) + return fnv1.Resource(resource=resource.dict_to_struct({"status": status}), ready=ready) + + def _to_dict(msg: message.Message) -> dict: """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" return json.loads(json_format.MessageToJson(msg, sort_keys=True)) -def _compose_cases() -> list[Case]: - mapping = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "MetricMapping", - "metadata": {"name": "my-engine"}, - "spec": { - "metrics": [{"from": "my_engine_queued", "to": "modelplane_requests_waiting"}], - }, - } - cluster = resource.dict_to_struct( - {"apiVersion": "modelplane.ai/v1alpha1", "kind": "InferenceCluster", "metadata": {"name": "prod-us-east"}} - ) - - def req(xr: dict, clusters: list | None) -> fnv1.RunFunctionRequest: - r = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(xr))), - ) - if clusters is not None: - r.required_resources["clusters"].items.extend([fnv1.Resource(resource=c) for c in clusters]) - return r - - def want(ready: fnv1.Ready, status: dict | None, cond: fnv1.Condition) -> fnv1.RunFunctionResponse: - composite = fnv1.Resource(ready=ready) - if status is not None: - composite.resource.CopyFrom(resource.dict_to_struct(status)) - return fnv1.RunFunctionResponse( +# This function composes nothing, so desired carries only the composite. +COMPOSE_CASES = [ + # The function counts the clusters without reading them, so the two are + # identical. + Case( + name="TwoClusters", + reason="With two inference clusters resolved, the mapping is Accepted and Ready, and its status counts both.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_metric_mapping()), + required_resources={ + "clusters": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "prod-us-east"}, + } + ), + ), + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "prod-us-east"}, + } + ), + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=composite), - conditions=[cond], + desired=fnv1.State(composite=_desired_metric_mapping(status={"clusters": 2}, ready=fnv1.READY_TRUE)), context=structpb.Struct(), requirements=fnv1.Requirements( resources={ - "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster") - } + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, ), - ) - - return [ - Case( - name="ready, and says how many clusters took the renames", - req=req(mapping, [cluster, cluster]), - want=want( - fnv1.READY_TRUE, - {"status": {"clusters": 2}}, + conditions=[ fnv1.Condition( type="Accepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Available", message="Renaming 1 metric(s) on 2 inference cluster(s)", ), - ), + ], + ), + ), + Case( + name="NoClusters", + reason="With the clusters requirement resolved to none, the mapping isn't Ready and its status counts no clusters.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_metric_mapping()), + required_resources={"clusters": fnv1.Resources()}, ), - Case( - name="not ready when no cluster exists to render into", - req=req(mapping, []), - want=want( - fnv1.READY_FALSE, - {"status": {"clusters": 0}}, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_metric_mapping(status={"clusters": 0}, ready=fnv1.READY_FALSE)), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, + ), + conditions=[ fnv1.Condition( type="Accepted", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters", message="No inference cluster to render these renames into", ), - ), + ], + ), + ), + Case( + name="ClustersUnresolved", + reason="Until the clusters requirement resolves, the mapping waits for it, not Ready and with no status.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_metric_mapping()), ), - Case( - name="waits for the clusters to resolve", - req=req(mapping, None), - want=want( - fnv1.READY_FALSE, - None, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_metric_mapping(status=None, ready=fnv1.READY_FALSE)), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, + ), + conditions=[ fnv1.Condition( type="Accepted", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForClusters", message="Waiting for the inference clusters to resolve", ), - ), + ], ), - ] + ), +] -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: - """The function reports whether a mapping reaches any cluster.""" + """RunFunction reports whether the mapping reaches any inference cluster.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-model-cache/tests/test_fn.py b/functions/compose-model-cache/tests/test_fn.py index b1f607ff5..a2262eb69 100644 --- a/functions/compose-model-cache/tests/test_fn.py +++ b/functions/compose-model-cache/tests/test_fn.py @@ -30,857 +30,1178 @@ from models.ai.modelplane.modelcache import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -# A fixed transition time keeps observed conditions deterministic. -_TRANSITION_TIME = datetime.datetime(2026, 6, 8, tzinfo=datetime.UTC) - @dataclasses.dataclass class Case: """A test case for compose-model-cache.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse -# The XR used across cases: a HuggingFace ModelCache in the ml-team namespace. -# Both the PVC and Job derive their names from -# resource.child_name("modelcache", "ml-team", "qwen", ...). -def _cache_xr(**hf_extra: Any) -> v1alpha1.ModelCache: - return v1alpha1.ModelCache( - metadata=metav1.ObjectMeta(name="qwen", namespace="ml-team"), - spec=v1alpha1.Spec( - source="HuggingFace", - huggingFace=v1alpha1.HuggingFace(repo="Qwen/Qwen3-0.6B", sizeGiB=20, **hf_extra), +def _model_cache( + *, revision: str | None, auth_secret: v1alpha1.AuthSecret | None, status: v1alpha1.Status | None +) -> fnv1.Resource: + """The qwen ModelCache XR, with status unless it's None.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ModelCache( + metadata=metav1.ObjectMeta(name="qwen", namespace="ml-team"), + spec=v1alpha1.Spec( + source="HuggingFace", + huggingFace=v1alpha1.HuggingFace( + repo="Qwen/Qwen3-0.6B", + sizeGiB=20, + revision=revision, + authSecret=auth_secret, + ), + ), + status=status, + ).model_dump(exclude_none=True, mode="json", by_alias=True) ), ) -def _cluster_dict(name: str, pc: str, *, source: str = "GKE", storage_class: str | None = None) -> dict: - """An InferenceCluster as Crossplane returns it in a required-resource set. - - The cache PVC's StorageClass comes from status.cache.storageClassName, which - the InferenceCluster relays from its backing cluster. The match gate requires - both providerConfigRef AND status.cache, so every matchable fixture reports - one. Defaults to the source's effective class (GKE -> modelplane-rwx, - EKS -> modelplane-rwx-efs) unless overridden. - """ - blocks = { - "GKE": {"gke": {"project": "my-project", "region": "us-central1"}}, - "EKS": {"eks": {"region": "us-west-2"}}, - } - if storage_class is None: - storage_class = "modelplane-rwx-efs" if source == "EKS" else "modelplane-rwx" - return { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceCluster", - "metadata": {"name": name}, - "spec": {"cluster": {"source": source, **blocks[source]}}, - "status": { - "providerConfigRef": {"name": pc}, - "cache": {"storageClassName": storage_class}, - }, - } - +def _desired_model_cache(*, summary: str, clusters: list[dict], ready: fnv1.Ready) -> fnv1.Resource: + """The desired ModelCache XR, reporting summary as its ready count and each cluster's phase.""" + return fnv1.Resource( + resource=resource.dict_to_struct({"status": {"summary": {"ready": summary}, "clusters": clusters}}), + ready=ready, + ) -def _observed_object(manifest_status: dict, *, ready: bool = False) -> dict: - """A full Object envelope as provider-kubernetes observes it. - Carries the echoed remote status under status.atProvider.manifest.status - (read by derive_cluster_phase) and, when `ready`, the Object's own Ready - condition (read by mark_ready_resources, populated by DeriveFromCelQuery). - """ - status: dict[str, Any] = {"atProvider": {"manifest": {"status": manifest_status}}} - if ready: - status["conditions"] = [ +def _cluster(*, name: str, provider_config: str, cluster: dict, storage_class: str) -> fnv1.Resource: + """An InferenceCluster reporting the providerConfigRef and cache StorageClass the function matches on.""" + return fnv1.Resource( + resource=resource.dict_to_struct( { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2026-06-08T00:00:00Z", - }, - ] - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": {"forProvider": {"manifest": {}}}, - "status": status, - } - - -def _auth_secret(*, data: dict[str, str] | None = None) -> dict: - """The control-plane authSecret as Crossplane returns it in a required- - resource set: a core/v1 Secret with base64 `data`.""" - return { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": "hf-token", "namespace": "ml-team"}, - "type": "Opaque", - "data": {"HF_TOKEN": _TOKEN_B64} if data is None else data, - } - - -def _req( - xr: v1alpha1.ModelCache, - clusters: list[dict], - observed: dict[str, dict] | None = None, - auth: dict | None = None, -) -> fnv1.RunFunctionRequest: - """Build a request the way the repo's other function tests do. - - - XR goes in observed.composite via dict_to_struct(model_dump(mode="json")). - - Resolved clusters go in the `clusters` required-resource set. - - `auth`, when given, is the resolved control-plane Secret in the - `auth-secret` required-resource set. - - `observed` maps a desired-resource key -> an observed Object envelope. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json")), - ), + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": name}, + "spec": {"cluster": cluster}, + "status": { + "providerConfigRef": {"name": provider_config}, + "cache": {"storageClassName": storage_class}, + }, + } ), ) - # Touch the key so a resolved-but-empty match (clusters == []) is present - # with no items, the way Crossplane returns it - distinct from an unresolved - # requirement, whose key is absent. - req.required_resources["clusters"].items.extend( - fnv1.Resource(resource=resource.dict_to_struct(c)) for c in clusters - ) - if auth is not None: - req.required_resources["auth-secret"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(auth)), - ) - for key, obj in (observed or {}).items(): - req.observed.resources[key].resource.update(obj) - return req -# The function requires every InferenceCluster with a bare selector, so every -# response echoes this selector under requirements. -_CLUSTERS_SELECTOR = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceCluster", -) - -# When the cache references an authSecret, the function requires that Secret by -# name from the XR's namespace; the response echoes this selector. -_AUTH_SELECTOR = fnv1.ResourceSelector( - api_version="v1", - kind="Secret", - match_name="hf-token", - namespace="ml-team", -) - -# The hydration shell script the Job runs (no revision, no auth secret). No -# --local-dir: HF_HUB_CACHE (below) points `hf download` at the mount, so it -# stages in HuggingFace's cache layout and a serving pod can load by repo id. -_HYDRATE_CMD = ( - "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " - "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B; " - "touch /mnt/artifact/.modelplane-hydrated" -) -# With a pinned revision (case 2 wires --revision and the HF_TOKEN env). -_HYDRATE_CMD_REVISION = ( - "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " - "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B --revision main; " - "touch /mnt/artifact/.modelplane-hydrated" -) -# Set on the hydration Job so `hf download` writes the hub cache layout into -# the mount rather than the container's default ~/.cache. -_HYDRATE_CACHE_ENV = {"name": "HF_HUB_CACHE", "value": "/mnt/artifact"} - -_PVC_NAME = "modelcache-ml-team-qwen-17db2" -_JOB_NAME = "modelcache-ml-team-qwen-hydrate-256ec" -_AUTH_NAME = "modelcache-ml-team-qwen-auth-ae01b" -_LABELS = {"modelplane.ai/modelcache": "qwen"} - -# The token data the control-plane authSecret carries, base64 as the API server -# stores it. Propagated verbatim into the workload-cluster Secret's data. -_TOKEN_B64 = "aGYtdG9rZW4tdmFsdWU=" +def _auth_secret(*, data: dict[str, str]) -> fnv1.Resource: + """The ModelCache's control-plane authSecret, with base64 data.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "hf-token", "namespace": "ml-team"}, + "type": "Opaque", + "data": data, + } + ), + ) -def _pvc_object(pc: str, *, storage_class: str = "modelplane-rwx") -> dict: - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": { - "forProvider": { - "manifest": { - "apiVersion": "v1", - "kind": "PersistentVolumeClaim", - "metadata": {"name": _PVC_NAME, "namespace": "mp-ml-team-51733", "labels": _LABELS}, - "spec": { - "accessModes": ["ReadWriteMany"], - "resources": {"requests": {"storage": "20Gi"}}, - "storageClassName": storage_class, +def _pvc_object(*, provider_config: str, storage_class: str, ready: fnv1.Ready) -> fnv1.Resource: + """The Object wrapping the qwen ModelCache's cache PVC on a workload cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "PersistentVolumeClaim", + "metadata": { + "name": "modelcache-ml-team-qwen-17db2", + "namespace": "mp-ml-team-51733", + "labels": {"modelplane.ai/modelcache": "qwen"}, + }, + "spec": { + "accessModes": ["ReadWriteMany"], + "resources": {"requests": {"storage": "20Gi"}}, + "storageClassName": storage_class, + }, + }, }, + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": provider_config}, + "readiness": {"celQuery": 'object.status.phase == "Bound"', "policy": "DeriveFromCelQuery"}, }, - }, - "providerConfigRef": {"kind": "ClusterProviderConfig", "name": pc}, - "readiness": {"celQuery": 'object.status.phase == "Bound"', "policy": "DeriveFromCelQuery"}, - }, - } + } + ), + ready=ready, + ) -def _job_object(pc: str, *, command: str = _HYDRATE_CMD, env: list | None = None) -> dict: - # HF_HUB_CACHE leads the Job's env on every source; anything the caller - # passes (an authSecret's HF_TOKEN) follows it. - env = [_HYDRATE_CACHE_ENV, *(env or [])] - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": { - "forProvider": { - "manifest": { - "apiVersion": "batch/v1", - "kind": "Job", - "metadata": {"name": _JOB_NAME, "namespace": "mp-ml-team-51733", "labels": _LABELS}, - "spec": { - "backoffLimit": 3, - "ttlSecondsAfterFinished": 180, - "template": { - "metadata": {"labels": _LABELS}, +def _job_object(*, provider_config: str, command: str, env: list[dict]) -> fnv1.Resource: + """The Object wrapping the qwen ModelCache's hydration Job, which mounts the cache PVC.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "forProvider": { + "manifest": { + "apiVersion": "batch/v1", + "kind": "Job", + "metadata": { + "name": "modelcache-ml-team-qwen-hydrate-256ec", + "namespace": "mp-ml-team-51733", + "labels": {"modelplane.ai/modelcache": "qwen"}, + }, "spec": { - "restartPolicy": "OnFailure", - "containers": [ - { - "name": "hydrate", - "image": "python:3.11-slim", - "command": ["/bin/sh", "-c", command], - "env": env, - "volumeMounts": [{"name": "artifact", "mountPath": "/mnt/artifact"}], + "backoffLimit": 3, + "ttlSecondsAfterFinished": 180, + "template": { + "metadata": {"labels": {"modelplane.ai/modelcache": "qwen"}}, + "spec": { + "restartPolicy": "OnFailure", + "containers": [ + { + "name": "hydrate", + "image": "python:3.11-slim", + "command": ["/bin/sh", "-c", command], + "env": env, + "volumeMounts": [{"name": "artifact", "mountPath": "/mnt/artifact"}], + }, + ], + "volumes": [ + { + "name": "artifact", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}, + }, + ], }, - ], - "volumes": [ - {"name": "artifact", "persistentVolumeClaim": {"claimName": _PVC_NAME}}, - ], + }, }, }, }, + "managementPolicies": ["Observe", "Create", "Update", "LateInitialize"], + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": provider_config}, + "readiness": { + "celQuery": 'object.status.conditions.exists(c, c.type == "Complete" && c.status == "True")', + "policy": "DeriveFromCelQuery", + }, }, - }, - "managementPolicies": ["Observe", "Create", "Update", "LateInitialize"], - "providerConfigRef": {"kind": "ClusterProviderConfig", "name": pc}, - "readiness": { - "celQuery": 'object.status.conditions.exists(c, c.type == "Complete" && c.status == "True")', - "policy": "DeriveFromCelQuery", - }, - }, - } + } + ), + ) -def _auth_object(pc: str) -> dict: - """The workload-cluster Secret Object propagating the HF token. No readiness - block: a Secret has no status, so the Object uses default readiness.""" - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": { - "forProvider": { - "manifest": { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": _AUTH_NAME, "namespace": "mp-ml-team-51733", "labels": _LABELS}, - "data": {"HF_TOKEN": _TOKEN_B64}, +def _observed_pvc_object() -> fnv1.Resource: + """The cache PVC's Object as observed, with the PVC Bound and the Object Ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": {"forProvider": {"manifest": {}}}, + "status": { + "atProvider": {"manifest": {"status": {"phase": "Bound"}}}, + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + }, + ], }, - }, - "providerConfigRef": {"kind": "ClusterProviderConfig", "name": pc}, - }, + } + ), + ) + + +def _observed_job_object(*, condition: str, ready: bool) -> fnv1.Resource: + """The hydration Job's Object as observed, the Job reporting condition, with a Ready condition only if ready.""" + status: dict[str, Any] = { + "atProvider": {"manifest": {"status": {"conditions": [{"type": condition, "status": "True"}]}}}, } + if ready: + status["conditions"] = [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + }, + ] + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": {"forProvider": {"manifest": {}}}, + "status": status, + } + ), + ) -def _compose_cases() -> list[Case]: - """The cases for test_compose, built by code that derives them from shared parts.""" - # --- Case 1: GKE cluster, first pass. Composes the RWX PVC + hydration - # Job per matched cluster; nothing observed yet so phase is Pending and - # ArtifactReady is Hydrating. Emits the one-time "Staging" event. --- - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Pending"}], - }, - }, +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +# Every case is a HuggingFace ModelCache named qwen in the ml-team namespace. The +# function requires every InferenceCluster with a bare selector, so every +# response carries that requirement. +COMPOSE_CASES = [ + # The Job sets HF_HUB_CACHE rather than passing --local-dir, so `hf download` + # writes HuggingFace's cache layout to the mount and a serving pod can load + # the model by repo id. + Case( + name="GKEFirstPass", + reason="With nothing observed yet, a ModelCache composes a PVC and a hydration Job on its GKE cluster and emits a Staging event.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache(revision=None, auth_secret=None, status=None), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="0/1", + clusters=[{"name": "cluster-a", "phase": "Pending"}], + ready=fnv1.READY_UNSPECIFIED, ), + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_UNSPECIFIED, + ), + "hydrate-cluster-a": _job_object( + provider_config="cluster-a-pc", + command=( + "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " + "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B; " + "touch /mnt/artifact/.modelplane-hydrated" + ), + env=[{"name": "HF_HUB_CACHE", "value": "/mnt/artifact"}], + ), + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), - "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + ), + ), + # The function copies the token's base64 verbatim to a workload-cluster + # Secret, and the Job's HF_TOKEN references that Secret, because the + # control-plane Secret isn't on the workload cluster. A Secret has no + # status, so its Object has no readiness block and uses default readiness. + Case( + name="RevisionAndAuthSecret", + reason="A ModelCache with a revision and a resolved authSecret passes --revision to hf download and gives the Job the token as HF_TOKEN.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision="main", + auth_secret=v1alpha1.AuthSecret(name="hf-token"), + status=None, + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], + ), + "auth-secret": fnv1.Resources(items=[_auth_secret(data={"HF_TOKEN": "aGYtdG9rZW4tdmFsdWU="})]), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - want1.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 2: GKE cluster with a pinned revision + auth secret. The Job - # command gains --revision, and the function propagates the token to a - # workload-cluster Secret (auth-cluster-a) whose name the Job's HF_TOKEN - # env references - not the user's control-plane Secret name. --- - xr2 = _cache_xr(revision="main", authSecret=v1alpha1.AuthSecret(name="hf-token")) - env2 = [ - { - "name": "HF_TOKEN", - "valueFrom": {"secretKeyRef": {"name": _AUTH_NAME, "key": "HF_TOKEN"}}, - }, - ] - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Pending"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="0/1", + clusters=[{"name": "cluster-a", "phase": "Pending"}], + ready=fnv1.READY_UNSPECIFIED, ), + resources={ + "auth-cluster-a": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Secret", + "metadata": { + "name": "modelcache-ml-team-qwen-auth-ae01b", + "namespace": "mp-ml-team-51733", + "labels": {"modelplane.ai/modelcache": "qwen"}, + }, + "data": {"HF_TOKEN": "aGYtdG9rZW4tdmFsdWU="}, + }, + }, + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + }, + } + ), + ), + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_UNSPECIFIED, + ), + "hydrate-cluster-a": _job_object( + provider_config="cluster-a-pc", + command=( + "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " + "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B --revision main; " + "touch /mnt/artifact/.modelplane-hydrated" + ), + env=[ + {"name": "HF_HUB_CACHE", "value": "/mnt/artifact"}, + { + "name": "HF_TOKEN", + "valueFrom": { + "secretKeyRef": {"name": "modelcache-ml-team-qwen-auth-ae01b", "key": "HF_TOKEN"}, + }, + }, + ], + ), + }, ), - resources={ - "auth-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_auth_object("cluster-a-pc"))), - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), - "hydrate-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct( - _job_object("cluster-a-pc", command=_HYDRATE_CMD_REVISION, env=env2), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "auth-secret": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="hf-token", + namespace="ml-team", ), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + ), + ), + Case( + name="StorageClassFromStatus", + reason="A ModelCache on an EKS cluster gives its PVC the EFS StorageClass the cluster reports in status.cache.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache(revision=None, auth_secret=None, status=None), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="eks-a", + provider_config="eks-a-pc", + cluster={"source": "EKS", "eks": {"region": "us-west-2"}}, + storage_class="modelplane-rwx-efs", + ), + ], ), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - want2.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want2.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 3: EKS cluster reporting an EFS RWX class on status.cache. The - # PVC sources its storageClassName from status.cache (modelplane-rwx-efs), - # not the GKE/Filestore one. --- - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "eks-a", "phase": "Pending"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="0/1", + clusters=[{"name": "eks-a", "phase": "Pending"}], + ready=fnv1.READY_UNSPECIFIED, ), + resources={ + "pvc-eks-a": _pvc_object( + provider_config="eks-a-pc", + storage_class="modelplane-rwx-efs", + ready=fnv1.READY_UNSPECIFIED, + ), + "hydrate-eks-a": _job_object( + provider_config="eks-a-pc", + command=( + "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " + "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B; " + "touch /mnt/artifact/.modelplane-hydrated" + ), + env=[{"name": "HF_HUB_CACHE", "value": "/mnt/artifact"}], + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: eks-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, ), - resources={ - "pvc-eks-a": fnv1.Resource( - resource=resource.dict_to_struct( - _pvc_object("eks-a-pc", storage_class="modelplane-rwx-efs"), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + ), + ), + # The observed PVC suppresses the Staging event, and the XR becoming Ready + # emits the staged one. + Case( + name="JobComplete", + reason="With its PVC bound and its hydration Job complete, a ModelCache drops the Job and reports Ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache(revision=None, auth_secret=None, status=None), + resources={ + "pvc-cluster-a": _observed_pvc_object(), + "hydrate-cluster-a": _observed_job_object(condition="Complete", ready=True), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="1/1", + clusters=[{"name": "cluster-a", "phase": "Ready"}], + ready=fnv1.READY_TRUE, + ), + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_TRUE, ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + ), + ), + # The observed PVC suppresses the Staging event. + Case( + name="PVCBound", + reason="With its PVC bound and no Job observed, a ModelCache composes the Job and reports the cluster Hydrating.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache(revision=None, auth_secret=None, status=None), + resources={ + "pvc-cluster-a": _observed_pvc_object(), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], ), - "hydrate-eks-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("eks-a-pc"))), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: eks-a", - ), - ], - context=structpb.Struct(), - ) - want3.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 4: ready. The observed PVC + Job Objects each carry their own - # Ready condition (from DeriveFromCelQuery), and the wrapped manifest - # status shows PVC Bound + Job succeeded. Phase Ready, both Objects - # marked ready, summary 1/1, XR ready, ArtifactReady Staged. The - # already-composed PVC suppresses the "Staging" event; the - # not-previously-ready -> ready transition emits the "staged" event. --- - observed4 = { - "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), - } - pvc_ready = _pvc_object("cluster-a-pc") - want4 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="0/1", + clusters=[{"name": "cluster-a", "phase": "Hydrating"}], + ready=fnv1.READY_UNSPECIFIED, ), - ready=fnv1.READY_TRUE, + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_TRUE, + ), + "hydrate-cluster-a": _job_object( + provider_config="cluster-a-pc", + command=( + "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " + "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B; " + "touch /mnt/artifact/.modelplane-hydrated" + ), + env=[{"name": "HF_HUB_CACHE", "value": "/mnt/artifact"}], + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + ), + ), + Case( + name="JobFailed", + reason="With its PVC bound and its hydration Job failed, a ModelCache reports the cluster Failed rather than Hydrating.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache(revision=None, auth_secret=None, status=None), + resources={ + "pvc-cluster-a": _observed_pvc_object(), + "hydrate-cluster-a": _observed_job_object(condition="Failed", ready=False), + }, ), - resources={ - # Job dropped once Ready; only the PVC remains composed. - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(pvc_ready), ready=fnv1.READY_TRUE), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], + ), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - results=[ - fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), - ], - context=structpb.Struct(), - ) - want4.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 5: hydrating. PVC Bound (Object Ready) but the Job hasn't - # completed, so phase is Hydrating, only the PVC is marked ready, summary - # 0/1, and the XR is not ready. No transition event fires. --- - observed5 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} - want5 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Hydrating"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="0/1", + clusters=[{"name": "cluster-a", "phase": "Failed"}], + ready=fnv1.READY_UNSPECIFIED, ), + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_TRUE, + ), + "hydrate-cluster-a": _job_object( + provider_config="cluster-a-pc", + command=( + "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " + "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B; " + "touch /mnt/artifact/.modelplane-hydrated" + ), + env=[{"name": "HF_HUB_CACHE", "value": "/mnt/artifact"}], + ), + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Failed"), + ], + ), + ), + # Cluster a is Ready, so its Job is dropped. + Case( + name="OneOfTwoReady", + reason="With one of two clusters hydrated and the other still hydrating, a ModelCache reports 1/2 ready and the artifact Partial.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache(revision=None, auth_secret=None, status=None), + resources={ + "pvc-a": _observed_pvc_object(), + "hydrate-a": _observed_job_object(condition="Complete", ready=True), + "pvc-b": _observed_pvc_object(), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="a", + provider_config="a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + _cluster( + name="b", + provider_config="b-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], ), - "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - context=structpb.Struct(), - ) - want5.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 6: failed. The Job reports a Failed condition (and is NOT - # Ready). A Failed Job takes precedence over PVC binding, so phase is - # Failed, only the PVC is marked ready, summary 0/1, XR not ready, and - # ArtifactReady is False with reason Failed. --- - observed6 = { - "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Failed", "status": "True"}]}), - } - want6 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Failed"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="1/2", + clusters=[{"name": "a", "phase": "Ready"}, {"name": "b", "phase": "Hydrating"}], + ready=fnv1.READY_UNSPECIFIED, ), + resources={ + "pvc-a": _pvc_object(provider_config="a-pc", storage_class="modelplane-rwx", ready=fnv1.READY_TRUE), + "pvc-b": _pvc_object(provider_config="b-pc", storage_class="modelplane-rwx", ready=fnv1.READY_TRUE), + "hydrate-b": _job_object( + provider_config="b-pc", + command=( + "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " + "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B; " + "touch /mnt/artifact/.modelplane-hydrated" + ), + env=[{"name": "HF_HUB_CACHE", "value": "/mnt/artifact"}], + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Partial"), + ], + ), + ), + # The TTL controller cleans up a finished Job, leaving only the PVC. The + # status latch keeps the cluster Ready, and the XR was already Ready, so no + # event fires. + Case( + name="JobCleanedUp", + reason="With the XR's status already reporting its cluster Ready and only the PVC observed, a ModelCache stays Ready and doesn't compose the Job again.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision=None, + auth_secret=None, + status=v1alpha1.Status( + summary=v1alpha1.Summary(ready="1/1"), + clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], + conditions=[ + v1alpha1.Condition( + type="Ready", + status="True", + reason="Available", + lastTransitionTime=datetime.datetime(2026, 6, 8, tzinfo=datetime.UTC), + ), + ], + ), + ), + resources={ + "pvc-cluster-a": _observed_pvc_object(), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], ), - "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Failed"), - ], - context=structpb.Struct(), - ) - want6.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 7: partial (1/2). Cluster a is Ready (PVC Bound + Job - # succeeded), cluster b is still Hydrating (PVC Bound only). Summary - # 1/2, ArtifactReady False with reason Partial, XR not ready. --- - observed7 = { - "pvc-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), - "pvc-b": _observed_object({"phase": "Bound"}, ready=True), - } - want7 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/2"}, - "clusters": [ - {"name": "a", "phase": "Ready"}, - {"name": "b", "phase": "Hydrating"}, - ], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="1/1", + clusters=[{"name": "cluster-a", "phase": "Ready"}], + ready=fnv1.READY_TRUE, ), + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_TRUE, + ), + }, ), - resources={ - # Cluster a is Ready, so its Job is dropped; b is still Hydrating. - "pvc-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("a-pc")), ready=fnv1.READY_TRUE), - "pvc-b": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("b-pc")), ready=fnv1.READY_TRUE), - "hydrate-b": fnv1.Resource(resource=resource.dict_to_struct(_job_object("b-pc"))), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + ), + ), + Case( + name="AuthSecretUnresolved", + reason="Until Crossplane resolves its authSecret, a ModelCache requires the Secret and composes nothing.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision=None, + auth_secret=v1alpha1.AuthSecret(name="hf-token"), + status=None, + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], + ), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Partial"), - ], - context=structpb.Struct(), - ) - want7.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 8: latch. A previously-Ready cluster whose hydration Job was - # dropped (and TTL-cleaned), so only the PVC is observed now. The status - # latch keeps phase Ready and the Job is not re-composed, PVC marked - # ready, summary 1/1, XR ready. Already-ready, so no transition event. --- - xr_ready = _cache_xr() - xr_ready.status = v1alpha1.Status( - summary=v1alpha1.Summary(ready="1/1"), - clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], - conditions=[ - v1alpha1.Condition( - type="Ready", - status="True", - reason="Available", - lastTransitionTime=_TRANSITION_TIME, - ), - ], - ) - observed8 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} - want8 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "auth-secret": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="hf-token", + namespace="ml-team", + ), + }, + ), + ), + ), + # The PVC doesn't need the token, so it still composes and a cache isn't + # pruned for a missing one, but the Job and token Secret are held back. The + # XR is marked not ready, since the PVC alone would make it ready once it + # binds. + Case( + name="TokenMissing", + reason="With its authSecret lacking the HF_TOKEN key, a ModelCache composes only the PVC, warns, and reports AuthSecretMissing.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision=None, + auth_secret=v1alpha1.AuthSecret(name="hf-token"), + status=None, ), - ready=fnv1.READY_TRUE, ), - resources={ - # Latched Ready with the Job already dropped: only the PVC. - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], ), + "auth-secret": fnv1.Resources(items=[_auth_secret(data={"OTHER": "aGYtdG9rZW4tdmFsdWU="})]), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - context=structpb.Struct(), - ) - want8.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 9: authSecret referenced but not yet resolved. The function - # requires both the clusters and the auth Secret, then returns early - # (no resources, status, or conditions) until Crossplane resolves the - # Secret and re-calls it. --- - xr9 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - want9 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(), - context=structpb.Struct(), - ) - want9.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want9.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 10: authSecret resolved but the Secret lacks the referenced - # key (here it carries OTHER, not HF_TOKEN). The PVC still composes - it - # doesn't depend on the token, so a cache isn't pruned for a missing one - # - but the hydration Job and token Secret are held back. ArtifactReady - # is False with reason AuthSecretMissing, and a warning names the Secret - # and key so the user can fix it instead of seeing the XR stall. The XR - # is marked not ready, since the PVC alone would make it ready once it - # binds. --- - want10 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Pending"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="0/1", + clusters=[{"name": "cluster-a", "phase": "Pending"}], + ready=fnv1.READY_FALSE, ), - ready=fnv1.READY_FALSE, + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_UNSPECIFIED, + ), + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="authSecret ml-team/hf-token is missing or has no key 'HF_TOKEN'", + ), + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "auth-secret": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="hf-token", + namespace="ml-team", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="AuthSecretMissing"), + ], + ), + ), + # An empty token value is as broken as a missing key: the Job would run with + # an empty HF_TOKEN. + Case( + name="TokenEmpty", + reason="With its authSecret's HF_TOKEN value empty, a ModelCache composes only the PVC, warns, and reports AuthSecretMissing.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision=None, + auth_secret=v1alpha1.AuthSecret(name="hf-token"), + status=None, + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], + ), + "auth-secret": fnv1.Resources(items=[_auth_secret(data={"HF_TOKEN": ""})]), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="AuthSecretMissing"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_WARNING, - message="authSecret ml-team/hf-token is missing or has no key 'HF_TOKEN'", - ), - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - want10.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want10.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 11: Ready cluster with an authSecret. The token is only needed - # while hydrating, so once the cluster is Ready the auth Secret is dropped - # alongside the Job (only the PVC remains composed), even though the - # control-plane Secret still resolves. Keeps the token from lingering on - # the inference cluster after hydration. --- - xr11 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - observed11 = { - "auth-cluster-a": _observed_object({}, ready=True), - "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), - } - want11 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="0/1", + clusters=[{"name": "cluster-a", "phase": "Pending"}], + ready=fnv1.READY_FALSE, ), - ready=fnv1.READY_TRUE, + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_UNSPECIFIED, + ), + }, ), - resources={ - # Ready: the auth Secret and Job are both dropped, only the PVC remains. - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="authSecret ml-team/hf-token is missing or has no key 'HF_TOKEN'", ), + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "auth-secret": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="hf-token", + namespace="ml-team", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="AuthSecretMissing"), + ], + ), + ), + # The token is only needed while hydrating, so it's dropped even though the + # control-plane Secret still resolves. That keeps the token off the inference + # cluster after hydration. + Case( + name="AuthSecretJobComplete", + reason="With its hydration Job complete, a ModelCache with an authSecret drops the workload-cluster token Secret along with the Job.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision=None, + auth_secret=v1alpha1.AuthSecret(name="hf-token"), + status=None, + ), + resources={ + "auth-cluster-a": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": {"forProvider": {"manifest": {}}}, + "status": { + "atProvider": {"manifest": {"status": {}}}, + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + }, + ], + }, + } + ), + ), + "pvc-cluster-a": _observed_pvc_object(), + "hydrate-cluster-a": _observed_job_object(condition="Complete", ready=True), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], + ), + "auth-secret": fnv1.Resources(items=[_auth_secret(data={"HF_TOKEN": "aGYtdG9rZW4tdmFsdWU="})]), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - results=[ - fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), - ], - context=structpb.Struct(), - ) - want11.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want11.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 12: token rotated away after a cache is Ready. A latched-Ready - # cluster whose authSecret now resolves without the key. The PVC keeps - # composing (and stays Ready via the status latch) rather than being - # pruned, and because hydration is already done the missing token is - # neither reported (ArtifactReady stays Staged) nor warned. --- - xr12 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - xr12.status = v1alpha1.Status( - summary=v1alpha1.Summary(ready="1/1"), - clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], - conditions=[ - v1alpha1.Condition( - type="Ready", - status="True", - reason="Available", - lastTransitionTime=_TRANSITION_TIME, - ), - ], - ) - observed12 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} - want12 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="1/1", + clusters=[{"name": "cluster-a", "phase": "Ready"}], + ready=fnv1.READY_TRUE, + ), + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_TRUE, + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "auth-secret": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="hf-token", + namespace="ml-team", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + ), + ), + # The PVC doesn't need the token, so it isn't pruned, and hydration is done, + # so the missing token is neither reported nor warned. + Case( + name="TokenRotatedAway", + reason="With the XR's status already reporting its cluster Ready, a ModelCache whose authSecret lacks the HF_TOKEN key keeps its PVC and stays Ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision=None, + auth_secret=v1alpha1.AuthSecret(name="hf-token"), + status=v1alpha1.Status( + summary=v1alpha1.Summary(ready="1/1"), + clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], + conditions=[ + v1alpha1.Condition( + type="Ready", + status="True", + reason="Available", + lastTransitionTime=datetime.datetime(2026, 6, 8, tzinfo=datetime.UTC), + ), + ], + ), ), - ready=fnv1.READY_TRUE, + resources={ + "pvc-cluster-a": _observed_pvc_object(), + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], ), + "auth-secret": fnv1.Resources(items=[_auth_secret(data={"OTHER": "aGYtdG9rZW4tdmFsdWU="})]), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - context=structpb.Struct(), - ) - want12.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want12.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 13: authSecret missing AND no clusters matched. NoClusters is - # the dominant signal - the cache can't progress regardless of the token - # - so both conditions report NoClusters and the missing token is neither - # reported nor warned. --- - xr13 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - want13 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"summary": {"ready": "0/0"}, "clusters": []}}), - ready=fnv1.READY_FALSE, - ), - ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), - ], - context=structpb.Struct(), - ) - want13.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want13.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - return [ - Case( - name="GKE cluster first pass composes RWX PVC and hydration Job", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")]), - want=want1, - ), - Case( - name="HuggingFace revision and auth secret wire --revision and HF_TOKEN", - req=_req(xr2, [_cluster_dict("cluster-a", "cluster-a-pc")], auth=_auth_secret()), - want=want2, - ), - Case( - name="EKS cluster PVC sources the EFS class from status.cache", - req=_req(_cache_xr(), [_cluster_dict("eks-a", "eks-a-pc", source="EKS")]), - want=want3, - ), - Case( - name="PVC bound and Job complete reports Ready", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed4), - want=want4, - ), - Case( - name="PVC bound but Job running reports Hydrating", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed5), - want=want5, - ), - Case( - name="failed Job reports Failed and takes precedence over PVC binding", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed6), - want=want6, - ), - Case( - name="one of two clusters ready reports partial", - req=_req(_cache_xr(), [_cluster_dict("a", "a-pc"), _cluster_dict("b", "b-pc")], observed7), - want=want7, - ), - Case( - name="hydrated cluster stays Ready after its Job is TTL-cleaned", - req=_req(xr_ready, [_cluster_dict("cluster-a", "cluster-a-pc")], observed8), - want=want8, - ), - Case( - name="authSecret unresolved requires it and returns early", - req=_req(xr9, [_cluster_dict("cluster-a", "cluster-a-pc")]), - want=want9, - ), - Case( - name="authSecret resolved without the referenced key composes PVC and warns", - req=_req( - xr9, - [_cluster_dict("cluster-a", "cluster-a-pc")], - auth=_auth_secret(data={"OTHER": _TOKEN_B64}), - ), - want=want10, - ), - Case( - name="authSecret resolved with an empty token value composes PVC and warns", - req=_req( - xr9, - [_cluster_dict("cluster-a", "cluster-a-pc")], - auth=_auth_secret(data={"HF_TOKEN": ""}), - ), - want=want10, - ), - Case( - name="Ready cluster drops the auth Secret with the Job", - req=_req( - xr11, - [_cluster_dict("cluster-a", "cluster-a-pc")], - observed11, - auth=_auth_secret(), - ), - want=want11, - ), - Case( - name="token rotated away after Ready keeps the PVC and stays Ready", - req=_req( - xr12, - [_cluster_dict("cluster-a", "cluster-a-pc")], - observed12, - auth=_auth_secret(data={"OTHER": _TOKEN_B64}), - ), - want=want12, - ), - Case( - name="authSecret missing with no clusters reports NoClusters not AuthSecretMissing", - req=_req(xr13, [], auth=_auth_secret(data={"OTHER": _TOKEN_B64})), - want=want13, - ), - ] - - -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="1/1", + clusters=[{"name": "cluster-a", "phase": "Ready"}], + ready=fnv1.READY_TRUE, + ), + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "auth-secret": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="hf-token", + namespace="ml-team", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + ), + ), + # Crossplane reports a selector that matches nothing as a present but empty + # entry, unlike an unresolved one, whose key is absent. NoClusters dominates, + # since the cache can't progress whatever the token, so the missing token is + # neither reported nor warned. + Case( + name="NoClustersTokenMissing", + reason="With no InferenceClusters and its authSecret lacking the HF_TOKEN key, a ModelCache reports NoClusters rather than AuthSecretMissing.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision=None, + auth_secret=v1alpha1.AuthSecret(name="hf-token"), + status=None, + ), + ), + required_resources={ + "clusters": fnv1.Resources(), + "auth-secret": fnv1.Resources(items=[_auth_secret(data={"OTHER": "aGYtdG9rZW4tdmFsdWU="})]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache(summary="0/0", clusters=[], ready=fnv1.READY_FALSE), + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "auth-secret": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="hf-token", + namespace="ml-team", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), + ], + ), + ), +] -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction composes a ModelCache's PVC and hydration Job per cluster and reports their progress.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-model-deployment/tests/test_cel.py b/functions/compose-model-deployment/tests/test_cel.py index 7a2e3fd9b..72359fe8f 100644 --- a/functions/compose-model-deployment/tests/test_cel.py +++ b/functions/compose-model-deployment/tests/test_cel.py @@ -26,183 +26,307 @@ from function import cel -def _device( - driver: str = "gpu.nvidia.com", - attributes: dict | None = None, - capacity: dict | None = None, - **extra: object, -) -> dict: - """A pool device in the raw dict shape cel.Program.matches expects.""" - return { - "driver": driver, - "attributes": attributes or {}, - "capacity": capacity or {}, - **extra, - } - - @dataclasses.dataclass class Case: + """A test case for matching a DRA CEL selector against a device.""" + name: str + reason: str expr: str device: dict want: bool -# Reusable device fixtures. -_GPU = _device( - driver="gpu.nvidia.com", - attributes={ - "architecture": {"string": "Hopper"}, - "cudaComputeCapability": {"version": "9.5.3"}, - # A qualified name lands under its own domain, not the driver's. - "resource.kubernetes.io/pcieRoot": {"string": "pci0"}, - }, - capacity={"memory": {"value": "141Gi"}}, -) -_NIC = _device(driver="nic.nvidia.com", attributes={"linkType": {"string": "infiniband"}}) +def _hopper_gpu() -> dict: + """A gpu.nvidia.com Hopper GPU with CUDA compute capability 9.5.3 and 141Gi of memory, on PCIe root pci0.""" + return { + "driver": "gpu.nvidia.com", + "attributes": { + "architecture": {"string": "Hopper"}, + "cudaComputeCapability": {"version": "9.5.3"}, + "resource.kubernetes.io/pcieRoot": {"string": "pci0"}, + }, + "capacity": {"memory": {"value": "141Gi"}}, + } + -_ATTR = 'device.attributes["gpu.nvidia.com"]' -_CAP = 'device.capacity["gpu.nvidia.com"]' +def _scalar_device(*, x: dict) -> dict: + """A gpu.nvidia.com device whose one attribute, x, is the given typed scalar.""" + return {"driver": "gpu.nvidia.com", "attributes": {"x": x}, "capacity": {}} -# Device fixtures for the verbatim selector examples in the DRA docs. -# (k8s.io/docs concept page and the allocate-devices-dra task page.) -_LARGE_BLACK = _device( - driver="resource-driver.example.com", - attributes={"color": {"string": "black"}, "size": {"string": "large"}}, -) -_SMALL_WHITE = _device( - driver="resource-driver.example.com", - attributes={"color": {"string": "white"}, "size": {"string": "small"}}, -) -_EXAMPLE_GPU = _device( - driver="gpu.example.com", - attributes={"type": {"string": "gpu"}}, -) -_GPU_64GI = _device( - driver="driver.example.com", - attributes={"type": {"string": "gpu"}}, - capacity={"memory": {"value": "64Gi"}}, -) + +def _example_device(*, color: str, size: str) -> dict: + """A resource-driver.example.com device, as in the DRA docs' examples.""" + return { + "driver": "resource-driver.example.com", + "attributes": {"color": {"string": color}, "size": {"string": size}}, + "capacity": {}, + } MATCHES_CASES = [ # driver. - Case(name="driver equals", expr='device.driver == "gpu.nvidia.com"', device=_GPU, want=True), - Case(name="driver not equals", expr='device.driver == "nic.nvidia.com"', device=_GPU, want=False), + Case( + name="DriverMatches", + reason="A selector naming the device's driver matches.", + expr='device.driver == "gpu.nvidia.com"', + device=_hopper_gpu(), + want=True, + ), + Case( + name="DriverMismatch", + reason="A selector naming another driver doesn't match.", + expr='device.driver == "nic.nvidia.com"', + device=_hopper_gpu(), + want=False, + ), # Quantity comparison + methods. Case( - name="quantity compareTo ge", - expr=f'{_CAP}.memory.compareTo(quantity("141Gi")) >= 0', - device=_GPU, + name="QuantityCompareTo", + reason="141Gi of GPU memory compares at least equal to 141Gi.", + expr='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("141Gi")) >= 0', + device=_hopper_gpu(), want=True, ), Case( - name="quantity compareTo too big", - expr=f'{_CAP}.memory.compareTo(quantity("200Gi")) >= 0', - device=_GPU, + name="QuantityTooBig", + reason="141Gi of GPU memory doesn't compare at least equal to 200Gi.", + expr='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("200Gi")) >= 0', + device=_hopper_gpu(), want=False, ), Case( - name="quantity isGreaterThan", - expr=f'{_CAP}.memory.isGreaterThan(quantity("80Gi"))', - device=_GPU, + name="QuantityIsGreaterThan", + reason="141Gi of GPU memory is greater than 80Gi.", + expr='device.capacity["gpu.nvidia.com"].memory.isGreaterThan(quantity("80Gi"))', + device=_hopper_gpu(), + want=True, + ), + Case( + name="QuantityIsLessThan", + reason="141Gi of GPU memory is less than 200Gi.", + expr='device.capacity["gpu.nvidia.com"].memory.isLessThan(quantity("200Gi"))', + device=_hopper_gpu(), want=True, ), - Case(name="quantity isLessThan", expr=f'{_CAP}.memory.isLessThan(quantity("200Gi"))', device=_GPU, want=True), - Case(name="quantity sign", expr=f"{_CAP}.memory.sign() == 1", device=_GPU, want=True), - Case(name="quantity asInteger", expr=f"{_CAP}.memory.asInteger() == {141 * 2**30}", device=_GPU, want=True), - Case(name="quantity isInteger", expr=f"{_CAP}.memory.isInteger()", device=_GPU, want=True), + # Upstream sign is global-only, so q.sign() is a compile error there. This + # pins the member form we accept anyway, a documented divergence in cel.py + # that only makes us more permissive. Case( - name="quantity add", - expr=f'{_CAP}.memory.add(quantity("1Gi")).compareTo(quantity("142Gi")) == 0', - device=_GPU, + name="QuantitySignMember", + reason="The member form of sign() finds 141Gi of GPU memory positive.", + expr='device.capacity["gpu.nvidia.com"].memory.sign() == 1', + device=_hopper_gpu(), want=True, ), - Case(name="isQuantity true", expr='isQuantity("1.3Gi")', device=_GPU, want=True), - Case(name="isQuantity false", expr='isQuantity("200K")', device=_GPU, want=False), + Case( + name="QuantityAsInteger", + reason="141Gi of GPU memory is the integer 151397597184.", + expr='device.capacity["gpu.nvidia.com"].memory.asInteger() == 151397597184', + device=_hopper_gpu(), + want=True, + ), + Case( + name="QuantityIsInteger", + reason="141Gi of GPU memory is an integer.", + expr='device.capacity["gpu.nvidia.com"].memory.isInteger()', + device=_hopper_gpu(), + want=True, + ), + Case( + name="QuantityAdd", + reason="141Gi of GPU memory plus 1Gi compares equal to 142Gi.", + expr='device.capacity["gpu.nvidia.com"].memory.add(quantity("1Gi")).compareTo(quantity("142Gi")) == 0', + device=_hopper_gpu(), + want=True, + ), + Case( + name="IsQuantity", + reason="isQuantity accepts 1.3Gi.", + expr='isQuantity("1.3Gi")', + device=_hopper_gpu(), + want=True, + ), + Case( + name="IsQuantityFalse", + reason="isQuantity rejects 200K, because Kubernetes spells the kilo suffix as a lower-case k.", + expr='isQuantity("200K")', + device=_hopper_gpu(), + want=False, + ), # Semver comparison + methods. Case( - name="semver isGreaterThan", - expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.0.0"))', - device=_GPU, + name="SemverIsGreaterThan", + reason="A CUDA compute capability of 9.5.3 is greater than 9.0.0.", + expr='device.attributes["gpu.nvidia.com"].cudaComputeCapability.isGreaterThan(semver("9.0.0"))', + device=_hopper_gpu(), + want=True, + ), + Case( + name="SemverNotGreater", + reason="A CUDA compute capability of 9.5.3 isn't greater than 9.9.0.", + expr='device.attributes["gpu.nvidia.com"].cudaComputeCapability.isGreaterThan(semver("9.9.0"))', + device=_hopper_gpu(), + want=False, + ), + Case( + name="SemverMajor", + reason="A CUDA compute capability of 9.5.3 has major version 9.", + expr='device.attributes["gpu.nvidia.com"].cudaComputeCapability.major() == 9', + device=_hopper_gpu(), + want=True, + ), + Case( + name="SemverMinor", + reason="A CUDA compute capability of 9.5.3 has minor version 5.", + expr='device.attributes["gpu.nvidia.com"].cudaComputeCapability.minor() == 5', + device=_hopper_gpu(), + want=True, + ), + Case( + name="SemverPatch", + reason="A CUDA compute capability of 9.5.3 has patch version 3.", + expr='device.attributes["gpu.nvidia.com"].cudaComputeCapability.patch() == 3', + device=_hopper_gpu(), + want=True, + ), + Case( + name="SemverEquality", + reason="A CUDA compute capability of 9.5.3 equals semver 9.5.3.", + expr='device.attributes["gpu.nvidia.com"].cudaComputeCapability == semver("9.5.3")', + device=_hopper_gpu(), want=True, ), Case( - name="semver not greater", - expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.9.0"))', - device=_GPU, + name="IsSemver", + reason="isSemver accepts 1.0.0.", + expr='isSemver("1.0.0")', + device=_hopper_gpu(), + want=True, + ), + Case( + name="IsSemverShort", + reason="isSemver rejects 1.0, which has no patch version.", + expr='isSemver("1.0")', + device=_hopper_gpu(), want=False, ), - Case(name="semver major", expr=f"{_ATTR}.cudaComputeCapability.major() == 9", device=_GPU, want=True), - Case(name="semver minor", expr=f"{_ATTR}.cudaComputeCapability.minor() == 5", device=_GPU, want=True), - Case(name="semver patch", expr=f"{_ATTR}.cudaComputeCapability.patch() == 3", device=_GPU, want=True), - Case(name="semver equality", expr=f'{_ATTR}.cudaComputeCapability == semver("9.5.3")', device=_GPU, want=True), - Case(name="isSemver strict true", expr='isSemver("1.0.0")', device=_GPU, want=True), - Case(name="isSemver strict rejects short", expr='isSemver("1.0")', device=_GPU, want=False), - Case(name="isSemver normalize accepts short", expr='isSemver("1.0", true)', device=_GPU, want=True), - Case(name="semver normalize overload", expr='semver("v1.0", true).major() == 1', device=_GPU, want=True), + Case( + name="IsSemverNormalize", + reason="isSemver with normalize accepts 1.0, which has no patch version.", + expr='isSemver("1.0", true)', + device=_hopper_gpu(), + want=True, + ), + Case( + name="SemverNormalize", + reason="semver() with normalize parses v1.0 as major version 1.", + expr='semver("v1.0", true).major() == 1', + device=_hopper_gpu(), + want=True, + ), # Typed scalar attributes (resolve straight to the value, no .string). - Case(name="string attribute", expr=f'{_ATTR}.architecture == "Hopper"', device=_GPU, want=True), Case( - name="string attribute mismatch", + name="StringAttribute", + reason="A string attribute reads as its value, Hopper.", + expr='device.attributes["gpu.nvidia.com"].architecture == "Hopper"', + device=_hopper_gpu(), + want=True, + ), + Case( + name="NonGPUDriverDomain", + reason="A NIC's string attribute reads under its own driver's domain.", expr='device.attributes["nic.nvidia.com"].linkType == "infiniband"', - device=_NIC, + device={"driver": "nic.nvidia.com", "attributes": {"linkType": {"string": "infiniband"}}, "capacity": {}}, want=True, ), Case( - name="bool attribute true", - expr=f"{_ATTR}.x", - device=_device(attributes={"x": {"bool": True}}), + name="BoolAttributeTrue", + reason="A true bool attribute matches on its own.", + expr='device.attributes["gpu.nvidia.com"].x', + device=_scalar_device(x={"bool": True}), want=True, ), Case( - name="bool attribute false", - expr=f"{_ATTR}.x", - device=_device(attributes={"x": {"bool": False}}), + name="BoolAttributeFalse", + reason="A false bool attribute doesn't match.", + expr='device.attributes["gpu.nvidia.com"].x', + device=_scalar_device(x={"bool": False}), want=False, ), Case( - name="int attribute", - expr=f"{_ATTR}.x >= 8", - device=_device(attributes={"x": {"int": 8}}), + name="IntAttribute", + reason="An int attribute of 8 satisfies >= 8.", + expr='device.attributes["gpu.nvidia.com"].x >= 8', + device=_scalar_device(x={"int": 8}), want=True, ), Case( - name="int attribute below", - expr=f"{_ATTR}.x >= 8", - device=_device(attributes={"x": {"int": 4}}), + name="IntAttributeBelow", + reason="An int attribute of 4 doesn't satisfy >= 8.", + expr='device.attributes["gpu.nvidia.com"].x >= 8', + device=_scalar_device(x={"int": 4}), want=False, ), # Qualified names split into their own domain. Case( - name="qualified name under its domain", + name="QualifiedAttributeName", + reason="The qualified attribute resource.kubernetes.io/pcieRoot reads under its own domain.", expr='device.attributes["resource.kubernetes.io"].pcieRoot == "pci0"', - device=_GPU, + device=_hopper_gpu(), + want=True, + ), + # The same input as StringAttribute, kept to mirror upstream's separate + # driver-name-qualifier row. + Case( + name="DriverNameQualifier", + reason="A bare attribute name reads under the device's driver domain.", + expr='device.attributes["gpu.nvidia.com"].architecture == "Hopper"', + device=_hopper_gpu(), want=True, ), - Case(name="bare name under driver domain", expr=f'{_ATTR}.architecture == "Hopper"', device=_GPU, want=True), # Non-matches that must not raise. Case( - name="two-component version is non-match", - expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("8.0.0"))', - device=_device(attributes={"cudaComputeCapability": {"version": "9.0"}}), + name="TwoPartVersion", + reason="A version attribute of 9.0 isn't valid semver, so the selector doesn't match rather than raising.", + expr='device.attributes["gpu.nvidia.com"].cudaComputeCapability.isGreaterThan(semver("8.0.0"))', + device={ + "driver": "gpu.nvidia.com", + "attributes": {"cudaComputeCapability": {"version": "9.0"}}, + "capacity": {}, + }, want=False, ), + # 10Mo isn't a valid quantity, but 10M wouldn't match this selector either, + # so a parser that read 10Mo as 10M would pass too. Case( - name="malformed quantity is non-match", - expr=f'{_CAP}.memory.compareTo(quantity("1Gi")) >= 0', - device=_device(capacity={"memory": {"value": "10Mo"}}), + name="MalformedQuantity", + reason="A selector for at least 1Gi of memory doesn't match a capacity of 10Mo, and doesn't raise.", + expr='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("1Gi")) >= 0', + device={"driver": "gpu.nvidia.com", "attributes": {}, "capacity": {"memory": {"value": "10Mo"}}}, + want=False, + ), + Case( + name="UnknownAttribute", + reason="Reading an attribute the device lacks doesn't match rather than raising.", + expr='device.attributes["gpu.nvidia.com"].nope == "x"', + device=_hopper_gpu(), want=False, ), - Case(name="unknown id is non-match", expr=f'{_ATTR}.nope == "x"', device=_GPU, want=False), # A non-bool selector must not spuriously match. Upstream rejects it # at compile time; we treat a non-bool result as a non-match. - Case(name="non-bool string selector is non-match", expr='"5"', device=_GPU, want=False), Case( - name="non-bool int selector is non-match", - expr=f"{_ATTR}.x", - device=_device(attributes={"x": {"int": 5}}), + name="NonBoolString", + reason="A selector that evaluates to a string doesn't match.", + expr='"5"', + device=_hopper_gpu(), + want=False, + ), + Case( + name="NonBoolInt", + reason="A selector that evaluates to an int doesn't match.", + expr='device.attributes["gpu.nvidia.com"].x', + device=_scalar_device(x={"int": 5}), want=False, ), # Domain presence. Upstream's domain-presence idiom is "" in @@ -211,94 +335,119 @@ class Case: # compile error on a real cluster (celpy accepts it - see cel.py's # documented divergences). An unknown domain is simply absent (False), # not present-but-empty. - Case(name="unknown domain absent", expr='"other.com" in device.attributes', device=_GPU, want=False), - Case(name="known domain present", expr='"gpu.nvidia.com" in device.attributes', device=_GPU, want=True), - # Reading an unknown domain resolves to an empty map (not an error), - # so an id lookup under it is a non-match rather than a failure. Case( - name="unknown domain id is non-match", + name="DomainCheckNegative", + reason="A domain the device has no attributes under isn't in device.attributes.", + expr='"other.com" in device.attributes', + device=_hopper_gpu(), + want=False, + ), + Case( + name="DomainCheckPositive", + reason="The device's driver domain is in device.attributes.", + expr='"gpu.nvidia.com" in device.attributes', + device=_hopper_gpu(), + want=True, + ), + # Upstream errors on the id, not the domain. matches() turns either error + # into a non-match. + Case( + name="UnknownDomainAttribute", + reason="Reading an attribute under a domain the device lacks doesn't match rather than raising.", expr='device.attributes["other.com"].x == "y"', - device=_GPU, + device=_hopper_gpu(), want=False, ), - # Guard a domain read with the in idiom before indexing it. Case( - name="guarded known domain", - expr=f'"gpu.nvidia.com" in device.attributes && {_ATTR}.architecture == "Hopper"', - device=_GPU, + name="GuardedDomain", + reason="A domain read guarded by the in idiom matches when the domain is present.", + expr='"gpu.nvidia.com" in device.attributes && device.attributes["gpu.nvidia.com"].architecture == "Hopper"', + device=_hopper_gpu(), want=True, ), - # The full design selector. Case( - name="full design expression", + name="DesignSelector", + reason="The design's selector, compute capability above 9.0.0 and at least 141Gi of memory, matches a Hopper GPU.", expr=( - f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.0.0")) && ' - f'{_CAP}.memory.compareTo(quantity("141Gi")) >= 0' + 'device.attributes["gpu.nvidia.com"].cudaComputeCapability.isGreaterThan(semver("9.0.0")) && ' + 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("141Gi")) >= 0' ), - device=_GPU, + device=_hopper_gpu(), want=True, ), - # Verbatim selector examples from the DRA docs, each against a device - # that should and should not match. + # Verbatim selector examples from the DRA docs (the k8s.io concept page + # and the allocate-devices-dra task page), each against a device that + # should match, and all but small-white against one that shouldn't. Case( - name="docs: large-black subrequest matches", + name="DocsLargeBlack", + reason="The docs' large-black selector matches a large black device.", expr=( 'device.attributes["resource-driver.example.com"].color == "black" && ' 'device.attributes["resource-driver.example.com"].size == "large"' ), - device=_LARGE_BLACK, + device=_example_device(color="black", size="large"), want=True, ), Case( - name="docs: large-black subrequest rejects small-white", + name="DocsLargeBlackRejects", + reason="The docs' large-black selector rejects a small white device.", expr=( 'device.attributes["resource-driver.example.com"].color == "black" && ' 'device.attributes["resource-driver.example.com"].size == "large"' ), - device=_SMALL_WHITE, + device=_example_device(color="white", size="small"), want=False, ), Case( - name="docs: small-white subrequest matches", + name="DocsSmallWhite", + reason="The docs' small-white selector matches a small white device.", expr=( 'device.attributes["resource-driver.example.com"].color == "white" && ' 'device.attributes["resource-driver.example.com"].size == "small"' ), - device=_SMALL_WHITE, + device=_example_device(color="white", size="small"), want=True, ), Case( - name="docs: extended-resource DeviceClass selector matches", + name="DocsDeviceClass", + reason="The docs' extended-resource DeviceClass selector matches a gpu.example.com GPU.", expr="device.driver == 'gpu.example.com' && device.attributes['gpu.example.com'].type == 'gpu'", - device=_EXAMPLE_GPU, + device={"driver": "gpu.example.com", "attributes": {"type": {"string": "gpu"}}, "capacity": {}}, want=True, ), Case( - name="docs: extended-resource DeviceClass selector rejects other driver", + name="DocsDeviceClassRejects", + reason="The docs' extended-resource DeviceClass selector rejects a device of another driver.", expr="device.driver == 'gpu.example.com' && device.attributes['gpu.example.com'].type == 'gpu'", - device=_NIC, + device={"driver": "nic.nvidia.com", "attributes": {"linkType": {"string": "infiniband"}}, "capacity": {}}, want=False, ), Case( - name="docs: ResourceClaim type+memory selector matches", + name="DocsResourceClaim", + reason="The docs' ResourceClaim selector for a 64Gi GPU matches a GPU with 64Gi of memory.", expr=( 'device.attributes["driver.example.com"].type == "gpu" && ' 'device.capacity["driver.example.com"].memory == quantity("64Gi")' ), - device=_GPU_64GI, + device={ + "driver": "driver.example.com", + "attributes": {"type": {"string": "gpu"}}, + "capacity": {"memory": {"value": "64Gi"}}, + }, want=True, ), Case( - name="docs: ResourceClaim type+memory selector rejects wrong memory", + name="DocsResourceClaimRejects", + reason="The docs' ResourceClaim selector for a 64Gi GPU rejects a GPU with 32Gi of memory.", expr=( 'device.attributes["driver.example.com"].type == "gpu" && ' 'device.capacity["driver.example.com"].memory == quantity("64Gi")' ), - device=_device( - driver="driver.example.com", - attributes={"type": {"string": "gpu"}}, - capacity={"memory": {"value": "32Gi"}}, - ), + device={ + "driver": "driver.example.com", + "attributes": {"type": {"string": "gpu"}}, + "capacity": {"memory": {"value": "32Gi"}}, + }, want=False, ), ] @@ -308,10 +457,10 @@ class Case: def test_matches(case: Case) -> None: """A DRA CEL selector matches a device as it does upstream.""" got = cel.Program(case.expr).matches(case.device) - assert got == case.want + assert got == case.want, case.reason -def test_compile_invalid_expression_raises() -> None: +def test_compile_error() -> None: """A malformed expression fails to compile.""" with pytest.raises(cel.CELCompileError, match=r"not \) valid \("): cel.Program("not ) valid (") diff --git a/functions/compose-model-deployment/tests/test_fn.py b/functions/compose-model-deployment/tests/test_fn.py index f937959ab..6a39f27ad 100644 --- a/functions/compose-model-deployment/tests/test_fn.py +++ b/functions/compose-model-deployment/tests/test_fn.py @@ -16,9 +16,8 @@ import asyncio import dataclasses -import datetime import json -from typing import Any +from typing import Any, Literal import pytest from crossplane.function import resource @@ -27,1698 +26,2284 @@ from google.protobuf import duration_pb2 as durationpb from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb -from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 -from models.ai.modelplane.modelcache import v1alpha1 as mcv1alpha1 from models.ai.modelplane.modeldeployment import v1alpha1 from models.ai.modelplane.modelreplica import v1alpha1 as mrv1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -# The selector used on the deployment's single GPU request, echoed verbatim -# into each resolved device request. -_GPU_CEL = 'device.driver == "gpu.nvidia.com"' -# A fixed transition time keeps observed conditions deterministic. -_TRANSITION_TIME = datetime.datetime(2025, 1, 1, tzinfo=datetime.UTC) - -# The resolved DRA device requests the scheduler stamps onto each ModelReplica -# member, derived from the deployment's nodeSelector matched against the -# cluster's GPU device (deviceClassName gpu.nvidia.com). -_DEVICE_REQUESTS = [ - { - "name": "gpu", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "selectors": [{"cel": _GPU_CEL}], - } -] +@dataclasses.dataclass +class ComposeCase: + """A test case for RunFunction.""" -# The single Standalone-member engine every fixture deployment uses. -_ENGINE = v1alpha1.Engine( - name="main", - members=[ - v1alpha1.Member( - role="Standalone", - nodeSelector=v1alpha1.NodeSelector( - devices=[v1alpha1.Device(name="gpu", count=1, selectors=[v1alpha1.Selector(cel=_GPU_CEL)])], - ), - template=v1alpha1.Template( - spec=v1alpha1.Spec( - containers=[ - v1alpha1.Container( - name="engine", - image="vllm/vllm-openai:latest", - args=["--model=Qwen/Qwen3-0.6B"], - ), - ], - ), - ), - ), - ], -) + name: str + reason: str + req: fnv1.RunFunctionRequest + want: fnv1.RunFunctionResponse -# A Standalone engine with no container args, for the co-location case. -_ENGINE_NO_ARGS = v1alpha1.Engine( - name="main", - members=[ - v1alpha1.Member( - role="Standalone", - nodeSelector=v1alpha1.NodeSelector( - devices=[v1alpha1.Device(name="gpu", count=1, selectors=[v1alpha1.Selector(cel=_GPU_CEL)])], - ), - template=v1alpha1.Template( - spec=v1alpha1.Spec(containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")]), - ), - ), - ], -) +@dataclasses.dataclass +class ResolveRequiredCase: + """A test case for fn.resolve_required.""" -def _replica_engines(*, args: bool = True) -> list: - """The composed ModelReplica's spec.engines: one engine whose Standalone - member carries the matched pool and its resolved device requests. + name: str + reason: str + req: fnv1.RunFunctionRequest + # resolve_required's name parameter, renamed so it doesn't collide with + # the case's own name. + requirement: str + want: tuple[fn.Resolution, dict | None] - args toggles the engine container's --model arg, matching the fixture - deployment a want is built from. - Every container carries MODELPLANE_SERVED_MODEL_NAME, ahead of any env the - user wrote, so an arg can reference it. It's how an engine comes up under the - name Modelplane routes to instead of Modelplane having to be told what the - engine was started with. - """ - container: dict[str, Any] = { - "name": "engine", - "image": "vllm/vllm-openai:latest", - "env": [{"name": "MODELPLANE_SERVED_MODEL_NAME", "value": "ml-team/my-model"}], - } - if args: - container["args"] = ["--model=Qwen/Qwen3-0.6B"] - return [ - { - "name": "main", - "copies": 1, - "members": [ - { - # No worker block: it's only set on Worker members, and - # the XRD deliberately has no schema default (defaults - # apply before CEL validation, which forbids worker on a - # Standalone). - "role": "Standalone", - "nodePoolName": "default", - "deviceRequests": _DEVICE_REQUESTS, - "template": {"spec": {"containers": [container]}}, - } - ], - } - ] +@dataclasses.dataclass +class InjectNameCase: + """A test case for fn._inject_served_model_name.""" + name: str + reason: str + template: mrv1alpha1.Template + served: str + want: mrv1alpha1.Template -# The composed spec.engines for the args-bearing fixture deployment, shared by -# most wants below. -_REPLICA_ENGINES = _replica_engines() -_REPLICA_ENGINES_NO_ARGS = _replica_engines(args=False) -# The composed spec.engines for a PrefillDecode deployment: the standard -# single-GPU Standalone engine, one marked Prefill and one Decode. -_PD_REPLICA_ENGINES = [ - {**_replica_engines()[0], "name": "prefill", "phase": "Prefill"}, - {**_replica_engines()[0], "name": "decode", "phase": "Decode"}, -] +@dataclasses.dataclass +class ServedModelNameCase: + """A test case for fn.served_model_name.""" -# A one-replica deployment requesting a single GPU. Reused across most cases. -_XR = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE])), - ), -).model_dump(exclude_none=True, mode="json") + name: str + reason: str + namespace: str + deployment: str + want: str -def _cluster( - name: str, +def _model_deployment( *, - ready: bool = True, - hostname: str | None = "cluster.clusters.example.com", - nodes: int = 2, - placement_labels: dict[str, str] | None = None, - cache_storage: bool = False, -) -> dict: - """An InferenceCluster input fixture, dumped to a dict. - - A ready cluster has a Ready=True condition and a gateway hostname. ready=False - flips the condition to Unavailable; hostname=None drops the gateway entirely - (mirroring an offline cluster). nodes=0 yields a pool with no capacity. - cache_storage reports an RWX StorageClass in status.cache, which a cluster - needs to stage a ModelCache. - """ - return icv1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta(name=name), - spec=icv1alpha1.Spec( - cluster=icv1alpha1.Cluster( - source="Existing", - existing=icv1alpha1.Existing(secretRef=icv1alpha1.SecretRef(name="k")), - ), - placement=( - icv1alpha1.Placement(metadata=icv1alpha1.Metadata(labels=placement_labels)) - if placement_labels - else None + replicas: int, + template_labels: dict[str, str] | None, + cluster_selector: v1alpha1.ClusterSelector | None, + model_cache: str | None, + serving_mode: Literal["Unified", "PrefillDecode"] | None, + engines: list[dict[str, Any]], + args: list[str] | None, +) -> fnv1.Resource: + """The observed ModelDeployment my-model, each of its engines running one Standalone GPU member.""" + member = v1alpha1.Member( + role="Standalone", + nodeSelector=v1alpha1.NodeSelector( + devices=[ + v1alpha1.Device( + name="gpu", + count=1, + selectors=[v1alpha1.Selector(cel='device.driver == "gpu.nvidia.com"')], + ), + ], + ), + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest", args=args)], ), ), - status=icv1alpha1.Status( - conditions=[ - icv1alpha1.Condition( - type="Ready", - status="True" if ready else "False", - reason="Available" if ready else "Unavailable", - lastTransitionTime=_TRANSITION_TIME, - ) - ], - gateway=icv1alpha1.Gateway(address="10.0.0.1", hostname=hostname) if hostname else None, - cache=icv1alpha1.CacheModel(storageClassName="rwx") if cache_storage else None, - providerConfigRef=icv1alpha1.ProviderConfigRef(name=name), - gpuPools=[ - icv1alpha1.GpuPool( - name="default", - nodes=nodes, - devices=[ - icv1alpha1.Device( - name="gpu", - claim="DRA", - driver="gpu.nvidia.com", - deviceClassName="gpu.nvidia.com", - count=1, - ) - ], - ) - ], + ) + xr = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=replicas, + template=v1alpha1.TemplateModel( + metadata=v1alpha1.Metadata(labels=template_labels) if template_labels is not None else None, + spec=v1alpha1.SpecModel( + clusterSelector=cluster_selector, + modelCacheRef=v1alpha1.ModelCacheRef(name=model_cache) if model_cache is not None else None, + serving=v1alpha1.Serving(mode=serving_mode) if serving_mode is not None else None, + engines=[v1alpha1.Engine(**engine, members=[member]) for engine in engines], + ), + ), ), - ).model_dump(exclude_none=True, mode="json") - - -# A ready cluster with a two-node GPU pool. Reused across most cases. -_CLUSTER_A = _cluster("cluster-a") - -# cluster-a with the RWX storage a ModelCache needs. -_CLUSTER_A_CACHE = _cluster("cluster-a", cache_storage=True) - - -def _cache(name: str, *, match_labels: dict[str, str] | None = None) -> dict: - """Build an observed ModelCache in the ml-team namespace. + ) + return fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json", by_alias=True))) - match_labels, when given, sets spec.clusterSelector.matchLabels - the - footprint the deployment scheduler intersects with its own selector. - """ - selector = mcv1alpha1.ClusterSelector(matchLabels=match_labels) if match_labels else None - return mcv1alpha1.ModelCache( - metadata=metav1.ObjectMeta(name=name, namespace="ml-team"), - spec=mcv1alpha1.Spec( - source="HuggingFace", - huggingFace=mcv1alpha1.HuggingFace(repo="Qwen/Qwen2.5-7B", sizeGiB=20), - clusterSelector=selector, - ), - ).model_dump(exclude_none=True, mode="json") +def _desired_model_deployment(*, total_replicas: int, ready_replicas: int, ready: fnv1.Ready) -> fnv1.Resource: + """The desired ModelDeployment, with how many replicas it scheduled and how many are ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": total_replicas, "ready": ready_replicas}}}), + ready=ready, + ) -def _replica_status(replica: dict, *, ready: bool) -> dict: - """Return a copy of an observed ModelReplica with a Ready condition. - compose_endpoints gates each ModelEndpoint on its ModelReplica reporting - Ready=True - the replica's engines are serving and its remote Service and - HTTPRoute exist - so the endpoint never advertises a backend still warming - up (#102). Tests stamp the condition on the observed replica to drive that - gate. - """ - replica = dict(replica) - replica["status"] = { +def _cluster( + *, + name: str, + ready: bool, + gateway_hostname: str | None, + nodes: int, + cache_storage: bool, + placement_labels: dict[str, str] | None, +) -> fnv1.Resource: + """A required InferenceCluster with one pool of single-GPU nodes.""" + spec: dict[str, Any] = { + "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k", "key": "kubeconfig"}}}, + "stack": "Standard", + } + if placement_labels is not None: + spec["placement"] = {"metadata": {"labels": placement_labels}} + status: dict[str, Any] = { "conditions": [ { "type": "Ready", "status": "True" if ready else "False", - "reason": "Available" if ready else "Creating", + "reason": "Available" if ready else "Unavailable", "lastTransitionTime": "2025-01-01T00:00:00Z", } - ] + ], + "providerConfigRef": {"name": name}, + "gpuPools": [ + { + "name": "default", + "nodes": nodes, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + } + ], + } + ], + } + if gateway_hostname is not None: + status["gateway"] = {"address": "10.0.0.1", "hostname": gateway_hostname} + if cache_storage: + status["cache"] = {"storageClassName": "rwx"} + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": name}, + "spec": spec, + "status": status, + } + ) + ) + + +def _cache(*, cluster_selector: dict | None) -> fnv1.Resource: + """The ModelCache qwen, which a deployment references by modelCacheRef.""" + spec: dict[str, Any] = { + "source": "HuggingFace", + "huggingFace": {"repo": "Qwen/Qwen2.5-7B", "sizeGiB": 20}, } - return replica + if cluster_selector is not None: + spec["clusterSelector"] = cluster_selector + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelCache", + "metadata": {"name": "qwen", "namespace": "ml-team"}, + "spec": spec, + } + ) + ) -# An existing ModelReplica pinned to cluster-a, observed across cases 5 and 6. -_EXISTING_REPLICA = mrv1alpha1.ModelReplica( - metadata=metav1.ObjectMeta( - name="my-model-5ab63", - namespace="ml-team", - labels={ - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", +def _observed_replica(*, ready: bool | None) -> fnv1.Resource: + """The ModelReplica my-model has on cluster-a, with no Ready condition if ready is None.""" + replica: dict[str, Any] = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, }, - ), - spec=mrv1alpha1.SpecModel( - clusterName="cluster-a", - engines=[ - mrv1alpha1.Engine( - name="main", - copies=1, - members=[ - mrv1alpha1.Member( - role="Standalone", - nodePoolName="default", - deviceRequests=[ - mrv1alpha1.DeviceRequest( - name="gpu", - deviceClassName="gpu.nvidia.com", - count=1, - selectors=[mrv1alpha1.Selector(cel=_GPU_CEL)], - ), - ], - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")], - ), - ), - ), - ], - ), - ], - ), -).model_dump(exclude_none=True, mode="json") - -# The ModelReplica a deployment that references ModelCache qwen composes on -# cluster-b, as the first replica there. -_CACHED_REPLICA_B = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-f0b76", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-b", - "modelplane.ai/replica-index": "0", + "spec": { + "clusterName": "cluster-a", + "engines": [ + { + "name": "main", + "copies": 1, + "members": [ + { + "role": "Standalone", + "nodePoolName": "default", + "deviceRequests": [ + { + "name": "gpu", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [{"cel": 'device.driver == "gpu.nvidia.com"'}], + } + ], + "template": { + "spec": {"containers": [{"name": "engine", "image": "vllm/vllm-openai:latest"}]} + }, + } + ], + } + ], }, - }, - "spec": {"clusterName": "cluster-b", "modelCacheRef": {"name": "qwen"}, "engines": _REPLICA_ENGINES}, -} + } + if ready is not None: + replica["status"] = { + "conditions": [ + { + "type": "Ready", + "status": "True" if ready else "False", + "reason": "Available" if ready else "Creating", + "lastTransitionTime": "2025-01-01T00:00:00Z", + } + ] + } + return fnv1.Resource(resource=resource.dict_to_struct(replica)) -# The requirements selectors every want echoes back. Both are bare selectors -# matching all resources of the kind. -_CLUSTER_SEL = fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster") -_REPLICA_SEL = fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica") + +def _observed_endpoint() -> fnv1.Resource: + """The ModelEndpoint my-model has for its replica on cluster-a, as observed.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, + } + ) + ) -def _req( - xr: dict, +def _composed_replica( *, - clusters: list[dict] | None = None, - replicas: list[dict] | None = None, - observed: dict | None = None, - cache: dict | None = None, - cache_resolved_empty: bool = False, -) -> fnv1.RunFunctionRequest: - """Build a RunFunctionRequest with the standard required_resources. - - clusters and replicas populate the "clusters" and "all-replicas" required - resources respectively; both keys are always present (empty when there are - no items). cache, when given, populates the "cache" required resource the - function declares for a deployment that sets modelCacheRef. - cache_resolved_empty marks the "cache" requirement resolved-but-empty (the - key present with no items, i.e. the cache doesn't exist) - distinct from - omitting it, which leaves the requirement unresolved. observed populates - observed.resources. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), - resources={key: fnv1.Resource(resource=resource.dict_to_struct(r)) for key, r in (observed or {}).items()}, + name: str, + cluster: str, + labels: dict[str, str], + model_cache: str | None, + serving_mode: str | None, + engines: list[dict[str, str]], + args: list[str] | None, + ready: fnv1.Ready, +) -> fnv1.Resource: + """A composed ModelReplica of my-model, each of its engines running one Standalone GPU member.""" + container: dict[str, Any] = {"name": "engine", "image": "vllm/vllm-openai:latest"} + if args is not None: + container["args"] = args + # Every container carries MODELPLANE_SERVED_MODEL_NAME, ahead of any env the + # user wrote, so an arg can reference it. It's how an engine comes up under + # the name Modelplane routes to, instead of Modelplane having to be told + # what it was started with. + container["env"] = [{"name": "MODELPLANE_SERVED_MODEL_NAME", "value": "ml-team/my-model"}] + member = { + # No worker block: it's only set on Worker members, and the XRD + # deliberately has no schema default (defaults apply before CEL + # validation, which forbids worker on a Standalone). + "role": "Standalone", + "nodePoolName": "default", + # The deployment's nodeSelector matched against the cluster's GPU device + # (deviceClassName gpu.nvidia.com), with its CEL selector echoed + # verbatim. + "deviceRequests": [ + { + "name": "gpu", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [{"cel": 'device.driver == "gpu.nvidia.com"'}], + } + ], + "template": {"spec": {"containers": [container]}}, + } + spec: dict[str, Any] = {"clusterName": cluster} + if model_cache is not None: + spec["modelCacheRef"] = {"name": model_cache} + if serving_mode is not None: + spec["serving"] = {"mode": serving_mode} + spec["engines"] = [{**engine, "copies": 1, "members": [member]} for engine in engines] + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": {"name": name, "namespace": "ml-team", "labels": labels}, + "spec": spec, + } ), + ready=ready, ) - if clusters: - for c in clusters: - req.required_resources["clusters"].items.append(fnv1.Resource(resource=resource.dict_to_struct(c))) - else: - req.required_resources["clusters"].SetInParent() - if replicas: - for r in replicas: - req.required_resources["all-replicas"].items.append(fnv1.Resource(resource=resource.dict_to_struct(r))) - else: - req.required_resources["all-replicas"].SetInParent() - if cache is not None: - req.required_resources["cache"].items.append(fnv1.Resource(resource=resource.dict_to_struct(cache))) - elif cache_resolved_empty: - req.required_resources["cache"].SetInParent() - return req - -def _want( - resp: fnv1.RunFunctionResponse, - *, - cluster_labels: dict[str, str] | None = None, - cache_name: str | None = None, -) -> fnv1.RunFunctionResponse: - """Attach the requirements selectors a reconcile echoes back. - cluster_labels narrows the "clusters" selector (the intersection of the - deployment's and the referenced cache's clusterSelectors). cache_name adds - the "cache" selector the function declares for a referenced ModelCache. - """ - cluster_sel = fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster") - if cluster_labels: - cluster_sel.match_labels.labels.update(cluster_labels) - resp.requirements.resources["clusters"].CopyFrom(cluster_sel) - resp.requirements.resources["all-replicas"].CopyFrom(_REPLICA_SEL) - if cache_name is not None: - resp.requirements.resources["cache"].CopyFrom( - fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="ModelCache", - match_name=cache_name, - namespace="ml-team", - ) +def _composed_endpoint(*, labels: dict[str, str]) -> fnv1.Resource: + """The ModelEndpoint for my-model's replica on cluster-a, as composed.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": {"name": "my-model-5ab63", "namespace": "ml-team", "labels": labels}, + "spec": { + "origin": "https://cluster.clusters.example.com", + "api": {"schema": "OpenAI", "prefix": "/ml-team/my-model-5ab63/v1"}, + "model": "ml-team/my-model", + }, + } ) - return resp - - -@dataclasses.dataclass -class Case: - """A test case for compose-model-deployment.""" - - name: str - req: fnv1.RunFunctionRequest - want: fnv1.RunFunctionResponse - + ) -def _compose_cases() -> list[Case]: - """The cases for test_compose, and the deployments they share.""" - # A deployment that sets spec.modelCacheRef. - xr_cached = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - spec=v1alpha1.SpecModel( - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), - engines=[_ENGINE], - ) - ), - ), - ).model_dump(exclude_none=True, mode="json") - # A cached deployment that also sets its own clusterSelector, so the - # scheduler intersects it with the cache's footprint. - xr_cached_selector = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - spec=v1alpha1.SpecModel( - clusterSelector=v1alpha1.ClusterSelector(matchLabels={"region": "us-east"}), - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), - engines=[_ENGINE], - ) - ), - ), - ).model_dump(exclude_none=True, mode="json") +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - # A two-replica deployment (no container args) for the co-location case. - xr_two = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=2, - template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE_NO_ARGS])), - ), - ).model_dump(exclude_none=True, mode="json") - # A disaggregated (PrefillDecode) deployment: a Prefill and a Decode engine. - xr_pd = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - spec=v1alpha1.SpecModel( - serving=v1alpha1.Serving(mode="PrefillDecode"), - engines=[ - _ENGINE.model_copy(update={"name": "prefill", "phase": "Prefill"}), - _ENGINE.model_copy(update={"name": "decode", "phase": "Decode"}), - ], - ) +COMPOSE_CASES = [ + # Routing must not advertise a backend whose pods are still warming up + # (#102). + ComposeCase( + name="FreshlyScheduled", + reason="A newly scheduled replica composes without an endpoint, because it isn't observed Ready yet.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + }, ), - ).model_dump(exclude_none=True, mode="json") - - # A deployment parked at zero replicas. - xr_zero = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=0, - template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE])), - ), - ).model_dump(exclude_none=True, mode="json") - - return [ - Case( - # First reconcile: the replica is composed but not yet observed - # Ready, so its endpoint is withheld - routing must not advertise - # a backend whose pods are still warming up (#102). - name="freshly scheduled replica composes no endpoint until ready", - req=_req(_XR, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, - }, - } - ), - ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) + }, ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], ), - Case( - # A replica that has gone not-Ready (e.g. a crash-loop after - # once serving) has its endpoint withdrawn: the previously - # observed endpoint is absent from desired, so Crossplane - # deletes it and traffic stops routing to the dead backend - # (#102). Omitting it from desired - not composing it - is what - # drives the deletion. - name="not-ready replica withdraws its endpoint", - req=_req( - _XR, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={ - "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=False), - "endpoint-cluster-a-0": { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, - }, + ), + # A replica can go not Ready after once serving, say in a crash-loop. + # Deleting its endpoint stops traffic routing to the dead backend (#102). + ComposeCase( + name="ReplicaNotReady", + reason="A replica observed not Ready loses its endpoint, which is left out of desired so Crossplane deletes it.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=False), + "endpoint-cluster-a-0": _observed_endpoint(), }, ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, - }, - } - ), - ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - context=structpb.Struct(), - ) + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], ), - Case( - name="no clusters produces warning", - req=_req(_XR, clusters=[]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoClusters", - ), - ], - results=[ - fnv1.Result(severity=fnv1.SEVERITY_WARNING, message="No InferenceClusters found"), - ], - context=structpb.Struct(), - ) + ), + ComposeCase( + name="NoClusters", + reason="With no InferenceClusters, the function composes nothing, warns, and reports NoClusters.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), ), + required_resources={ + "clusters": fnv1.Resources(), + "all-replicas": fnv1.Resources(), + }, ), - Case( - name="insufficient capacity produces no replicas", - req=_req(_XR, clusters=[_cluster("cluster-a", nodes=0)]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_FALSE, - ), - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="InsufficientCapacity", - message="0 of 1 replicas scheduled (checked 1 clusters)", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoReplicasScheduled", - ), - ], - context=structpb.Struct(), - ) + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + # Inline rather than _desired_model_deployment, because with no + # cluster the function returns before it writes any status. + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_WARNING, message="No InferenceClusters found"), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoClusters", + ), + ], ), - Case( - # Zero desired parks the deployment before resolve_inputs runs: - # no requirements are declared (the want carries none), nothing - # is composed, and both conditions read True with the - # NoReplicasDesired reason rather than a capacity failure. - name="scaled to zero composes nothing and reports NoReplicasDesired", - req=_req(xr_zero), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_TRUE, - ), + ), + ComposeCase( + name="InsufficientCapacity", + reason="A cluster with no nodes leaves the replica unscheduled, reported as InsufficientCapacity.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", - ), - ], - context=structpb.Struct(), ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=0, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + }, ), - Case( - # Scaling an existing deployment to zero: the observed replica - # and endpoint are absent from desired (pruned), and the - # transition is announced while they still exist. - name="scale to zero prunes observed replicas and emits an event", - req=_req( - xr_zero, - observed={ - "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True), - "endpoint-cluster-a-0": { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=0, ready_replicas=0, ready=fnv1.READY_FALSE), + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_TRUE, - ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="InsufficientCapacity", + message="0 of 1 replicas scheduled (checked 1 clusters)", ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scaled to zero: removing all replicas", - ), - ], - context=structpb.Struct(), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReplicasScheduled", + ), + ], + ), + ), + # Zero desired parks the deployment before resolve_inputs runs, so it + # reads as NoReplicasDesired rather than as a capacity failure. + ComposeCase( + name="ZeroReplicas", + reason="A deployment of zero replicas declares no requirements, composes nothing, and reports NoReplicasDesired.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=0, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources(), + "all-replicas": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=0, ready_replicas=0, ready=fnv1.READY_TRUE), ), + context=structpb.Struct(), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + ], ), - Case( - name="ready replica is preserved and keeps its endpoint", - req=_req( - _XR, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={ - "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True), - "endpoint-cluster-a-0": { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, - }, + ), + # The scale-down is announced while the replica and endpoint still exist. + ComposeCase( + name="ScaleToZero", + reason="Scaling a running deployment to zero prunes its replica and endpoint, and announces the scale-down.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=0, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=True), + "endpoint-cluster-a-0": _observed_endpoint(), }, ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 1}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "endpoint-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "origin": "https://cluster.clusters.example.com", - "api": { - "schema": "OpenAI", - "prefix": "/ml-team/my-model-5ab63/v1", - }, - "model": "ml-team/my-model", - }, - } - ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="AllReplicasReady", - message="1 of 1 ready", - ), - ], - context=structpb.Struct(), - ) + required_resources={ + "clusters": fnv1.Resources(), + "all-replicas": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=0, ready_replicas=0, ready=fnv1.READY_TRUE), ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scaled to zero: removing all replicas", + ), + ], + context=structpb.Struct(), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + ], ), - Case( - name="offline pinned cluster keeps replica but drops endpoint", - req=_req( - _XR, - clusters=[_cluster("cluster-a", ready=False, hostname=None)], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _EXISTING_REPLICA}, - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, - }, - } - ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - context=structpb.Struct(), - ) + ), + ComposeCase( + name="ReplicaReady", + reason="A replica observed Ready keeps its place and its endpoint, and the deployment reports all replicas ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=True), + "endpoint-cluster-a-0": _observed_endpoint(), + }, ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + }, ), - Case( - name="deleted pinned cluster triggers replica re-placement", - req=_req( - _XR, - clusters=[_cluster("cluster-b", hostname="cluster-b.clusters.example.com")], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-b-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-f0b76", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-b", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-b", - "engines": _REPLICA_ENGINES, - }, - } - ), - ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=1, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_TRUE, ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-b", - ), - ], - context=structpb.Struct(), - ) + "endpoint-cluster-a-0": _composed_endpoint( + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + } + ), + }, ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="AllReplicasReady", + message="1 of 1 ready", + ), + ], ), - Case( - name="modelCacheRef is propagated onto the composed replica", - req=_req(xr_cached, clusters=[_CLUSTER_A_CACHE], cache=_cache("qwen")), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "modelCacheRef": {"name": "qwen"}, - "engines": _REPLICA_ENGINES, - }, - } - ), - ), + ), + # Either the missing hostname or the missing Ready condition alone would + # hold back the endpoint, so this pins neither. + ComposeCase( + name="ClusterOffline", + reason="A replica stays on its cluster while the cluster is offline, and gets no endpoint, with no gateway hostname on the cluster and no Ready condition on the replica.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=None), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=False, + gateway_hostname=None, + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", ), - cache_name="qwen", + ], + ), + ), + ComposeCase( + name="ClusterDeleted", + reason="A replica whose cluster no longer exists is re-placed on another cluster.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=True), + }, ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-b", + ready=True, + gateway_hostname="cluster-b.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + }, ), - Case( - # The cache stages only to a subset of clusters; the scheduler - # intersects the cache's footprint with the deployment's own - # clusterSelector so replicas never land where the cache isn't. - name="cache clusterSelector is intersected with the deployment's", - req=_req( - xr_cached_selector, - clusters=[_CLUSTER_A_CACHE], - cache=_cache("qwen", match_labels={"tier": "gpu"}), - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "modelCacheRef": {"name": "qwen"}, - "engines": _REPLICA_ENGINES, - }, - } - ), - ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-b-0": _composed_replica( + name="my-model-f0b76", + cluster="cluster-b", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-b", + "modelplane.ai/replica-index": "0", }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-b", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + ), + ), + ComposeCase( + name="ModelCacheRef", + reason="A deployment's modelCacheRef is copied onto its replica, and ModelCacheResolved reports the cache resolved.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], ), - cluster_labels={"region": "us-east", "tier": "gpu"}, - cache_name="qwen", ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=True, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + "cache": fnv1.Resources(items=[_cache(cluster_selector=None)]), + }, ), - Case( - # compose-model-cache stages only onto clusters that report cache - # storage, so cluster-a, which reports none, can't host the - # replica's PVC. The replica lands on cluster-b, though cluster-a - # would win the tiebreak by name. - name="a cached replica lands only on a cluster with cache storage", - req=_req( - xr_cached, - clusters=[_CLUSTER_A, _cluster("cluster-b", cache_storage=True)], - cache=_cache("qwen"), - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-b-0": fnv1.Resource(resource=resource.dict_to_struct(_CACHED_REPLICA_B)), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-b", - ), - ], - context=structpb.Struct(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", ), - cache_name="qwen", + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + "cache": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelCache", + match_name="qwen", + namespace="ml-team", + ), + }, ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], ), - Case( - # A replica is running on cluster-a, which has no cache storage, - # say because the bug this guards against put it there. Its PVC - # never appears, so it's dropped and re-placed on cluster-b, the - # way a replica is when the cache's selector stops matching. - name="a running cached replica on a cluster without cache storage is re-placed", - req=_req( - xr_cached, - clusters=[_CLUSTER_A, _cluster("cluster-b", cache_storage=True)], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - cache=_cache("qwen"), - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-b-0": fnv1.Resource(resource=resource.dict_to_struct(_CACHED_REPLICA_B)), + ), + # The cache stages only onto the clusters its selector matches, so a + # replica must never land where the cache isn't. + ComposeCase( + name="CacheClusterSelector", + reason="The clusters requirement selects the labels of both the deployment's and the cache's clusterSelector.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=v1alpha1.ClusterSelector(matchLabels={"region": "us-east"}), + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=True, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + "cache": fnv1.Resources(items=[_cache(cluster_selector={"matchLabels": {"tier": "gpu"}})]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-b", - ), - ], - context=structpb.Struct(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", ), - cache_name="qwen", + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_labels=fnv1.MatchLabels(labels={"region": "us-east", "tier": "gpu"}), + ), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + "cache": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelCache", + match_name="qwen", + namespace="ml-team", + ), + }, ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], ), - Case( - # The only candidate has no cache storage, so the cache can't - # stage there and nothing is placed. ReplicasScheduled says why - # rather than blaming capacity. - name="no candidate with cache storage places nothing", - req=_req(xr_cached, clusters=[_CLUSTER_A], cache=_cache("qwen")), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_FALSE, - ), + ), + # compose-model-cache stages only onto clusters that report cache + # storage, so a cluster that reports none can't host the replica's PVC. + ComposeCase( + name="CacheStorageRequired", + reason="A cached replica skips cluster-a, which has no cache storage, for cluster-b, though cluster-a would win the tiebreak by name.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ), + _cluster( + name="cluster-b", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=True, + placement_labels=None, + ), + ] + ), + "all-replicas": fnv1.Resources(), + "cache": fnv1.Resources(items=[_cache(cluster_selector=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-b-0": _composed_replica( + name="my-model-f0b76", + cluster="cluster-b", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-b", + "modelplane.ai/replica-index": "0", + }, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoCacheStorage", - message="0 of 1 replicas scheduled: no candidate cluster has storage for ModelCache qwen", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoReplicasScheduled", - ), - ], - context=structpb.Struct(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-b", ), - cache_name="qwen", + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + "cache": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelCache", + match_name="qwen", + namespace="ml-team", + ), + }, ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], ), - Case( - # A referenced cache Crossplane hasn't fetched yet leaves the - # footprint unknown. With no replicas to retain, the function - # holds off placing any rather than risk landing them outside the - # footprint: fill is suppressed, so nothing is composed, and - # ModelCacheResolved=False (Unresolved) says why. The wait is - # transient and self-clearing, so it's a condition, not an event. - # The cluster and replica requirements are still declared so the - # cache can resolve alongside them. - name="unresolved cache suppresses new placement", - req=_req(xr_cached, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_FALSE, - ), + ), + # A replica placed before #186 was fixed, or before its cluster lost cache + # storage, can be running on a cluster without it. Its PVC never appears, + # so it's re-placed the way a replica is when the cache's selector stops + # matching. + ComposeCase( + name="RunningWithoutStorage", + reason="A cached replica running on a cluster without cache storage is re-placed on a cluster with it.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=True), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ), + _cluster( + name="cluster-b", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=True, + placement_labels=None, + ), + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + "cache": fnv1.Resources(items=[_cache(cluster_selector=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-b-0": _composed_replica( + name="my-model-f0b76", + cluster="cluster-b", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-b", + "modelplane.ai/replica-index": "0", + }, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelCacheUnresolved", - message="Waiting for ModelCache qwen", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="InsufficientCapacity", - message="0 of 1 replicas scheduled (checked 1 clusters)", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoReplicasScheduled", - ), - ], - context=structpb.Struct(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-b", ), - cache_name="qwen", + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + "cache": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelCache", + match_name="qwen", + namespace="ml-team", + ), + }, ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], ), - Case( - # The cache a live deployment depends on is deleted (the cache - # requirement resolves but matches nothing - ABSENT). The cache - # only matters when loading weights, which already happened, so - # its disappearance must not tear the deployment down: the - # existing replica is retained (retain ignores fill) even as - # ModelCacheResolved goes False (NotFound) and new placement is - # suppressed. - name="deleted cache retains existing replicas", - req=_req( - xr_cached, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - cache_resolved_empty=True, - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 1}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "modelCacheRef": {"name": "qwen"}, - "engines": _REPLICA_ENGINES, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "endpoint-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "origin": "https://cluster.clusters.example.com", - "api": { - "schema": "OpenAI", - "prefix": "/ml-team/my-model-5ab63/v1", - }, - "model": "ml-team/my-model", - }, - } - ), - ), + ), + ComposeCase( + name="NoCacheStorage", + reason="When no candidate cluster has cache storage nothing is placed, and ReplicasScheduled says so rather than blaming capacity.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + "cache": fnv1.Resources(items=[_cache(cluster_selector=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=0, ready_replicas=0, ready=fnv1.READY_FALSE), + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + "cache": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelCache", + match_name="qwen", + namespace="ml-team", + ), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoCacheStorage", + message="0 of 1 replicas scheduled: no candidate cluster has storage for ModelCache qwen", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReplicasScheduled", + ), + ], + ), + ), + # An unfetched cache leaves the footprint unknown, so with no replicas to + # retain the function holds off placing any rather than risk landing them + # outside it. The wait is transient and self-clearing, so it's a + # condition, not an event. The cluster and replica requirements are still + # declared so the cache can resolve alongside them. The request has no + # "cache" required resource, which is what marks it unresolved. + ComposeCase( + name="CacheUnresolved", + reason="Until Crossplane fetches the referenced cache no replica is placed, and ModelCacheResolved reports ModelCacheUnresolved.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=0, ready_replicas=0, ready=fnv1.READY_FALSE), + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + "cache": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelCache", + match_name="qwen", + namespace="ml-team", + ), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelCacheUnresolved", + message="Waiting for ModelCache qwen", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="InsufficientCapacity", + message="0 of 1 replicas scheduled (checked 1 clusters)", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReplicasScheduled", + ), + ], + ), + ), + # The "cache" requirement resolves but matches nothing. The cache only + # matters when loading weights, which already happened, so its + # disappearance must not tear the deployment down: the scheduler keeps a + # placed replica even while it may not place new ones. + ComposeCase( + name="CacheDeleted", + reason="When the referenced cache is deleted the running replica keeps its place and endpoint, and ModelCacheResolved reports ModelCacheNotFound.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=True), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + "cache": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=1, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_TRUE, ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelCacheNotFound", - message="ModelCache qwen not found; holding replica placement", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="AllReplicasReady", - message="1 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_WARNING, - message="ModelCache qwen not found; holding replica placement", - ), - ], - context=structpb.Struct(), + "endpoint-cluster-a-0": _composed_endpoint( + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + } + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="ModelCache qwen not found; holding replica placement", ), - cache_name="qwen", + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + "cache": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelCache", + match_name="qwen", + namespace="ml-team", + ), + }, ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelCacheNotFound", + message="ModelCache qwen not found; holding replica placement", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="AllReplicasReady", + message="1 of 1 ready", + ), + ], ), - Case( - name="two replicas co-locate on one cluster as distinct resources", - req=_req(xr_two, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 2, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES_NO_ARGS, - }, - } - ), - ), - "replica-cluster-a-1": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-609c5", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "1", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES_NO_ARGS, - }, - } - ), - ), + ), + ComposeCase( + name="TwoReplicasOneCluster", + reason="Two replicas placed on one cluster compose as two distinct ModelReplicas.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=2, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=None, + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=2, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=None, + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 2 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 2 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) + "replica-cluster-a-1": _composed_replica( + name="my-model-609c5", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "1", + }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 2 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 2 ready", + ), + ], ), - Case( - # PrefillDecode copies serving and each engine's phase onto the - # replica; the replica backend reads them to front the engines - # with an InferencePool + endpoint picker rather than a Service. - name="PrefillDecode copies serving and engine phases onto the replica", - req=_req(xr_pd, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "serving": {"mode": "PrefillDecode"}, - "engines": _PD_REPLICA_ENGINES, - }, - } - ), - ), + ), + # compose-model-replica reads serving and each engine's phase to pick + # disaggregated routing, which role-labels the phase engines and puts the + # pd-sidecar on decode. + ComposeCase( + name="PrefillDecode", + reason="A PrefillDecode deployment copies its serving mode and each engine's phase onto the replica.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode="PrefillDecode", + engines=[{"name": "prefill", "phase": "Prefill"}, {"name": "decode", "phase": "Decode"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache=None, + serving_mode="PrefillDecode", + engines=[ + {"name": "prefill", "phase": "Prefill"}, + {"name": "decode", "phase": "Decode"}, + ], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], ), - ] - - -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) -def test_compose(case: Case) -> None: - """The function fans out ModelReplicas and, once they're Ready, ModelEndpoints.""" - got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) - - -def _composed(resp: fnv1.RunFunctionResponse, kind: str) -> list[dict]: - """Composed desired resources of the given kind, as dicts.""" - result = [] - for r in resp.desired.resources.values(): - d = resource.struct_to_dict(r.resource) - if d.get("kind") == kind: - result.append(d) - return result - - -# spec.template.metadata.labels land on the composed ModelReplicas and -# ModelEndpoints, alongside the labels Modelplane manages. - - -def test_template_labels_stamped_on_replica_and_endpoint() -> None: - """Template labels land on the replica and endpoint beside the managed labels.""" - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - metadata=v1alpha1.Metadata(labels={"tier": "prod", "team": "search"}), - spec=v1alpha1.SpecModel(engines=[_ENGINE]), + ), + # The observed, Ready replica lets the endpoint compose this reconcile. + ComposeCase( + name="TemplateLabels", + reason="A deployment's template labels land on the replica and its endpoint, alongside the labels Modelplane manages.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels={"tier": "prod", "team": "search"}, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=True), + }, ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + }, ), - ).model_dump(exclude_none=True, mode="json") - # An observed, Ready replica lets the endpoint compose this reconcile. - req = _req( - xr, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - composed = _composed(got, "ModelReplica") + _composed(got, "ModelEndpoint") - assert len(composed) == 2, "expected one ModelReplica and one ModelEndpoint" - for obj in composed: - labels = obj["metadata"]["labels"] - assert labels.get("tier") == "prod" - assert labels.get("team") == "search" - assert labels.get("modelplane.ai/deployment") == "my-model" - assert labels.get("modelplane.ai/cluster") == "cluster-a" - assert labels.get("modelplane.ai/replica-index") == "0" - - -def test_template_labels_managed_labels_win_a_collision() -> None: - """A managed label beats a template label of the same key.""" - # The XRD's CEL rejects a template label under the modelplane.ai/ prefix, - # but the invariant lives in the function too: managed labels are stamped - # last, so a colliding label can't override them even if that CEL rule is - # relaxed or the function is reused elsewhere. - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - metadata=v1alpha1.Metadata(labels={"modelplane.ai/cluster": "wrong", "tier": "prod"}), - spec=v1alpha1.SpecModel(engines=[_ENGINE]), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=1, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "tier": "prod", + "team": "search", + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_TRUE, + ), + "endpoint-cluster-a-0": _composed_endpoint( + labels={ + "tier": "prod", + "team": "search", + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + } + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="AllReplicasReady", + message="1 of 1 ready", + ), + ], ), - ).model_dump(exclude_none=True, mode="json") - got = asyncio.run(fn.FunctionRunner().RunFunction(_req(xr, clusters=[_CLUSTER_A]), None)) - - replica = _composed(got, "ModelReplica")[0] - assert replica["metadata"]["labels"]["modelplane.ai/cluster"] == "cluster-a" - assert replica["metadata"]["labels"]["tier"] == "prod" - - -def test_resolve_required() -> None: - """resolve_required tells a found, a missing, and an unfetched requirement apart.""" - cache = {"apiVersion": "modelplane.ai/v1alpha1", "kind": "ModelCache", "metadata": {"name": "qwen"}} - - # PRESENT: the requirement resolved and matched a resource. - req = fnv1.RunFunctionRequest() - req.required_resources["cache"].items.append(fnv1.Resource(resource=resource.dict_to_struct(cache))) - assert fn.resolve_required(req, "cache") == (fn.Resolution.PRESENT, cache) - - # ABSENT: the requirement resolved but matched nothing (key present, no items). - req = fnv1.RunFunctionRequest() - req.required_resources["cache"].SetInParent() - assert fn.resolve_required(req, "cache") == (fn.Resolution.ABSENT, None) - - # UNRESOLVED: Crossplane has not fetched the requirement (key absent). - req = fnv1.RunFunctionRequest() - assert fn.resolve_required(req, "cache") == (fn.Resolution.UNRESOLVED, None) + ), + # The XRD's CEL rejects a template label under the modelplane.ai/ + # prefix, but the invariant lives in the function too: managed labels + # are stamped last, so they win even if that CEL rule is relaxed or the + # function is reused elsewhere. + ComposeCase( + name="ManagedLabelWins", + reason="A template label with the same key as a managed label is overridden by the managed one.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels={"modelplane.ai/cluster": "wrong", "tier": "prod"}, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "tier": "prod", + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + ), + ), + # Placement labels are the endpoint half of residency. A ModelService + # selects endpoints by label, so without them a region-scoped service + # can't select its own replicas, and nobody can label them by hand because + # Modelplane owns them. The gateway half is an InferenceGateway's + # serviceSelector. + ComposeCase( + name="PlacementLabels", + reason="A cluster's placement labels land on the replica composed there and on its endpoint.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=True), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels={"example.org/region": "eu"}, + ) + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=1, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "example.org/region": "eu", + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_TRUE, + ), + "endpoint-cluster-a-0": _composed_endpoint( + labels={ + "example.org/region": "eu", + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + } + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="AllReplicasReady", + message="1 of 1 ready", + ), + ], + ), + ), + # The cluster is the authority on where it is, so its placement + # labels are stamped after the deployment's own template labels. + ComposeCase( + name="PlacementLabelWins", + reason="A cluster's placement label overrides a template label of the same key.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels={"example.org/region": "wrong"}, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels={"example.org/region": "eu"}, + ) + ] + ), + "all-replicas": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "example.org/region": "eu", + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + ), + ), +] -# The name an engine is started under, and how it gets there. +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: ComposeCase) -> None: + """RunFunction fans out ModelReplicas and, once they're Ready, ModelEndpoints.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want), case.reason + + +RESOLVE_REQUIRED_CASES = [ + ResolveRequiredCase( + name="Present", + reason="A requirement that resolved and matched a resource is PRESENT, with that resource.", + req=fnv1.RunFunctionRequest( + required_resources={ + "cache": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelCache", + "metadata": {"name": "qwen"}, + } + ) + ) + ] + ), + }, + ), + requirement="cache", + want=( + fn.Resolution.PRESENT, + {"apiVersion": "modelplane.ai/v1alpha1", "kind": "ModelCache", "metadata": {"name": "qwen"}}, + ), + ), + # The SDK returns None for this and for a requirement Crossplane hasn't + # fetched alike. Only the requirement's key tells them apart. + ResolveRequiredCase( + name="Absent", + reason="A requirement that resolved but matched nothing is ABSENT.", + req=fnv1.RunFunctionRequest(required_resources={"cache": fnv1.Resources()}), + requirement="cache", + want=(fn.Resolution.ABSENT, None), + ), + ResolveRequiredCase( + name="Unresolved", + reason="A requirement Crossplane hasn't fetched is UNRESOLVED.", + req=fnv1.RunFunctionRequest(), + requirement="cache", + want=(fn.Resolution.UNRESOLVED, None), + ), +] -def test_served_model_name_goes_ahead_of_the_users_env() -> None: - """The served model name env var comes before the container's own env.""" - # Env expansion is left to right, so an arg or a later entry referencing - # $(MODELPLANE_SERVED_MODEL_NAME) only resolves if it's first. - template = mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container( - name="engine", - image="vllm/vllm-openai:latest", - env=[mrv1alpha1.EnvItem(name="HF_TOKEN", value="x")], - ) - ] - ) - ) - fn._inject_served_model_name(template, "ml-team/kimi-k2") - assert template.spec is not None - assert [(e.name, e.value) for e in template.spec.containers[0].env or []] == [ - ("MODELPLANE_SERVED_MODEL_NAME", "ml-team/kimi-k2"), - ("HF_TOKEN", "x"), - ] +@pytest.mark.parametrize("case", RESOLVE_REQUIRED_CASES, ids=lambda case: case.name) +def test_resolve_required(case: ResolveRequiredCase) -> None: + """resolve_required tells a found, a missing, and an unfetched requirement apart.""" + assert fn.resolve_required(case.req, case.requirement) == case.want, case.reason + + +INJECT_NAME_CASES = [ + # Env expansion is left to right, so an arg or a later entry + # referencing $(MODELPLANE_SERVED_MODEL_NAME) only resolves if it's + # first. + InjectNameCase( + name="UserEnv", + reason="The served model name goes ahead of the env the user wrote.", + template=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + env=[mrv1alpha1.EnvItem(name="HF_TOKEN", value="x")], + ) + ] + ) + ), + served="ml-team/kimi-k2", + want=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + env=[ + mrv1alpha1.EnvItem(name="MODELPLANE_SERVED_MODEL_NAME", value="ml-team/kimi-k2"), + mrv1alpha1.EnvItem(name="HF_TOKEN", value="x"), + ], + ) + ] + ) + ), + ), + # Modelplane decides this value. Honouring an override would let the + # engine answer to a name nothing routes to, which surfaces as a 404 + # from the engine rather than anything visible in status. + InjectNameCase( + name="UserOverride", + reason="A MODELPLANE_SERVED_MODEL_NAME the user wrote is replaced by the served model name.", + template=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + env=[mrv1alpha1.EnvItem(name="MODELPLANE_SERVED_MODEL_NAME", value="mine")], + ) + ] + ) + ), + served="ml-team/kimi-k2", + want=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + env=[mrv1alpha1.EnvItem(name="MODELPLANE_SERVED_MODEL_NAME", value="ml-team/kimi-k2")], + ) + ] + ) + ), + ), +] -def test_served_model_name_user_override_is_dropped() -> None: - """A user's own MODELPLANE_SERVED_MODEL_NAME is replaced, not kept.""" - # Modelplane decides this value. Honouring an override would let the engine - # answer to a name nothing routes to, which surfaces as a 404 from the - # engine rather than anything visible in status. - template = mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container( - name="engine", - image="vllm/vllm-openai:latest", - env=[mrv1alpha1.EnvItem(name="MODELPLANE_SERVED_MODEL_NAME", value="mine")], - ) - ] - ) - ) - fn._inject_served_model_name(template, "ml-team/kimi-k2") - assert template.spec is not None - assert [(e.name, e.value) for e in template.spec.containers[0].env or []] == [ - ("MODELPLANE_SERVED_MODEL_NAME", "ml-team/kimi-k2"), - ] +@pytest.mark.parametrize("case", INJECT_NAME_CASES, ids=lambda case: case.name) +def test_inject_name(case: InjectNameCase) -> None: + """_inject_served_model_name puts the served model name first in each container's env.""" + got = case.template.model_copy(deep=True) + fn._inject_served_model_name(got, case.served) + assert got.model_dump() == case.want.model_dump(), case.reason -def test_served_model_name_is_namespaced() -> None: - """served_model_name prefixes the deployment's name with its namespace.""" +SERVED_MODEL_NAME_CASES = [ # So two deployments in different namespaces can't collide, and a # ModelService can rewrite one name for a whole deployment. - assert fn.served_model_name("ml-team", "kimi-k2") == "ml-team/kimi-k2" - - -# A cluster's spec.placement.metadata.labels land on the ModelReplicas and -# ModelEndpoints composed there. -# -# This is the endpoint half of residency: a ModelService selects endpoints by -# label, so without it a region-scoped service can't select its own replicas, -# and nobody can label them by hand because Modelplane owns them. The gateway -# half is an InferenceGateway's serviceSelector. - - -def test_placement_labels_stamped_on_replica_and_endpoint() -> None: - """A cluster's placement labels land on the replica and endpoint composed there.""" - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE])), - ), - ).model_dump(exclude_none=True, mode="json") - req = _req( - xr, - clusters=[_cluster("cluster-a", placement_labels={"example.org/region": "eu"})], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - composed = _composed(got, "ModelReplica") + _composed(got, "ModelEndpoint") - assert len(composed) == 2, "expected one ModelReplica and one ModelEndpoint" - for obj in composed: - assert obj["metadata"]["labels"].get("example.org/region") == "eu" + ServedModelNameCase( + name="Namespaced", + reason="A deployment's served model name is its namespace and name, joined by a slash.", + namespace="ml-team", + deployment="kimi-k2", + want="ml-team/kimi-k2", + ), +] -def test_placement_labels_cluster_label_beats_a_template_label() -> None: - """A cluster's placement label beats a template label of the same key.""" - # The cluster is the authority on where it is, so its placement labels are - # stamped after the deployment's own template labels. - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - metadata=v1alpha1.Metadata(labels={"example.org/region": "wrong"}), - spec=v1alpha1.SpecModel(engines=[_ENGINE]), - ), - ), - ).model_dump(exclude_none=True, mode="json") - got = asyncio.run( - fn.FunctionRunner().RunFunction( - _req(xr, clusters=[_cluster("cluster-a", placement_labels={"example.org/region": "eu"})]), - None, - ) - ) - replica = _composed(got, "ModelReplica")[0] - assert replica["metadata"]["labels"]["example.org/region"] == "eu" +@pytest.mark.parametrize("case", SERVED_MODEL_NAME_CASES, ids=lambda case: case.name) +def test_served_model_name(case: ServedModelNameCase) -> None: + """served_model_name prefixes the deployment's name with its namespace.""" + assert fn.served_model_name(case.namespace, case.deployment) == case.want, case.reason diff --git a/functions/compose-model-deployment/tests/test_quantity.py b/functions/compose-model-deployment/tests/test_quantity.py index 5c0429415..594cd94cb 100644 --- a/functions/compose-model-deployment/tests/test_quantity.py +++ b/functions/compose-model-deployment/tests/test_quantity.py @@ -32,8 +32,9 @@ These expected values come from running the inputs through the real Kubernetes code, not from assertion by hand. When you change the quantity module or want to add a case, derive its expected value with the parity oracle in ./oracle (see -oracle/README.md) rather than reasoning about it - upstream has surprises (e.g. -binary-suffix overflow saturates to int64-max, so 8Ei == 10Ei). +the package comment in oracle/main.go) rather than reasoning about it - +upstream has surprises (e.g. binary-suffix overflow saturates to int64-max, so +8Ei == 10Ei). """ import dataclasses @@ -42,96 +43,282 @@ from function import cel, quantity -def _eval(expr: str) -> bool: - """Compile and evaluate a deviceless boolean CEL expression.""" - return cel.Program(expr).matches({}) - - @dataclasses.dataclass -class Case: +class QuantityCase: + """A test case for a quantity CEL expression.""" + name: str + reason: str expr: str want: bool @dataclasses.dataclass -class ParseErrCase: +class ParseRejectsCase: + """A test case for a string quantity.parse rejects.""" + name: str - input: str + reason: str + s: str + want: str QUANTITY_CASES = [ # parse + isQuantity. - Case(name="parse", expr='quantity("12Mi").compareTo(quantity("12Mi")) == 0', want=True), - Case(name="isQuantity int string", expr='isQuantity("20")', want=True), - Case(name="isQuantity megabytes", expr='isQuantity("20M")', want=True), - Case(name="isQuantity mebibytes", expr='isQuantity("20Mi")', want=True), - Case(name="isQuantity invalid suffix", expr='isQuantity("20Mo")', want=False), - Case(name="isQuantity passing regex bad suffix", expr='isQuantity("10Mm")', want=False), + QuantityCase( + name="Parse", + reason="quantity() parses 12Mi into a value that compares equal to 12Mi.", + expr='quantity("12Mi").compareTo(quantity("12Mi")) == 0', + want=True, + ), + QuantityCase( + name="IsQuantity", + reason="isQuantity accepts the plain integer 20.", + expr='isQuantity("20")', + want=True, + ), + QuantityCase( + name="IsQuantityMegabytes", + reason="isQuantity accepts 20M.", + expr='isQuantity("20M")', + want=True, + ), + QuantityCase( + name="IsQuantityMebibytes", + reason="isQuantity accepts 20Mi.", + expr='isQuantity("20Mi")', + want=True, + ), + QuantityCase( + name="IsQuantityInvalidSuffix", + reason="isQuantity rejects 20Mo, whose suffix is invalid.", + expr='isQuantity("20Mo")', + want=False, + ), + QuantityCase( + name="IsQuantityPassingRegex", + reason="isQuantity rejects 10Mm, which passes resource.Quantity's split regex but has no valid suffix.", + expr='isQuantity("10Mm")', + want=False, + ), # resource.Quantity accepts decimal exponents and nano/micro suffixes. - Case(name="isQuantity exponent lowercase", expr='isQuantity("256e3")', want=True), - Case(name="isQuantity exponent uppercase", expr='isQuantity("1E3")', want=True), - Case(name="exponent value", expr='quantity("256e3").compareTo(quantity("256000")) == 0', want=True), - Case(name="isQuantity nano", expr='isQuantity("100n")', want=True), - Case(name="isQuantity micro", expr='isQuantity("100u")', want=True), - Case(name="isQuantity trailing dot", expr='isQuantity("5.")', want=True), + QuantityCase( + name="IsQuantityExponent", + reason="isQuantity accepts 256e3, with a lower-case decimal exponent.", + expr='isQuantity("256e3")', + want=True, + ), + QuantityCase( + name="IsQuantityUpperExponent", + reason="isQuantity accepts 1E3, with an upper-case decimal exponent.", + expr='isQuantity("1E3")', + want=True, + ), + QuantityCase( + name="ExponentValue", + reason="256e3 compares equal to 256000.", + expr='quantity("256e3").compareTo(quantity("256000")) == 0', + want=True, + ), + QuantityCase( + name="IsQuantityNano", + reason="isQuantity accepts 100n, with the nano suffix.", + expr='isQuantity("100n")', + want=True, + ), + QuantityCase( + name="IsQuantityMicro", + reason="isQuantity accepts 100u, with the micro suffix.", + expr='isQuantity("100u")', + want=True, + ), + QuantityCase( + name="IsQuantityTrailingDot", + reason="isQuantity accepts 5., with a trailing dot.", + expr='isQuantity("5.")', + want=True, + ), # The quantity() constructor does NOT trim whitespace. - Case(name="isQuantity leading whitespace false", expr='isQuantity(" 5Gi")', want=False), - Case(name="isQuantity trailing whitespace false", expr='isQuantity("5Gi ")', want=False), - # Values equal at nano resolution compare equal (resource.Quantity.Cmp - # rounds to nano). - Case( - name="nano rounding equality", + QuantityCase( + name="IsQuantityLeadingWhitespace", + reason="isQuantity rejects a quantity with leading whitespace.", + expr='isQuantity(" 5Gi")', + want=False, + ), + QuantityCase( + name="IsQuantityTrailingWhitespace", + reason="isQuantity rejects a quantity with trailing whitespace.", + expr='isQuantity("5Gi ")', + want=False, + ), + # resource.ParseQuantity rounds up to nano. + QuantityCase( + name="NanoRounding", + reason="0.0000000004 and 0.000000001, equal at nano resolution, compare equal.", expr='quantity("0.0000000004").compareTo(quantity("0.000000001")) == 0', want=True, ), # doc-comment isQuantity examples. - Case(name="isQuantity 1.3G", expr='isQuantity("1.3G")', want=True), - Case(name="isQuantity 1.3Gi", expr='isQuantity("1.3Gi")', want=True), - Case(name="isQuantity comma", expr='isQuantity("1,3G")', want=False), - Case(name="isQuantity 10000k", expr='isQuantity("10000k")', want=True), - Case(name="isQuantity capital K", expr='isQuantity("200K")', want=False), - Case(name="isQuantity Three", expr='isQuantity("Three")', want=False), - Case(name="isQuantity bare suffix", expr='isQuantity("Mi")', want=False), + QuantityCase( + name="IsQuantityDecimalG", + reason="isQuantity accepts 1.3G.", + expr='isQuantity("1.3G")', + want=True, + ), + QuantityCase( + name="IsQuantityDecimalGi", + reason="isQuantity accepts 1.3Gi.", + expr='isQuantity("1.3Gi")', + want=True, + ), + QuantityCase( + name="IsQuantityComma", + reason="isQuantity rejects 1,3G, which has a comma for a decimal point.", + expr='isQuantity("1,3G")', + want=False, + ), + QuantityCase( + name="IsQuantityKilo", + reason="isQuantity accepts 10000k.", + expr='isQuantity("10000k")', + want=True, + ), + QuantityCase( + name="IsQuantityCapitalK", + reason="isQuantity rejects 200K, because Kubernetes spells the kilo suffix as a lower-case k.", + expr='isQuantity("200K")', + want=False, + ), + QuantityCase( + name="IsQuantityWord", + reason="isQuantity rejects the word Three.", + expr='isQuantity("Three")', + want=False, + ), + QuantityCase( + name="IsQuantityBareSuffix", + reason="isQuantity rejects Mi, a bare suffix.", + expr='isQuantity("Mi")', + want=False, + ), # equality. - Case(name="equality reflexivity", expr='quantity("200M") == quantity("200M")', want=True), - Case( - name="equality symmetry", + QuantityCase( + name="EqualityReflexivity", + reason="200M equals itself.", + expr='quantity("200M") == quantity("200M")', + want=True, + ), + QuantityCase( + name="EqualitySymmetry", + reason="200M equals 0.2G, and 0.2G equals 200M.", expr='quantity("200M") == quantity("0.2G") && quantity("0.2G") == quantity("200M")', want=True, ), - Case( - name="equality transitivity", + QuantityCase( + name="EqualityTransitivity", + reason="2M, 0.002G and 2000k all equal one another.", expr=( 'quantity("2M") == quantity("0.002G") && quantity("2000k") == quantity("2M") && ' 'quantity("0.002G") == quantity("2000k")' ), want=True, ), - Case(name="inequality", expr='quantity("200M") == quantity("0.3G")', want=False), + QuantityCase( + name="Inequality", + reason="200M doesn't equal 0.3G.", + expr='quantity("200M") == quantity("0.3G")', + want=False, + ), # isLessThan / isGreaterThan. - Case(name="less", expr='quantity("50M").isLessThan(quantity("50Mi"))', want=True), - Case(name="less obvious", expr='quantity("50M").isLessThan(quantity("100M"))', want=True), - Case(name="less false", expr='quantity("100M").isLessThan(quantity("50M"))', want=False), - Case(name="greater", expr='quantity("50Mi").isGreaterThan(quantity("50M"))', want=True), - Case(name="greater obvious", expr='quantity("150Mi").isGreaterThan(quantity("100Mi"))', want=True), - Case(name="greater false", expr='quantity("50M").isGreaterThan(quantity("100M"))', want=False), + QuantityCase( + name="IsLessThan", + reason="50M is less than 50Mi.", + expr='quantity("50M").isLessThan(quantity("50Mi"))', + want=True, + ), + QuantityCase( + name="IsLessThanObvious", + reason="50M is less than 100M.", + expr='quantity("50M").isLessThan(quantity("100M"))', + want=True, + ), + QuantityCase( + name="IsLessThanFalse", + reason="100M isn't less than 50M.", + expr='quantity("100M").isLessThan(quantity("50M"))', + want=False, + ), + QuantityCase( + name="IsGreaterThan", + reason="50Mi is greater than 50M.", + expr='quantity("50Mi").isGreaterThan(quantity("50M"))', + want=True, + ), + QuantityCase( + name="IsGreaterThanObvious", + reason="150Mi is greater than 100Mi.", + expr='quantity("150Mi").isGreaterThan(quantity("100Mi"))', + want=True, + ), + QuantityCase( + name="IsGreaterThanFalse", + reason="50M isn't greater than 100M.", + expr='quantity("50M").isGreaterThan(quantity("100M"))', + want=False, + ), # compareTo. - Case(name="compare equal", expr='quantity("200M").compareTo(quantity("0.2G")) == 0', want=True), - Case(name="compare less", expr='quantity("50M").compareTo(quantity("50Mi")) == -1', want=True), - Case(name="compare greater", expr='quantity("50Mi").compareTo(quantity("50M")) == 1', want=True), + QuantityCase( + name="CompareEqual", + reason="200M compares equal to 0.2G.", + expr='quantity("200M").compareTo(quantity("0.2G")) == 0', + want=True, + ), + QuantityCase( + name="CompareLess", + reason="50M compares less than 50Mi.", + expr='quantity("50M").compareTo(quantity("50Mi")) == -1', + want=True, + ), + QuantityCase( + name="CompareGreater", + reason="50Mi compares greater than 50M.", + expr='quantity("50Mi").compareTo(quantity("50M")) == 1', + want=True, + ), # add / sub (quantity and int overloads). - Case(name="add quantity", expr='quantity("50k").add(quantity("20")) == quantity("50.02k")', want=True), - Case(name="add int not less", expr='quantity("50k").add(20).isLessThan(quantity("50020"))', want=False), - Case(name="sub quantity", expr='quantity("50k").sub(quantity("20")) == quantity("49.98k")', want=True), - Case(name="sub int", expr='quantity("50k").sub(20) == quantity("49980")', want=True), - Case( - name="arith chain 1", + QuantityCase( + name="AddQuantity", + reason="50k plus the quantity 20 equals 50.02k.", + expr='quantity("50k").add(quantity("20")) == quantity("50.02k")', + want=True, + ), + QuantityCase( + name="AddInt", + reason="50k plus the int 20 isn't less than 50020.", + expr='quantity("50k").add(20).isLessThan(quantity("50020"))', + want=False, + ), + QuantityCase( + name="SubQuantity", + reason="50k minus the quantity 20 equals 49.98k.", + expr='quantity("50k").sub(quantity("20")) == quantity("49.98k")', + want=True, + ), + QuantityCase( + name="SubInt", + reason="50k minus the int 20 equals 49980.", + expr='quantity("50k").sub(20) == quantity("49980")', + want=True, + ), + QuantityCase( + name="ArithChain", + reason="50k plus 20 minus 100k is the integer -49980.", expr='quantity("50k").add(20).sub(quantity("100k")).asInteger() == -49980', want=True, ), - Case( - name="arith chain 2", + QuantityCase( + name="ArithChainLonger", + reason="50k plus 20 minus 100k minus -50000 is the integer 20.", expr='quantity("50k").add(20).sub(quantity("100k")).sub(-50000).asInteger() == 20', want=True, ), @@ -140,55 +327,127 @@ class ParseErrCase: # surface. celpy can't tell the two call styles apart, so we accept # both, but the test asserts the upstream-correct global form (see # cel.py's documented divergences). - Case(name="sign positive", expr='sign(quantity("50k")) == 1', want=True), - Case(name="sign negative", expr='sign(quantity("-50k")) == -1', want=True), - Case(name="sign zero", expr='sign(quantity("0")) == 0', want=True), + QuantityCase( + name="SignPositive", + reason="The sign of 50k is 1.", + expr='sign(quantity("50k")) == 1', + want=True, + ), + QuantityCase( + name="SignNegative", + reason="The sign of -50k is -1.", + expr='sign(quantity("-50k")) == -1', + want=True, + ), + QuantityCase( + name="SignZero", + reason="The sign of 0 is 0.", + expr='sign(quantity("0")) == 0', + want=True, + ), # Binary-suffix overflow saturates to int64-max, keeping sign, so # 8Ei/10Ei/100Ei all compare equal to int64-max (resource.Quantity # stores BinarySI in an int64). Confirmed against resource.Quantity. - Case( - name="Ei saturates to int64 max", + QuantityCase( + name="EiSaturates", + reason="8Ei saturates, comparing equal to the int64 maximum.", expr='quantity("8Ei").compareTo(quantity("9223372036854775807")) == 0', want=True, ), - Case(name="8Ei equals 10Ei", expr='quantity("8Ei").compareTo(quantity("10Ei")) == 0', want=True), - Case(name="10Ei equals 100Ei", expr='quantity("10Ei").compareTo(quantity("100Ei")) == 0', want=True), - Case( - name="negative Ei saturates", + QuantityCase( + name="SaturatedEiEqual", + reason="8Ei and 10Ei both saturate, so they compare equal.", + expr='quantity("8Ei").compareTo(quantity("10Ei")) == 0', + want=True, + ), + QuantityCase( + name="LargerEiEqual", + reason="10Ei and 100Ei both saturate, so they compare equal.", + expr='quantity("10Ei").compareTo(quantity("100Ei")) == 0', + want=True, + ), + QuantityCase( + name="NegativeEiSaturates", + reason="-10Ei saturates, keeping its sign, to compare equal to minus the int64 maximum.", expr='quantity("-10Ei").compareTo(quantity("-9223372036854775807")) == 0', want=True, ), - # 7Ei is below int64-max, so it does NOT saturate and stays less. - Case(name="7Ei below saturation", expr='quantity("7Ei").isLessThan(quantity("8Ei"))', want=True), + QuantityCase( + name="EiBelowSaturation", + reason="7Ei is below the int64 maximum, so it doesn't saturate and is less than 8Ei.", + expr='quantity("7Ei").isLessThan(quantity("8Ei"))', + want=True, + ), # Large DECIMAL-path values do not saturate (only the binary path # does) and must not raise on nano-rounding. isQuantity must be true # and the value must round-trip. - Case(name="isQuantity 256E", expr='isQuantity("256E")', want=True), - Case(name="isQuantity 10E", expr='isQuantity("10E")', want=True), - Case(name="256E value", expr='quantity("256E").compareTo(quantity("256000000000000000000")) == 0', want=True), - Case(name="256E greater than 1Ei", expr='quantity("256E").isGreaterThan(quantity("1Ei"))', want=True), + QuantityCase( + name="IsQuantityLargeExa", + reason="isQuantity accepts 256E, a decimal quantity too large for int64.", + expr='isQuantity("256E")', + want=True, + ), + QuantityCase( + name="IsQuantityExa", + reason="isQuantity accepts 10E.", + expr='isQuantity("10E")', + want=True, + ), + QuantityCase( + name="LargeExaValue", + reason="256E compares equal to its full value, unsaturated.", + expr='quantity("256E").compareTo(quantity("256000000000000000000")) == 0', + want=True, + ), + QuantityCase( + name="LargeExaGreater", + reason="256E is greater than 1Ei.", + expr='quantity("256E").isGreaterThan(quantity("1Ei"))', + want=True, + ), # asInteger / isInteger. - Case(name="as integer", expr='quantity("50k").asInteger() == 50000', want=True), - Case(name="is integer true small", expr='quantity("50").isInteger()', want=True), - Case(name="is integer true big magnitude", expr='quantity("50000000G").isInteger()', want=True), - Case( - name="is integer false overflow", + QuantityCase( + name="AsInteger", + reason="50k as an integer is 50000.", + expr='quantity("50k").asInteger() == 50000', + want=True, + ), + QuantityCase( + name="IsInteger", + reason="50 is an integer.", + expr='quantity("50").isInteger()', + want=True, + ), + QuantityCase( + name="IsIntegerBig", + reason="50000000G is an integer.", + expr='quantity("50000000G").isInteger()', + want=True, + ), + QuantityCase( + name="IsIntegerOverflow", + reason="A quantity too large for int64 isn't an integer.", expr='quantity("9999999999999999999999999999999999999G").isInteger()', want=False, ), - # asInteger overflow is a runtime error upstream -> non-match here. - Case( - name="as integer overflow is non-match", + QuantityCase( + name="AsIntegerError", + reason="Converting a quantity too large for int64 to an integer doesn't match; upstream raises instead.", expr='quantity("9999999999999999999999999999999999999G").asInteger() > 0', want=False, ), # asApproximateFloat. - Case(name="as approximate float", expr='quantity("50.703k").asApproximateFloat() == 50703.0', want=True), - # An invalid suffix is a runtime error upstream -> non-match here. - # (Uses a member method upstream accepts, isGreaterThan, so the - # non-match is the parse failure, not a rejected call form.) - Case( - name="invalid suffix is non-match", + QuantityCase( + name="AsFloat", + reason="50.703k as an approximate float is 50703.0.", + expr='quantity("50.703k").asApproximateFloat() == 50703.0', + want=True, + ), + # isGreaterThan is a member method upstream accepts, so the non-match is + # the parse failure, not a rejected call form. + QuantityCase( + name="InvalidSuffix", + reason="Comparing 10Mo, whose suffix is invalid, doesn't match; upstream raises instead.", expr='quantity("10Mo").isGreaterThan(quantity("1"))', want=False, ), @@ -196,28 +455,63 @@ class ParseErrCase: @pytest.mark.parametrize("case", QUANTITY_CASES, ids=lambda case: case.name) -def test_quantity(case: Case) -> None: +def test_quantity(case: QuantityCase) -> None: """A quantity CEL expression evaluates as it does upstream.""" - assert _eval(case.expr) == case.want + got = cel.Program(case.expr).matches({}) + assert got == case.want, case.reason -# The bare-suffix row ("Mi") is a DELIBERATE divergence, not parity: upstream -# parses most bare suffixes as 0 but inconsistently errors on a few (see -# parse()'s docstring). We reject every bare suffix; no device capacity is -# ever a bare suffix. PARSE_REJECTS_CASES = [ - ParseErrCase(name="invalid suffix Mo", input="10Mo"), - ParseErrCase(name="passing regex bad suffix Mm", input="10Mm"), - ParseErrCase(name="capital K", input="200K"), - ParseErrCase(name="comma", input="1,3G"), - ParseErrCase(name="word", input="Three"), - ParseErrCase(name="bare suffix (deliberate divergence)", input="Mi"), - ParseErrCase(name="empty", input=""), + ParseRejectsCase( + name="InvalidSuffix", + reason="parse rejects 10Mo, whose suffix is invalid.", + s="10Mo", + want="invalid quantity: '10Mo'", + ), + ParseRejectsCase( + name="PassingRegex", + reason="parse rejects 10Mm, which passes resource.Quantity's split regex but has no valid suffix.", + s="10Mm", + want="invalid quantity: '10Mm'", + ), + ParseRejectsCase( + name="CapitalK", + reason="parse rejects 200K, because Kubernetes spells the kilo suffix as a lower-case k.", + s="200K", + want="invalid quantity: '200K'", + ), + ParseRejectsCase( + name="Comma", + reason="parse rejects 1,3G, which has a comma for a decimal point.", + s="1,3G", + want="invalid quantity: '1,3G'", + ), + ParseRejectsCase( + name="Word", + reason="parse rejects the word Three.", + s="Three", + want="invalid quantity: 'Three'", + ), + # A DELIBERATE divergence, not parity: upstream parses most bare suffixes + # as 0 but inconsistently errors on a few (see parse()'s docstring). We + # reject every bare suffix; no device capacity is ever a bare suffix. + ParseRejectsCase( + name="BareSuffix", + reason="parse rejects Mi, a bare suffix, where upstream parses most bare suffixes as 0.", + s="Mi", + want="invalid quantity: 'Mi'", + ), + ParseRejectsCase( + name="Empty", + reason="parse rejects an empty string.", + s="", + want="invalid quantity: ''", + ), ] @pytest.mark.parametrize("case", PARSE_REJECTS_CASES, ids=lambda case: case.name) -def test_parse_rejects(case: ParseErrCase) -> None: - """parse() rejects what resource.Quantity rejects, which drives the non-matches above.""" - with pytest.raises(ValueError, match="invalid quantity"): - quantity.parse(case.input) +def test_parse_rejects(case: ParseRejectsCase) -> None: + """parse() rejects what resource.Quantity rejects, plus every bare suffix, a deliberate divergence.""" + with pytest.raises(ValueError, match=case.want): + quantity.parse(case.s) diff --git a/functions/compose-model-deployment/tests/test_scheduling.py b/functions/compose-model-deployment/tests/test_scheduling.py index a584c9c9f..6d3ee98ed 100644 --- a/functions/compose-model-deployment/tests/test_scheduling.py +++ b/functions/compose-model-deployment/tests/test_scheduling.py @@ -25,6 +25,7 @@ import dataclasses import datetime +from typing import Literal import pytest from function import cel, scheduling @@ -33,17 +34,16 @@ from models.ai.modelplane.modelreplica import v1alpha1 as mrv1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -# A fixed transition time keeps observed conditions deterministic. -_TRANSITION_TIME = datetime.datetime(2025, 1, 1, tzinfo=datetime.UTC) - -# A GPU memory selector reused across cases. +# The selectors the cases' device requests use. They're named because which one +# a case uses is often what it tests, such as _MEM_200 against _MEM_LT_200, and +# an 80-character literal would bury that. A request's selectors reach the +# Candidate unchanged, so the expected DeviceRequests use them too. _MEM_141 = 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("141Gi")) >= 0' _MEM_200 = 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("200Gi")) >= 0' _MEM_LT_200 = 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("200Gi")) < 0' _IB = 'device.attributes["nic.nvidia.com"].linkType == "infiniband"' -# Default engine name used by the single-engine helpers below. -_ENGINE = "main" +_Role = Literal["Standalone", "Leader", "Worker"] @dataclasses.dataclass @@ -51,164 +51,80 @@ class Case: """A test case for scheduling.schedule.""" name: str + reason: str deployment: mdv1alpha1.ModelDeployment clusters: list[icv1alpha1.InferenceCluster] all_replicas: list[mrv1alpha1.ModelReplica] + fill: bool want: list[scheduling.Candidate] -def _request(name: str = "gpu", count: int = 1, cel_exprs: list[str] | None = None) -> mdv1alpha1.Device: - """A nodeSelector device request.""" - return mdv1alpha1.Device( - name=name, - count=count, - selectors=[mdv1alpha1.Selector(cel=c) for c in (cel_exprs or [_MEM_141])], - ) - - -def _template() -> mdv1alpha1.Template: - return mdv1alpha1.Template( - spec=mdv1alpha1.Spec( - containers=[mdv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")], +def _member(*, role: _Role, worker_nodes: int | None, devices: list[mdv1alpha1.Device] | None) -> mdv1alpha1.Member: + """A ModelDeployment member running vLLM, which claims nothing if devices is None.""" + return mdv1alpha1.Member( + role=role, + worker=mdv1alpha1.Worker(nodes=worker_nodes) if worker_nodes is not None else None, + nodeSelector=mdv1alpha1.NodeSelector(devices=devices) if devices is not None else None, + template=mdv1alpha1.Template( + spec=mdv1alpha1.Spec(containers=[mdv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")]), ), ) -def _node_selector(requests: list[mdv1alpha1.Device] | None) -> mdv1alpha1.NodeSelector: - return mdv1alpha1.NodeSelector(devices=requests if requests is not None else [_request()]) - - -def _engine( - name: str = _ENGINE, - *, - copies: int = 1, - pipeline: int = 1, - requests: list[mdv1alpha1.Device] | None = None, -) -> mdv1alpha1.Engine: - """An engine of homogeneous members. - - pipeline == 1 is a single Standalone member; pipeline > 1 is a Leader plus a - Worker spanning (pipeline - 1) nodes, so the engine spans `pipeline` nodes. - Engine copies multiply that, so node cost is pipeline * copies. Every member - carries the same nodeSelector; heterogeneous-member cases build their - members directly. - """ - if pipeline == 1: - members = [mdv1alpha1.Member(role="Standalone", nodeSelector=_node_selector(requests), template=_template())] - else: - members = [ - mdv1alpha1.Member(role="Leader", nodeSelector=_node_selector(requests), template=_template()), - mdv1alpha1.Member( - role="Worker", - worker=mdv1alpha1.Worker(nodes=pipeline - 1), - nodeSelector=_node_selector(requests), - template=_template(), - ), - ] - return mdv1alpha1.Engine(name=name, copies=copies, members=members) - - def _deployment( - name: str = "my-model", - replicas: int = 1, - pipeline: int = 1, - count: int = 1, - requests: list[mdv1alpha1.Device] | None = None, - engines: list[mdv1alpha1.Engine] | None = None, - tolerations: list[mdv1alpha1.Toleration] | None = None, + *, + replicas: int, + members: list[mdv1alpha1.Member], + tolerations: list[mdv1alpha1.Toleration] | None, ) -> mdv1alpha1.ModelDeployment: - """Construct a ModelDeployment. - - The single-engine helpers map a node shape onto one engine: pipeline sets the - engine's node span (a Standalone, or a Leader plus a Worker) and count sets - the engine's copies, so node cost is pipeline * count. Multi-engine cases - pass `engines` directly. - """ - if engines is None: - engines = [_engine(copies=count, pipeline=pipeline, requests=requests)] + """The ModelDeployment my-model, whose one engine, main, has the given members.""" return mdv1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name=name, namespace="ml-team"), + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), spec=mdv1alpha1.SpecModel1( replicas=replicas, template=mdv1alpha1.TemplateModel( - spec=mdv1alpha1.SpecModel(engines=engines, tolerations=tolerations), + spec=mdv1alpha1.SpecModel( + engines=[mdv1alpha1.Engine(name="main", copies=1, members=members)], + tolerations=tolerations, + ), ), ), ) -def _gpu_device( - name: str = "gpu", - *, - claim: str = "DRA", - driver: str = "gpu.nvidia.com", - device_class: str = "gpu.nvidia.com", - count: int = 1, - memory: str = "141Gi", -) -> dict: - """A GPU device dict for a pool, with memory capacity.""" - d = { - "name": name, - "claim": claim, - "driver": driver, - "count": count, - "capacity": {"memory": {"value": memory}}, - } - if claim == "DRA": - d["deviceClassName"] = device_class - return d - - -def _nic_device(*, link_type: str = "infiniband", count: int = 1) -> dict: - """A synthetic NIC device dict for a pool.""" - return { - "name": "nic", - "claim": "Synthetic", - "driver": "nic.nvidia.com", - "count": count, - "attributes": {"linkType": {"string": link_type}}, - } +def _gpu_device(*, name: str, claim: Literal["DRA", "Synthetic"], count: int, memory: str) -> icv1alpha1.Device: + """A gpu.nvidia.com GPU in a pool. Only a DRA device has a device class to claim it by.""" + return icv1alpha1.Device( + name=name, + claim=claim, + driver="gpu.nvidia.com", + deviceClassName="gpu.nvidia.com" if claim == "DRA" else None, + count=count, + capacity={"memory": icv1alpha1.Capacity(value=memory)}, + ) -def _pool(name: str, *, nodes: int = 2, devices: list[dict] | None = None) -> dict: - """A pool with devices, for nodeSelector tests.""" - return { - "name": name, - "nodes": nodes, - "devices": devices if devices is not None else [_gpu_device()], - } +def _nic_device(*, link_type: str) -> icv1alpha1.Device: + """A Synthetic NIC in a pool, which a nodeSelector can match but nothing claims.""" + return icv1alpha1.Device( + name="nic", + claim="Synthetic", + driver="nic.nvidia.com", + count=1, + attributes={"linkType": icv1alpha1.Attributes(string=link_type)}, + ) def _cluster( - name: str, *, - placement_labels: dict[str, str] | None = None, - ready: bool = True, - gateway_hostname: str = "cluster-a.clusters.example.com", - pools: list[dict] | None = None, - taints: list[icv1alpha1.Taint] | None = None, + name: str, + gateway_hostname: str | None, + ready: bool, + pools: list[icv1alpha1.GpuPool], + taints: list[icv1alpha1.Taint] | None, + placement_labels: dict[str, str] | None, ) -> icv1alpha1.InferenceCluster: - """Construct an InferenceCluster with the given readiness and pools. - - A "ready" cluster has a Ready=True condition and a gateway hostname. An - address alone is not enough: an InferenceGateway addresses a cluster by name. - Setting ready=False or gateway_hostname="" produces a degraded cluster - the scheduler will retain but not pick anew. - """ - if pools is None: - pools = [{"name": "default", "nodes": 2, "devices": [_gpu_device()]}] - - status = "True" if ready else "False" - reason = "Available" if ready else "Unavailable" - conditions = [ - icv1alpha1.Condition( - type="Ready", - status=status, - reason=reason, - lastTransitionTime=_TRANSITION_TIME, - ) - ] - + """An InferenceCluster with the given readiness, gateway hostname, GPU pools, taints and placement labels.""" return icv1alpha1.InferenceCluster( metadata=metav1.ObjectMeta(name=name), spec=icv1alpha1.Spec( @@ -219,1563 +135,5251 @@ def _cluster( taints=taints, placement=( icv1alpha1.Placement(metadata=icv1alpha1.Metadata(labels=placement_labels)) - if placement_labels + if placement_labels is not None else None ), ), status=icv1alpha1.Status( - conditions=conditions, - gateway=( - icv1alpha1.Gateway(address="10.0.0.1", hostname=gateway_hostname) - if gateway_hostname - else icv1alpha1.Gateway(address="10.0.0.1") - ), + conditions=[ + icv1alpha1.Condition( + type="Ready", + status="True" if ready else "False", + reason="Available" if ready else "Unavailable", + lastTransitionTime=datetime.datetime(2025, 1, 1, tzinfo=datetime.UTC), + ), + ], + # The scheduler needs a hostname, not just an address, because an + # InferenceGateway addresses a cluster by name. + gateway=icv1alpha1.Gateway(address="10.0.0.1", hostname=gateway_hostname), providerConfigRef=icv1alpha1.ProviderConfigRef(name=name), - gpuPools=[icv1alpha1.GpuPool(**p) for p in pools], + gpuPools=pools, ), ) -def _replica_device_requests() -> list[mrv1alpha1.DeviceRequest]: - return [ - mrv1alpha1.DeviceRequest( - name="gpu", - deviceClassName="gpu.nvidia.com", - count=1, - selectors=[mrv1alpha1.Selector(cel=_MEM_141)], - ), - ] - - -def _replica_engine( - name: str = _ENGINE, +def _replica_member( *, - pool: str = "default", - copies: int = 1, - pipeline: int = 1, -) -> mrv1alpha1.Engine: - """One engine of an observed ModelReplica, with per-member pool pins and resolved requests.""" - template = mrv1alpha1.Template( - spec=mrv1alpha1.Spec(containers=[mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")]), + role: _Role, + worker_nodes: int | None, + pool: str, + device_requests: list[mrv1alpha1.DeviceRequest] | None, +) -> mrv1alpha1.Member: + """A ModelReplica member running vLLM, pinned to pool, with its device requests resolved.""" + return mrv1alpha1.Member( + role=role, + worker=mrv1alpha1.Worker(nodes=worker_nodes) if worker_nodes is not None else None, + nodePoolName=pool, + deviceRequests=device_requests, + template=mrv1alpha1.Template( + spec=mrv1alpha1.Spec(containers=[mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")]), + ), ) - if pipeline == 1: - members = [ - mrv1alpha1.Member( - role="Standalone", nodePoolName=pool, deviceRequests=_replica_device_requests(), template=template - ) - ] - else: - members = [ - mrv1alpha1.Member( - role="Leader", nodePoolName=pool, deviceRequests=_replica_device_requests(), template=template - ), - mrv1alpha1.Member( - role="Worker", - worker=mrv1alpha1.Worker(nodes=pipeline - 1), - nodePoolName=pool, - deviceRequests=_replica_device_requests(), - template=template, - ), - ] - return mrv1alpha1.Engine(name=name, copies=copies, members=members) def _replica( - deployment_name: str, - cluster_name: str, - *, - pool: str = "default", - index: int = 0, - pipeline: int = 1, - count: int = 1, - engines: list[mrv1alpha1.Engine] | None = None, + *, name: str, deployment: str, cluster: str, index: int, members: list[mrv1alpha1.Member] ) -> mrv1alpha1.ModelReplica: - """Construct an observed ModelReplica pinned to a (cluster, index). - - Mirrors _deployment's single-engine mapping: pipeline sets the engine's node - span and count its copies, so node cost is pipeline * count. - """ - if engines is None: - engines = [_replica_engine(pool=pool, copies=count, pipeline=pipeline)] + """An observed ModelReplica of deployment, pinned to (cluster, index), whose one engine, main, has members.""" return mrv1alpha1.ModelReplica( metadata=metav1.ObjectMeta( - name=f"{deployment_name}-{cluster_name}-{index}", + name=name, namespace="ml-team", labels={ - "modelplane.ai/deployment": deployment_name, - "modelplane.ai/cluster": cluster_name, + "modelplane.ai/deployment": deployment, + "modelplane.ai/cluster": cluster, "modelplane.ai/replica-index": str(index), }, ), - spec=mrv1alpha1.SpecModel(clusterName=cluster_name, engines=engines), - ) - - -def _replica_with_pool( - deployment_name: str, - cluster_name: str, - *, - pool: str, - index: int = 0, - pipeline: int = 1, - count: int = 1, -) -> mrv1alpha1.ModelReplica: - """An observed ModelReplica pinned to a cluster AND a specific node pool.""" - return _replica(deployment_name, cluster_name, pool=pool, index=index, pipeline=pipeline, count=count) - - -def _collision_replica( - object_name: str, - cluster_name: str, - *, - pool: str, - index: int = 0, -) -> mrv1alpha1.ModelReplica: - """A my-model replica with an explicit object name, to force an identity clash. - - Carries the my-model deployment label and a chosen (cluster, index) so it - collides with a normal my-model replica, while its distinct metadata.name is - the tiebreak the scheduler sorts on. - """ - r = _replica_with_pool("my-model", cluster_name, pool=pool, index=index) - assert r.metadata is not None - r.metadata.name = object_name - return r - - -# Convenience: the resolved DeviceRequest for a default GPU request matching a -# default pool, used in expected candidates for nodeSelector cases. -def _resolved(name: str = "gpu", count: int = 1, cel_exprs: list[str] | None = None) -> scheduling.DeviceRequest: - return scheduling.DeviceRequest( - name=name, - device_class_name="gpu.nvidia.com", - count=count, - cel_selectors=cel_exprs or [_MEM_141], + spec=mrv1alpha1.SpecModel( + clusterName=cluster, engines=[mrv1alpha1.Engine(name="main", copies=1, members=members)] + ), ) -def _placement( +def _candidate( *, - name: str = _ENGINE, - pool: str = "default", - device_requests: list[scheduling.DeviceRequest] | None = None, - pipeline: int = 1, -) -> scheduling.EnginePlacement: - """An expected EnginePlacement: one member placement per member. - - Mirrors _engine's member shape: pipeline == 1 is a single Standalone, - pipeline > 1 a Leader plus a Worker, all on the same pool with the same - resolved requests. - """ - dr = device_requests if device_requests is not None else [_resolved()] - if pipeline == 1: - members = [scheduling.MemberPlacement(role="Standalone", pool=pool, device_requests=dr)] - else: - members = [ - scheduling.MemberPlacement(role="Leader", pool=pool, device_requests=dr), - scheduling.MemberPlacement(role="Worker", pool=pool, device_requests=dr), - ] - return scheduling.EnginePlacement(name=name, members=members) - - -# Convenience: build an expected Candidate defaulting to index 0, so the many -# single-replica-per-cluster cases stay terse. A placed or retained replica -# resolves to one engine on the default pool with the default GPU request; a -# degraded/unplaced cluster carries no gateway. Cases that need a specific pool, -# request, or engine layout pass `engines` explicitly. -def _cand( name: str, - *, - index: int = 0, - pool: str = "default", - device_requests: list[scheduling.DeviceRequest] | None = None, - pipeline: int = 1, - engines: list[scheduling.EnginePlacement] | None = None, - gateway_hostname: str = "", - placement_labels: dict[str, str] | None = None, + index: int, + gateway_hostname: str, + placement_labels: dict[str, str], + members: list[scheduling.MemberPlacement], ) -> scheduling.Candidate: - if engines is None: - engines = [_placement(pool=pool, device_requests=device_requests, pipeline=pipeline)] + """An expected Candidate for (name, index), whose one engine, main, places members.""" return scheduling.Candidate( name=name, index=index, - engines=engines, gateway_hostname=gateway_hostname, - placement_labels=placement_labels or {}, + placement_labels=placement_labels, + engines=[scheduling.EnginePlacement(name="main", members=members)], ) -# Deployments use the default single-GPU nodeSelector request (any pool's GPU -# device satisfies it), so these focus on placement rather than pool matching; -# NODE_SELECTOR_CASES covers request-to-device matching. SCHEDULE_CASES = [ + # Placement: retain, spread, scale and capacity. Every member requests one + # GPU matching _MEM_141, which every pool's 141Gi GPU satisfies, so these + # cases focus on placement rather than pool matching. Case( - name="no clusters returns no candidates", - deployment=_deployment(), + name="NoClusters", + reason="With no clusters, schedule places nothing.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[], all_replicas=[], + fill=True, want=[], ), Case( - name="single ready cluster is picked", - deployment=_deployment(), - clusters=[_cluster("cluster-a")], - all_replicas=[], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default")], - ), - Case( - name="not-ready cluster is not picked for a new replica", - deployment=_deployment(), - clusters=[_cluster("cluster-a", ready=False)], - all_replicas=[], - want=[], - ), - Case( - name="cluster without gateway address is not picked", - deployment=_deployment(), - clusters=[_cluster("cluster-a", gateway_hostname="")], - all_replicas=[], - want=[], - ), - Case( - name="multi-node deployment needs enough nodes", - deployment=_deployment(pipeline=4), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], - all_replicas=[], - want=[], - ), - Case( - name="existing replica is retained on its pinned cluster", - deployment=_deployment(), - clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - # cluster-a wins even though cluster-b is also viable. The pin - # still matches, so it's retained with its resolved pool/requests. - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="degraded pinned cluster is retained with empty gateway", - deployment=_deployment(), - clusters=[_cluster("cluster-a", ready=False, gateway_hostname="")], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="")], - ), - Case( - name="deleted pinned cluster triggers re-placement", - deployment=_deployment(), - clusters=[_cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com")], - all_replicas=[_replica("my-model", "cluster-a")], - want=[_cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default")], - ), - Case( - name="scale up places new replicas on additional clusters", - deployment=_deployment(replicas=2), + name="OneReadyCluster", + reason="A new replica lands on the one ready cluster, with no placement labels because the cluster declares none.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ], - all_replicas=[_replica("my-model", "cluster-a")], - want=[ - _cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com"), - _cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default"), + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) ], - ), - Case( - name="scale up with no extra capacity returns only retained", - deployment=_deployment(replicas=2), - # Single-node pool, already filled by the retained replica, so no - # second replica can be placed - not even on the same cluster. - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default")], - ), - Case( - name="two replicas pack onto one cluster when it is the only option", - deployment=_deployment(replicas=2), - # One cluster, a 2-node pool, two 1-node replicas. With nowhere - # to spread, both pack onto cluster-a at indices 0 and 1. - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], all_replicas=[], + fill=True, want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), Case( - name="two replicas spread across two clusters before packing", - deployment=_deployment(replicas=2), - # Both clusters can hold two replicas, but we prefer one each. + name="ClusterNotReady", + reason="A cluster that isn't Ready gets no new replica.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=2)]), _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=2)], - ), + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=False, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) ], all_replicas=[], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], + fill=True, + want=[], ), Case( - name="three replicas spread first then pack the remainder", - deployment=_deployment(replicas=3), - # Two clusters, plenty of room. Spread gives a, b one each, then - # the third lands back on cluster-a (lowest load, name tiebreak). + name="NoGatewayHostname", + reason="A cluster without a gateway hostname gets no new replica.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], - ), + name="cluster-a", + gateway_hostname=None, + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) ], all_replicas=[], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], + fill=True, + want=[], ), Case( - name="capacity forces packing past the spread preference", - deployment=_deployment(replicas=3), - # cluster-b holds one replica; cluster-a has room for the rest. - # Spread puts one on each, then the third can't fit on b (full), - # so it packs onto a. + name="TooFewNodes", + reason="A Leader and a three-node Worker need four nodes, so a two-node pool hosts no replica.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + _member( + role="Worker", + worker_nodes=3, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=1)], - ), + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) ], all_replicas=[], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], + fill=True, + want=[], ), Case( - name="new replica spreads onto an empty cluster before doubling up", - deployment=_deployment(replicas=2), - # cluster-a already hosts a replica; cluster-b is empty. The new - # replica prefers empty cluster-b over packing onto a. + name="RetainPinned", + reason="An existing replica on cluster-a stays there with its resolved pool and requests, where fresh placement would also put it as the lower name.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), _cluster( - "cluster-b", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, ), ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), Case( - name="new replica takes the lowest free index on a packed cluster", - deployment=_deployment(replicas=3), - # Only cluster-a exists, already hosting indices 0 and 2 (1 was - # deleted). The new replica fills the gap at index 1. - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], + name="DegradedClusterRetained", + reason="A replica stays on a cluster that isn't Ready and has no gateway hostname, and its candidate's gateway hostname is empty.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname=None, + ready=False, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=2), + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) ], + fill=True, want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=2, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _candidate( + name="cluster-a", + index=0, + gateway_hostname="", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), Case( - name="scale down packs off by dropping the highest index first", - deployment=_deployment(replicas=2), - # cluster-a hosts indices 0 and 1; cluster-b hosts index 0. Three - # replicas, want two. Highest index (a/1) is dropped, keeping the - # spread across a/0 and b/0. + name="ClusterDeleted", + reason="A replica whose cluster no longer exists is re-placed on another cluster.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), _cluster( - "cluster-b", + name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], - ), + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) ], all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=1), - _replica_with_pool("my-model", "cluster-b", pool="default", index=0), + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) ], + fill=True, want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="retained replica is charged at its own node cost, not the new shape", - # The deployment's workers grew to pipeline=4 (4 nodes/replica), - # but the existing replica was created at pipeline=2 and is - # retained (no nodeSelector change rolls it). It still consumes - # only its original 2 nodes. The pool has 6, so a second replica - # at the new 4-node cost must still fit (6 - 2 = 4). Regression: - # charging the retained replica at the new shape (4) would leave - # 2 free and wrongly refuse the placement. - deployment=_deployment(replicas=2, pipeline=4), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=6)])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default", pipeline=2)], - # The retained replica is re-stamped to the deployment's current - # pipeline=4 shape but still charged its observed 2 nodes in the - # ledger. - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pipeline=4), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pipeline=4), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), Case( - name="scale down drops from the most-loaded cluster to preserve spread", - deployment=_deployment(replicas=2), - # cluster-a hosts two replicas, cluster-b one. Scaling 3->2 must - # drop a's extra (a/1), NOT b's sole replica - otherwise we'd - # leave a packed and b empty, the opposite of spread. b's index - # is 3 (higher than a/1) to prove we drop by cluster load, not by - # a global index comparison. + name="ScaleUp", + reason="Scaling from one replica to two places the new one on the cluster that has none.", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), _cluster( - "cluster-b", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, ), ], all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=1), - _replica_with_pool("my-model", "cluster-b", pool="default", index=3), + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) ], + fill=True, want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=3, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), Case( - name="co-located replicas are both retained across a reconcile", - deployment=_deployment(replicas=2), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], - all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=1), - ], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + name="ScaleUpNoCapacity", + reason="Scaling up when the retained replica fills the only one-node pool places no new replica.", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=1, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="PackOneCluster", + reason="With one cluster to place on, two replicas pack onto it at indices 0 and 1.", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="SpreadTwoClusters", + reason="Two replicas spread one to each of two clusters, though either cluster could hold both.", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="SpreadThenPack", + reason="Three replicas spread across two clusters, then the third lands on cluster-a, the lower name at equal load.", + deployment=_deployment( + replicas=3, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="OneNodePoolFills", + reason="Three replicas spread one to each cluster, filling cluster-b's one-node pool, and the third lands on cluster-a, which the name tiebreak picks whether or not cluster-b is full.", + deployment=_deployment( + replicas=3, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=1, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="SpreadToEmpty", + reason="A new replica goes to empty cluster-b rather than doubling up on cluster-a, which already hosts one.", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="LowestFreeIndex", + reason="A new replica on a cluster that hosts indices 0 and 2 takes the free index 1.", + deployment=_deployment( + replicas=3, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-a-2", + deployment="my-model", + cluster="cluster-a", + index=2, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=2, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="ScaleDownHighestIndex", + reason="Scaling three replicas to two drops cluster-a's index 1, keeping one replica on each cluster.", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-a-1", + deployment="my-model", + cluster="cluster-a", + index=1, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-b-0", + deployment="my-model", + cluster="cluster-b", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # The existing replica was created with a 1-node Worker, and is retained + # because no nodeSelector change rolls it. It's re-stamped to the + # deployment's current shape. Charging the retained replica at the new + # shape would leave 2 nodes free and wrongly refuse the placement. + Case( + name="RetainedNodeCost", + reason="A retained replica costs the two nodes it was placed with, not the deployment's new four, so a second replica fits the six-node pool.", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + _member( + role="Worker", + worker_nodes=3, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=6, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Leader", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ), + _replica_member( + role="Worker", + worker_nodes=1, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ), + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + scheduling.MemberPlacement( + role="Worker", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + scheduling.MemberPlacement( + role="Worker", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # Dropping cluster-b's only replica would leave cluster-a packed and + # cluster-b empty, the opposite of spread. cluster-b's replica has index 3, + # higher than a/1, to prove we drop by cluster load, not by a global index + # comparison. + Case( + name="ScaleDownMostLoaded", + reason="Scaling three replicas to two drops cluster-a's second replica rather than cluster-b's only one.", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-a-1", + deployment="my-model", + cluster="cluster-a", + index=1, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-b-3", + deployment="my-model", + cluster="cluster-b", + index=3, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=3, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="RetainColocated", + reason="Two existing replicas on cluster-a, the only cluster, stay at a/0 and a/1, where fresh placement would also put them.", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-a-1", + deployment="my-model", + cluster="cluster-a", + index=1, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="ScaleDownNameTiebreak", + reason="Scaling two replicas at index 0 down to one keeps cluster-a's, the lower name.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[ + _replica( + name="my-model-cluster-b-0", + deployment="my-model", + cluster="cluster-b", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="AlphabeticalPlacement", + reason="New replicas go to clusters in name order, whatever order the clusters arrive in.", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-c", + gateway_hostname="cluster-c.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="OtherDeploymentCapacity", + reason="Another deployment's replica fills the only node, leaving no room for a new replica.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=1, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="other-model-cluster-a-0", + deployment="other-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[], + ), + Case( + name="OwnReplicaFillsPool", + reason="An existing replica filling cluster-a's one-node pool stays there, and with the deployment already at its one replica, schedule builds no capacity ledger that could charge it.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=1, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # other-model's pods are pinned to a node label no node carries, so + # they're unschedulable and occupy nothing. Charging the unattributable + # replica would wrongly report the cluster full. + Case( + name="OtherDeletedPool", + reason="Another deployment's replica pinned to a pool the cluster no longer publishes uses no node, so a new replica takes the free one.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=1, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="other-model-cluster-a-0", + deployment="other-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="gone", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # The schedule must be a function of state, not of delivery order. Both + # pools match, so either would be a valid placement - only determinism is + # under test. + Case( + name="CollisionByName", + reason="Of two replicas on the same cluster and index, the one whose name sorts first is retained, whatever their input order.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="a", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ), + icv1alpha1.GpuPool( + name="b", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ), + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0-dup", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="b", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="a", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="a", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # fill=False retains existing replicas but places no new ones. A caller + # passes fill=False when it can't yet trust the candidate set (for the + # ModelDeployment, when a referenced ModelCache is unresolved). Retain runs + # unconditionally; only the placement of new replicas is held. + Case( + name="NoFillNoReplicas", + reason="Without fill and with no existing replicas, nothing is placed.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=False, + want=[], + ), + Case( + name="NoFillRetains", + reason="Without fill, an existing replica is still retained.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=False, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="NoFillScaleUp", + reason="Without fill, scaling to three replicas keeps the existing one and places no more.", + deployment=_deployment( + replicas=3, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=False, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # nodeSelector device-request matching and pool pinning. + Case( + name="RequestMatches", + reason="A device request that matches a pool's GPU places the replica in that pool.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="RequestMismatch", + reason="A request for a 200Gi GPU matches no 141Gi GPU, so nothing is placed.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_200)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], + ), + Case( + name="CountNotCovered", + reason="A request for 8 GPUs doesn't fit a device with only 4, so nothing is placed.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=8, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=4, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], + ), + # `d.count or 1` would read a published 0 as 1, and place a replica whose + # ResourceClaim no device could satisfy. The status schema permits 0 even + # though an InferenceClass device count is floored at 1. + Case( + name="ZeroDeviceCount", + reason="A device published with count 0 satisfies no request, so nothing is placed.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=0, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], + ), + Case( + name="ZeroNodeCount", + reason="A pool with a matching GPU but no nodes, say one autoscaled to zero, hosts no replica.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=0, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], + ), + Case( + name="SyntheticNICUnclaimed", + reason="A synthetic NIC counts toward matching a pool but, unlike the DRA GPU, isn't among the resolved requests.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="MissingNIC", + reason="A pool without the NIC a request selects hosts no replica.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], + ), + # DRA allocates distinct devices per request. + Case( + name="OneDeviceTwoRequests", + reason="Two requests can't both claim one GPU device of count 1, so the pool doesn't match.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu-a", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="gpu-b", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], + ), + # Capacity is consumed across requests. Checking each request against the + # full device count on its own would let these two share 8 GPUs. + Case( + name="RequestsExceedCount", + reason="Two requests for 5 GPUs each need 10, more than the device's 8, so the pool doesn't match.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu-a", count=5, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="gpu-b", count=5, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=8, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], + ), + Case( + name="RequestsShareDevice", + reason="Two requests for 4 GPUs each fit a device of 8, and both resolve.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu-a", count=4, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="gpu-b", count=4, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=8, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu-a", + device_class_name="gpu.nvidia.com", + count=4, + cel_selectors=[_MEM_141], + ), + scheduling.DeviceRequest( + name="gpu-b", + device_class_name="gpu.nvidia.com", + count=4, + cel_selectors=[_MEM_141], + ), + ], + ), + ], + ), + ], + ), + Case( + name="NICPicksPool", + reason="Of two pools with a GPU, only the one with an infiniband NIC matches, so it's picked though it's listed second.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="dev", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="gpudirect-tcpx"), + ], + ), + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ), + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # The replica's serving workload would have no ResourceClaim to bind GPUs + # through. + Case( + name="SyntheticOnly", + reason="A request that matches only a synthetic NIC leaves nothing to claim, so nothing is placed.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ) + ], + taints=None, + placement_labels=None, + ) ], + all_replicas=[], + fill=True, + want=[], ), Case( - name="scale down across clusters drops higher cluster name at equal index", - deployment=_deployment(replicas=1), + name="RetainPinnedPool", + reason="A replica pinned to the cluster's only pool stays on it, where fresh placement would also put it.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) ], all_replicas=[ - _replica("my-model", "cluster-b"), - _replica("my-model", "cluster-a"), + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="frontier", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], - # Both at index 0, so the (index, name) tiebreak keeps cluster-a. - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], ), + # A claimable GPU keeps both pools viable hosts; the synthetic + # NIC's link type is the drifting discriminator. Case( - name="new placement is alphabetical for determinism", - deployment=_deployment(replicas=2), + name="SelectorDrift", + reason="A replica pinned to a pool its selector no longer matches is re-placed onto the pool that matches.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)]), + ], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-c", gateway_hostname="cluster-c.clusters.example.com"), - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="a", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="gpudirect-tcpx"), + ], + ), + icv1alpha1.GpuPool( + name="b", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ), + ], + taints=None, + placement_labels=None, + ) ], - all_replicas=[], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="a", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="b", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="PinStillMatches", + reason="A replica stays on pool a while that pool still matches, where fresh placement would also put it as the first of two matching pools.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="a", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ), + icv1alpha1.GpuPool( + name="b", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ), + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="a", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, want=[ - _cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default"), + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="a", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), Case( - name="other deployment's replicas consume node capacity", - deployment=_deployment(pipeline=1), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], - # other-model occupies the single node on cluster-a. - all_replicas=[_replica("other-model", "cluster-a")], + name="NoPoolMatches", + reason="A replica whose pool no longer matches is dropped when no other pool matches either.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="a", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="gpudirect-tcpx"), + ], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="a", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, want=[], ), Case( - name="our own observed replicas don't double-count against us", - deployment=_deployment(pipeline=1), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - # Retained on its pin: the single node it already occupies isn't - # charged against itself, so it stays rather than being evicted. - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="replica labeled for our deployment but pinned to unknown cluster is ignored", - deployment=_deployment(), - clusters=[_cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com")], - all_replicas=[_replica("my-model", "cluster-a")], - want=[_cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default")], - ), - Case( - name="another deployment pinned to a deleted pool consumes no capacity", - # other-model is pinned to pool "gone", which the cluster no - # longer publishes. Its pods are pinned to a node label no node - # carries, so they're unschedulable and occupy nothing. The one - # published node on "frontier" is therefore free for our replica. - # Charging the unattributable replica would wrongly report the - # cluster full. - deployment=_deployment(requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], - all_replicas=[_replica_with_pool("other-model", "cluster-a", pool="gone")], + name="UnpublishedPool", + reason="A replica pinned to a pool the cluster no longer publishes is re-placed onto one that matches.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # a/1 is pinned to the pool gone, which the cluster doesn't publish, so it + # never held one of frontier's nodes. The ledger would ignore it even if it + # charged dropped replicas, so this can't show that dropping one frees its + # node for the refill. + Case( + name="RefillBesideRetained", + reason="A replica dropped for its unpublished pool is re-placed at index 1 on frontier, taking the node a/0 leaves free in the two-node pool.", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="frontier", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-a-1", + deployment="my-model", + cluster="cluster-a", + index=1, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="gone", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="CountPerPool", + reason="A request for 8 GPUs pins to pool b, which has 8, rather than counting pool a's 4 toward a cluster-wide sum.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=8, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="a", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=4, memory="141Gi")] + ), + icv1alpha1.GpuPool( + name="b", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=8, memory="141Gi")] + ), + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, want=[ - _cand( + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="b", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=8, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # Per-member placement: single-pool engines, rejection when no pool fits, + # and claimless ride-along members. + Case( + name="OnePoolAllMembers", + reason="A leader that matches both pools and a worker that matches only big both go on big.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_200)])], + ), + ], + tolerations=None, + ), + clusters=[ + _cluster( name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="small", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ), + icv1alpha1.GpuPool( + name="big", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="200Gi")] + ), + ], + taints=None, + placement_labels=None, ) ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="big", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + scheduling.MemberPlacement( + role="Worker", + pool="big", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_200], + ) + ], + ), + ], + ), + ], ), + # The scheduler never splits an engine across pools, because it can't tell + # whether big and small share a fabric (#149). Case( - name="colliding (cluster, index) retains deterministically by replica name", - # Two of our replicas collide on (cluster-a, index 0) with - # different pinned pools. Retain keeps the first by replica name - # (my-model-cluster-a-0 on "a" sorts before the "-dup" replica on - # "b"), independent of input order, so the schedule is a function - # of state not of delivery order. Both pools match, so either - # would be a valid placement - only determinism is under test. - deployment=_deployment(requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), + name="NoSinglePool", + reason="A leader that fits only big and a worker that fits only small leave the replica unplaced.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_200)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_LT_200)])], + ), + ], + tolerations=None, + ), clusters=[ _cluster( - "cluster-a", - pools=[ - _pool("a", devices=[_gpu_device()]), - _pool("b", devices=[_gpu_device()]), - ], - ) - ], - all_replicas=[ - _collision_replica("my-model-cluster-a-0-dup", "cluster-a", pool="b", index=0), - _replica_with_pool("my-model", "cluster-a", pool="a", index=0), - ], - want=[ - _cand( name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", - pool="a", - device_requests=[_resolved()], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="small", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ), + icv1alpha1.GpuPool( + name="big", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="200Gi")] + ), + ], + taints=None, + placement_labels=None, ) ], - ), -] - - -@pytest.mark.parametrize("case", SCHEDULE_CASES, ids=lambda case: case.name) -def test_schedule(case: Case) -> None: - """The scheduler retains existing pins and places new replicas.""" - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) - assert got == case.want - - -# A caller passes fill=False when it can't yet trust the candidate set (for the -# ModelDeployment, when a referenced ModelCache is unresolved). Retain runs -# unconditionally; only the placement of new replicas is held. -FILL_FALSE_CASES = [ - Case( - name="no replicas yet: nothing is placed", - deployment=_deployment(), - clusters=[_cluster("cluster-a")], all_replicas=[], + fill=True, want=[], ), Case( - name="existing replica is retained despite fill=False", - deployment=_deployment(), - clusters=[_cluster("cluster-a")], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="scale-up shortfall is not filled, only the retained replica remains", - deployment=_deployment(replicas=3), + name="GangTooBig", + reason="A two-node gang whose only matching pool has one node goes unplaced.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), -] - - -@pytest.mark.parametrize("case", FILL_FALSE_CASES, ids=lambda case: case.name) -def test_fill_false_is_retain_only(case: Case) -> None: - """fill=False retains existing replicas but places no new ones.""" - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas, fill=False) - assert got == case.want - - -# nodeSelector device-request matching and pool pinning. -NODE_SELECTOR_CASES = [ - Case( - name="matching request picks the cluster and records the pool", - deployment=_deployment(requests=[_request(cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], - all_replicas=[], - want=[ - _cand( + _cluster( name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="small", nodes=8, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="40Gi")] + ), + icv1alpha1.GpuPool( + name="big", nodes=1, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ), + ], + taints=None, + placement_labels=None, ) ], - ), - Case( - name="non-matching request filters the cluster out", - deployment=_deployment(requests=[_request(cel_exprs=[_MEM_200])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], - all_replicas=[], - want=[], - ), - Case( - name="device count not covered filters out", - # Request 8 GPUs, pool device has only 4. - deployment=_deployment(requests=[_request(count=8, cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=4)])])], all_replicas=[], + fill=True, want=[], ), Case( - name="published device count of zero satisfies no request", - # A pool device published with count 0 must read as "none - # available", not default to 1. Regression: `d.count or 1` - # treated 0 as 1 and placed a replica whose ResourceClaim no - # device could satisfy. The status schema permits 0 even though - # an InferenceClass device count is floored at 1. - deployment=_deployment(requests=[_request(count=1, cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=0)])])], - all_replicas=[], - want=[], - ), - Case( - name="published pool node count of zero hosts nothing", - # An autoscaled-to-zero pool has a matching GPU device but no - # nodes, so it can host no replica. - deployment=_deployment(requests=[_request(count=1, cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=0)])], + name="GangToOtherCluster", + reason="A two-node gang too big for cluster-a's matching pool lands whole on cluster-b.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="small", nodes=8, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="40Gi")] + ), + icv1alpha1.GpuPool( + name="big", nodes=1, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ), + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="big", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], all_replicas=[], - want=[], + fill=True, + want=[ + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="big", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + scheduling.MemberPlacement( + role="Worker", + pool="big", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], ), + # The whole engine could land on pool a, where the worker claims, but + # that would run the leader without the GPU it asked for. Case( - name="synthetic NIC device matches but is not in resolved requests", + name="ClaimableElsewhere", + reason="A leader that matches only a synthetic device in pool a goes with its gang to pool b, where it can claim a GPU.", deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_200)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, ), clusters=[ _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device()])], + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="a", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _gpu_device(name="syn", claim="Synthetic", count=1, memory="200Gi"), + ], + ), + icv1alpha1.GpuPool( + name="b", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="200Gi")] + ), + ], + taints=None, + placement_labels=None, ) ], all_replicas=[], - # Only the claim: DRA gpu request is resolved; the synthetic nic - # matched for scheduling but isn't claimed. + fill=True, want=[ - _cand( + _candidate( name="cluster-a", + index=0, gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved(name="gpu")], - ) + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="b", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_200], + ) + ], + ), + scheduling.MemberPlacement( + role="Worker", + pool="b", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), + # A selector that matches only a synthetic device on every pool is + # deliberate: it pins without claiming. Case( - name="multi-device: missing NIC filters the pool out", - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device()])])], - all_replicas=[], - want=[], - ), - Case( - name="two requests cannot both claim one single-count device", - # Two distinct requests, each matching the same single GPU - # device. DRA allocates distinct devices per request, so a - # count:1 device can satisfy only one. The pool must not match. + name="SyntheticEverywhere", + reason="A leader that matches only the pool's synthetic NIC places claimless alongside its claiming worker.", deployment=_deployment( - requests=[ - _request(name="gpu-a", cel_exprs=[_MEM_141]), - _request(name="gpu-b", cel_exprs=[_MEM_141]), - ] + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=1)])])], + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ) + ], + taints=None, + placement_labels=None, + ) + ], all_replicas=[], - want=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="frontier", + device_requests=[], + ), + scheduling.MemberPlacement( + role="Worker", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], ), Case( - name="two requests against one device must fit within its count", - # Two count:5 requests need 10 GPUs total; the device has 8. - # Capacity is consumed across requests, so the pool must not - # match (regression: an earlier version checked each request - # against the full device count independently). + name="MemberMatchesNowhere", + reason="A worker that matches no pool leaves the whole replica unplaced.", deployment=_deployment( - requests=[ - _request(name="gpu-a", count=5, cel_exprs=[_MEM_141]), - _request(name="gpu-b", count=5, cel_exprs=[_MEM_141]), - ] + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_200)])], + ), + ], + tolerations=None, ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=8)])])], + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], all_replicas=[], + fill=True, want=[], ), Case( - name="two requests sharing a device fit when count covers both", - # 8-GPU device, two count:4 requests = 8 total. Both resolve. + name="ClaimlessLeader", + reason="A leader with no nodeSelector follows the worker's pool at no node cost, so the gang fits a one-node pool.", deployment=_deployment( - requests=[ - _request(name="gpu-a", count=4, cel_exprs=[_MEM_141]), - _request(name="gpu-b", count=4, cel_exprs=[_MEM_141]), - ] + replicas=1, + members=[ + _member(role="Leader", worker_nodes=None, devices=None), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=8)])])], + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=1, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], all_replicas=[], + fill=True, want=[ - _cand( + _candidate( name="cluster-a", + index=0, gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[ - _resolved(name="gpu-a", count=4), - _resolved(name="gpu-b", count=4), + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="frontier", + device_requests=[], + ), + scheduling.MemberPlacement( + role="Worker", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), ], - ) + ), ], ), Case( - name="first matching pool wins (deterministic)", - # Both pools carry a claimable GPU; the synthetic NIC's link type - # is the discriminator. Only the infiniband pool satisfies the - # nic selector, so it wins regardless of pool order. + name="RetainedClaimless", + reason="A replica pinned to the cluster's only pool, claimless leader and worker alike, stays there, where fresh placement would also put both members.", deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] + replicas=1, + members=[ + _member(role="Leader", worker_nodes=None, devices=None), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool("dev", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")]), - _pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), + icv1alpha1.GpuPool( + name="frontier", + nodes=1, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) ], + taints=None, + placement_labels=None, ) ], - all_replicas=[], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member(role="Leader", worker_nodes=None, pool="frontier", device_requests=None), + _replica_member( + role="Worker", + worker_nodes=1, + pool="frontier", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ), + ], + ) + ], + fill=True, want=[ - _cand( + _candidate( name="cluster-a", + index=0, gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved(name="gpu")], - ) + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="frontier", + device_requests=[], + ), + scheduling.MemberPlacement( + role="Worker", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), + # other-model's claimless leader shares its worker's node. Charging the + # claimless leader a node would wrongly report insufficient capacity. Case( - name="synthetic-only selector leaves nothing to claim, pool ineligible", - # The sole request matches a synthetic NIC. The replica's serving - # workload would have no ResourceClaim to bind GPUs through, so - # the pool is not a viable host and nothing is scheduled. - deployment=_deployment(requests=[_request(name="nic", cel_exprs=[_IB])]), + name="OtherClaimless", + reason="Another deployment's claimless leader uses no node, leaving one of two free for a new replica.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")])], + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, ) ], - all_replicas=[], - want=[], - ), - Case( - name="retained replica keeps its pinned pool", - deployment=_deployment(requests=[_request(cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="frontier")], + all_replicas=[ + _replica( + name="other-model-cluster-a-0", + deployment="other-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member(role="Leader", worker_nodes=None, pool="default", device_requests=None), + _replica_member( + role="Worker", + worker_nodes=1, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ), + ], + ) + ], + fill=True, want=[ - _cand( + _candidate( name="cluster-a", + index=0, gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], - ) + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), Case( - name="selector drift re-places replica onto a now-matching pool", - # A claimable GPU keeps both pools viable hosts; the synthetic - # NIC's link type is the drifting discriminator. + name="ShapeChange", + reason="A replica placed as one Standalone member is re-placed when the deployment grows a Worker.", deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool("a", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")]), - _pool("b", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) ], ) ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], + fill=True, want=[ - _cand( + _candidate( name="cluster-a", + index=0, gateway_hostname="cluster-a.clusters.example.com", - pool="b", - device_requests=[_resolved(name="gpu")], - ) + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + scheduling.MemberPlacement( + role="Worker", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), + # Taints on InferenceClusters gate placement; a matching toleration on the + # ModelDeployment overrides them. NoSchedule keeps new replicas off a + # cluster but leaves existing ones; NoExecute additionally drains the + # existing ones, which fill reschedules onto a tolerated cluster. Case( - name="pinned pool that still matches stays pinned (attribute drift is sticky)", + name="NoScheduleTaint", + reason="A NoSchedule taint keeps a new replica off cluster-a, so it lands on cluster-b.", deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool("a", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), - _pool("b", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) ], - ) + taints=[icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule")], + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], + all_replicas=[], + fill=True, want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="a", - device_requests=[_resolved(name="gpu")], - ) + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), Case( - name="no matching pool anywhere drops the replica entirely", + name="NoScheduleRetains", + reason="A NoSchedule taint leaves an existing replica on its cluster.", deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, ), clusters=[ _cluster( - "cluster-a", - pools=[_pool("a", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")])], + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=[icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule")], + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], ) ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], - want=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], ), Case( - name="replica with no pool pin is re-placed when a selector now applies", + name="TolerationAllows", + reason="A toleration for a cluster's taint lets a new replica land there.", deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=[mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists")], ), clusters=[ _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")])], - ) - ], - all_replicas=[_replica("my-model", "cluster-a")], - want=[ - _cand( name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved(name="gpu")], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=[icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule")], + placement_labels=None, ) ], - ), - Case( - name="dropping a non-matching replica frees its node for the refill", - # a/0 is pinned to a pool that still matches (retained). a/1 is - # pinned to a pool no longer published, so it's dropped and will - # be re-placed. The pool has just 2 nodes; both are notionally in - # use by a/0 and a/1. The refill must see a/1's node freeing up - # (it's being deleted) and re-place onto frontier at index 1. - # Regression: the ledger must not charge dropped replicas. - deployment=_deployment(replicas=2, requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=2)])], - all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="frontier", index=0), - _replica_with_pool("my-model", "cluster-a", pool="gone", index=1), - ], + all_replicas=[], + fill=True, want=[ - _cand( + _candidate( name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], - ), - _cand( - name="cluster-a", - index=1, - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], ), ], ), Case( - name="device count is checked against the pinned pool, not a cluster-wide sum", - # Request 8 GPUs. Pool 'a' has 4/node (doesn't fit); pool 'b' - # has 8 and does. The replica must pin to 'b'. - deployment=_deployment(requests=[_request(count=8, cel_exprs=[_MEM_141])]), + name="NoExecuteDrains", + reason="A NoExecute taint drains an existing replica, which is rescheduled onto an untainted cluster.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=[icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute")], + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, pools=[ - _pool("a", devices=[_gpu_device(count=4)]), - _pool("b", devices=[_gpu_device(count=8)]), + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) ], ) ], - all_replicas=[], + fill=True, want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="b", - device_requests=[_resolved(count=8)], - ) + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), -] - - -@pytest.mark.parametrize("case", NODE_SELECTOR_CASES, ids=lambda case: case.name) -def test_node_selector(case: Case) -> None: - """The scheduler places replicas only on pools whose devices satisfy the nodeSelector.""" - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) - assert got == case.want - - -def test_node_selector_invalid_cel_raises() -> None: - """A malformed expression raises CELCompileError, which the caller handles.""" - deployment = _deployment(requests=[_request(cel_exprs=["this is ) not valid ("])]) - with pytest.raises(cel.CELCompileError, match=r"this is \) not valid \("): - scheduling.schedule(deployment, [_cluster("cluster-a", pools=[_pool("frontier")])], []) - - -def _gang( - leader_requests: list[mdv1alpha1.Device] | None, - worker_requests: list[mdv1alpha1.Device] | None, - *, - worker_nodes: int = 1, -) -> mdv1alpha1.Engine: - """A Leader/Worker engine with per-member (possibly heterogeneous) selectors. - - Either member's requests may be None, meaning that member carries no - nodeSelector and claims nothing. - """ - leader = mdv1alpha1.Member(role="Leader", template=_template()) - if leader_requests is not None: - leader.nodeSelector = _node_selector(leader_requests) - worker = mdv1alpha1.Member(role="Worker", worker=mdv1alpha1.Worker(nodes=worker_nodes), template=_template()) - if worker_requests is not None: - worker.nodeSelector = _node_selector(worker_requests) - return mdv1alpha1.Engine(name=_ENGINE, members=[leader, worker]) - - -# Per-member placement: single-pool engines, rejection when no pool fits, and -# claimless ride-along members. -MEMBERS_CASES = [ - Case( - name="a single pool satisfying every member hosts the whole engine", - # The leader's request matches both pools; the worker's only - # matches big. The whole-engine pass must put both members on - # big - the one pool that satisfies them all. - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_200])], + Case( + name="NoExecuteTolerated", + reason="A toleration for cluster-a's NoExecute taint leaves the replica at cluster-a/0, as it would be even if retain drained it, since fill honours the toleration too.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], ) - ] + ], + tolerations=[mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists")], ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool("small", devices=[_gpu_device(memory="141Gi")]), - _pool("big", devices=[_gpu_device(memory="200Gi")]), + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) ], + taints=[icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute")], + placement_labels=None, ) ], - all_replicas=[], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, want=[ - scheduling.Candidate( + _candidate( name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="big", device_requests=[_resolved()]), - scheduling.MemberPlacement( - role="Worker", - pool="big", - device_requests=[_resolved(cel_exprs=[_MEM_200])], - ), + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) ], - ) + ), ], - ) + ), ], ), + # The deploy function surfaces the shortfall. Case( - name="members no single pool satisfies are not scheduled", - # The leader only fits big (>= 200Gi); the worker only fits - # small (< 200Gi). No single pool satisfies both. The scheduler - # never splits an engine across pools - it can't tell whether - # big and small share a fabric - so the engine is rejected and - # the replica goes unplaced (#149). + name="NoExecuteNowhere", + reason="A NoExecute taint drains the only replica, and with no other cluster nothing replaces it.", deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_200])], - [_request(cel_exprs=[_MEM_LT_200])], + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], ) - ] + ], + tolerations=None, ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool("small", devices=[_gpu_device(memory="141Gi")]), - _pool("big", devices=[_gpu_device(memory="200Gi")]), + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) ], + taints=[icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute")], + placement_labels=None, ) ], - all_replicas=[], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, want=[], ), Case( - name="a gang too big for its only matching pool is rejected", - # Both members match only big (>= 141Gi); small (40Gi) matches - # neither. big has one free node but the gang needs two. The - # engine doesn't fit any single pool, so it's rejected; with big - # the only cluster the replica goes unplaced. + name="WrongEffect", + reason="A NoSchedule toleration doesn't cover a NoExecute taint with the same key, so the replica is drained to cluster-b.", deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_141])], + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], ) - ] + ], + tolerations=[ + mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists", effect="NoSchedule") + ], ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=[icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute")], + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, pools=[ - _pool("small", nodes=8, devices=[_gpu_device(memory="40Gi")]), - _pool("big", nodes=1, devices=[_gpu_device(memory="141Gi")]), + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) ], ) ], - all_replicas=[], - want=[], + fill=True, + want=[ + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], ), Case( - name="a gang too big for one cluster's pool lands whole on another", - # Same gang. cluster-a's matching pool has only one free node - # (too few for the two-member gang), so the scheduler rejects - # cluster-a and places the whole gang on cluster-b, whose pool - # has room for both members. + name="SecondTaint", + reason="Tolerating one of a cluster's two taints isn't enough, so a new replica lands on cluster-b.", deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_141])], + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], ) - ] + ], + tolerations=[mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists")], ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool("small", nodes=8, devices=[_gpu_device(memory="40Gi")]), - _pool("big", nodes=1, devices=[_gpu_device(memory="141Gi")]), + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) ], + taints=[ + icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule"), + icv1alpha1.Taint(key="modelplane.ai/reserved", effect="NoSchedule"), + ], + placement_labels=None, ), _cluster( - "cluster-b", + name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("big", nodes=2, devices=[_gpu_device(memory="141Gi")])], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, ), ], all_replicas=[], + fill=True, want=[ - scheduling.Candidate( + _candidate( name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="big", device_requests=[_resolved()]), - scheduling.MemberPlacement(role="Worker", pool="big", device_requests=[_resolved()]), + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) ], - ) + ), ], - ) + ), ], ), Case( - name="a member claimable elsewhere is not stranded on a synthetic match", - # On pool-a the leader's request matches only a Synthetic - # device (nothing to claim) while the worker claims, so the - # whole engine *could* land there - but pool-b satisfies the - # leader claimably. The engine must go to pool-b; placing on - # pool-a would run the leader without the GPU it asked for. + name="EqualValueMatches", + reason="An Equal toleration whose key and value match the taint's lets a new replica land.", deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_200])], - [_request(cel_exprs=[_MEM_141])], + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], ) - ] + ], + tolerations=[mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="on")], ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool( - "a", - devices=[ - _gpu_device(memory="141Gi"), - _gpu_device(name="syn", claim="Synthetic", memory="200Gi"), - ], - ), - _pool("b", devices=[_gpu_device(memory="200Gi")]), + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) ], + taints=[icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule")], + placement_labels=None, ) ], all_replicas=[], + fill=True, want=[ - scheduling.Candidate( + _candidate( name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement( - role="Leader", - pool="b", - device_requests=[_resolved(cel_exprs=[_MEM_200])], - ), - scheduling.MemberPlacement( - role="Worker", - pool="b", - device_requests=[_resolved(cel_exprs=[_MEM_141])], - ), + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) ], - ) + ), ], - ) + ), ], ), Case( - name="a member synthetic-only everywhere places claimless with its gang", - # The leader's request matches only the pool's synthetic NIC on - # every pool - deliberate (a selector that pins without - # claiming). It places claimless alongside the claiming worker. + name="EqualValueDiffers", + reason="An Equal toleration whose value differs from the taint's keeps a new replica off.", deployment=_deployment( - engines=[ - _gang( - [_request(name="nic", cel_exprs=[_IB])], - [_request(cel_exprs=[_MEM_141])], + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], ) - ] + ], + tolerations=[mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="off")], ), clusters=[ _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device()])], - ) - ], - all_replicas=[], - want=[ - scheduling.Candidate( name="cluster-a", - index=0, gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), - scheduling.MemberPlacement(role="Worker", pool="frontier", device_requests=[_resolved()]), - ], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] ) ], + taints=[icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule")], + placement_labels=None, ) ], + all_replicas=[], + fill=True, + want=[], ), Case( - name="a member that matches nowhere fails the whole replica", + name="KeylessExists", + reason="An Exists toleration with no key tolerates both of a cluster's taints.", deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_200])], + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], ) - ] + ], + tolerations=[mdv1alpha1.Toleration(operator="Exists")], ), - clusters=[_cluster("cluster-a", pools=[_pool("default", devices=[_gpu_device(memory="141Gi")])])], - all_replicas=[], - want=[], - ), - Case( - name="a claimless leader rides along on its gang's pool at zero cost", - # The leader carries no nodeSelector: it claims nothing, follows - # the worker's pool, and costs no nodes - the 1-node pool fits - # the whole gang because only the worker occupies a node. - deployment=_deployment(engines=[_gang(None, [_request(cel_exprs=[_MEM_141])])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], - all_replicas=[], - want=[ - scheduling.Candidate( + clusters=[ + _cluster( name="cluster-a", - index=0, gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), - scheduling.MemberPlacement(role="Worker", pool="frontier", device_requests=[_resolved()]), - ], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] ) ], - ) - ], - ), - Case( - name="a retained replica's claimless member keeps its pin", - deployment=_deployment(engines=[_gang(None, [_request(cel_exprs=[_MEM_141])])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], - all_replicas=[ - _replica( - "my-model", - "cluster-a", - engines=[ - mrv1alpha1.Engine( - name=_ENGINE, - members=[ - mrv1alpha1.Member( - role="Leader", - nodePoolName="frontier", - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - mrv1alpha1.Member( - role="Worker", - worker=mrv1alpha1.Worker(nodes=1), - nodePoolName="frontier", - deviceRequests=_replica_device_requests(), - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - ], - ) + taints=[ + icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule"), + icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute"), ], + placement_labels=None, ) ], + all_replicas=[], + fill=True, want=[ - scheduling.Candidate( + _candidate( name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), - scheduling.MemberPlacement(role="Worker", pool="frontier", device_requests=[_resolved()]), + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) ], - ) + ), ], - ) + ), ], ), + # The labels go on to the ModelReplica and ModelEndpoint composed from the + # Candidate. This is how a self-hosted endpoint gets its region: a + # ModelService selects endpoints by label, so without them a region-scoped + # service can't select its own replicas, and it can't label them by hand + # because Modelplane owns them. Case( - name="another deployment's claimless member consumes no capacity", - # other-model's gang occupies only its worker's node: its - # claimless leader shares that node. The 2-node pool has 1 node - # free, so our 1-node deployment fits. Charging the claimless - # leader a node would wrongly report insufficient capacity. - deployment=_deployment(), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], - all_replicas=[ - _replica( - "other-model", - "cluster-a", - engines=[ - mrv1alpha1.Engine( - name="main", - members=[ - mrv1alpha1.Member( - role="Leader", - nodePoolName="default", - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - mrv1alpha1.Member( - role="Worker", - worker=mrv1alpha1.Worker(nodes=1), - nodePoolName="default", - deviceRequests=_replica_device_requests(), - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - ], + name="PlacementLabels", + reason="A cluster's placement labels reach the candidate placed there.", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] ) ], + taints=None, + placement_labels={"example.org/region": "eu"}, ) ], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="a member shape change re-places the replica", - # The deployment grew a Worker (Standalone -> Leader+Worker). - # The observed single-member replica no longer lines up, so it - # is re-placed with the new shape. - deployment=_deployment(engines=[_gang([_request()], [_request()])]), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], - all_replicas=[_replica("my-model", "cluster-a")], + all_replicas=[], + fill=True, want=[ - scheduling.Candidate( + _candidate( name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="default", device_requests=[_resolved()]), - scheduling.MemberPlacement(role="Worker", pool="default", device_requests=[_resolved()]), + placement_labels={"example.org/region": "eu"}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) ], - ) + ), ], - ) + ), ], ), ] -@pytest.mark.parametrize("case", MEMBERS_CASES, ids=lambda case: case.name) -def test_members(case: Case) -> None: - """The scheduler places every member of an engine on one pool that fits them all.""" - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) - assert got == case.want - - -# Taints on InferenceClusters gate placement; a matching toleration on the -# ModelDeployment overrides them. NoSchedule keeps new replicas off a cluster but -# leaves existing ones; NoExecute additionally drains the existing ones, which -# fill reschedules onto a tolerated cluster. -_MAINT = icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule") -_DECOMM = icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute") - - -def _names(got: list[scheduling.Candidate]) -> list[tuple[str, int]]: - return [(c.name, c.index) for c in got] - - -def test_taints_noschedule_keeps_new_replicas_off() -> None: - """A NoSchedule taint keeps a new replica off the cluster.""" - clusters = [ - _cluster("cluster-a", taints=[_MAINT]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1), clusters, []) - assert _names(got) == [("cluster-b", 0)] - - -def test_taints_noschedule_leaves_existing_replica_in_place() -> None: - """A NoSchedule taint leaves an existing replica where it is.""" - existing = _replica("my-model", "cluster-a") - got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a", taints=[_MAINT])], [existing]) - assert _names(got) == [("cluster-a", 0)] - - -def test_taints_toleration_allows_placement_on_tainted() -> None: - """A matching toleration lets a new replica onto a tainted cluster.""" - tol = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists") - got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), [_cluster("cluster-a", taints=[_MAINT])], []) - assert _names(got) == [("cluster-a", 0)] - - -def test_taints_noexecute_drains_and_reschedules() -> None: - """A NoExecute taint drains an existing replica onto another cluster.""" - existing = _replica("my-model", "cluster-a") - clusters = [ - _cluster("cluster-a", taints=[_DECOMM]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1), clusters, [existing]) - assert _names(got) == [("cluster-b", 0)] - - -def test_taints_noexecute_toleration_retains_in_place() -> None: - """A NoExecute toleration keeps an existing replica on a draining cluster.""" - tol = mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists") - existing = _replica("my-model", "cluster-a") - got = scheduling.schedule( - _deployment(replicas=1, tolerations=[tol]), [_cluster("cluster-a", taints=[_DECOMM])], [existing] - ) - assert _names(got) == [("cluster-a", 0)] - - -def test_taints_noexecute_drain_leaves_count_unmet_when_nowhere_to_go() -> None: - """Draining with no tolerated cluster to go to yields fewer than spec.replicas.""" - # The deploy function surfaces the shortfall. - existing = _replica("my-model", "cluster-a") - got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a", taints=[_DECOMM])], [existing]) - assert got == [] - - -def test_taints_noschedule_toleration_does_not_cover_a_noexecute_taint() -> None: - """A toleration that matches the key but not the effect doesn't tolerate.""" - # An operator who tolerates only NoSchedule is still drained by a NoExecute - # taint. - tol = mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists", effect="NoSchedule") - existing = _replica("my-model", "cluster-a") - clusters = [ - _cluster("cluster-a", taints=[_DECOMM]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, [existing]) - assert _names(got) == [("cluster-b", 0)] - - -def test_taints_untolerated_second_taint_still_repels() -> None: - """Any untolerated taint keeps new replicas off, even if another is tolerated.""" - other = icv1alpha1.Taint(key="modelplane.ai/reserved", effect="NoSchedule") - tol = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists") - clusters = [ - _cluster("cluster-a", taints=[_MAINT, other]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, []) - assert _names(got) == [("cluster-b", 0)] - - -def test_taints_equal_toleration_matches_on_value() -> None: - """An Equal toleration tolerates only when key and value both match.""" - match = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="on") - placed = scheduling.schedule( - _deployment(replicas=1, tolerations=[match]), [_cluster("cluster-a", taints=[_MAINT])], [] - ) - assert _names(placed) == [("cluster-a", 0)] - - mismatch = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="off") - repelled = scheduling.schedule( - _deployment(replicas=1, tolerations=[mismatch]), [_cluster("cluster-a", taints=[_MAINT])], [] - ) - assert repelled == [] - - -def test_taints_keyless_exists_tolerates_every_taint() -> None: - """An Exists toleration with no key tolerates any taint on the cluster.""" - tol = mdv1alpha1.Toleration(operator="Exists") - clusters = [_cluster("cluster-a", taints=[_MAINT, _DECOMM])] - got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, []) - assert _names(got) == [("cluster-a", 0)] - - -# A cluster's placement labels reach the Candidate, and so the ModelReplica and -# ModelEndpoint composed from it. -# -# This is how a self-hosted endpoint gets its region: a ModelService selects -# endpoints by label, so without it a region-scoped service can't select its own -# replicas, and it can't label them by hand because Modelplane owns them. - - -def test_placement_labels_reach_the_candidate() -> None: - """A cluster's placement labels reach the Candidate.""" - got = scheduling.schedule( - _deployment(replicas=1), - [_cluster("cluster-a", placement_labels={"example.org/region": "eu"})], - [], - ) - assert [c.placement_labels for c in got] == [{"example.org/region": "eu"}] +@pytest.mark.parametrize("case", SCHEDULE_CASES, ids=lambda case: case.name) +def test_schedule(case: Case) -> None: + """schedule() retains existing replicas and places new ones on clusters that can host them.""" + got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas, fill=case.fill) + assert got == case.want, case.reason -def test_placement_labels_a_cluster_declaring_none_yields_none() -> None: - """A cluster that declares no placement labels yields a Candidate with none.""" - got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a")], []) - assert [c.placement_labels for c in got] == [{}] +# SCHEDULE_CASES' want is a list of Candidates, so the one call that raises is +# tested here. +def test_schedule_error() -> None: + """A malformed expression raises CELCompileError, which the caller handles.""" + with pytest.raises(cel.CELCompileError, match=r"this is \) not valid \("): + scheduling.schedule( + _deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device( + name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel="this is ) not valid (")] + ) + ], + ) + ], + tolerations=None, + ), + [ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + [], + ) diff --git a/functions/compose-model-deployment/tests/test_semver.py b/functions/compose-model-deployment/tests/test_semver.py index 433a3f058..7ef6d4d8a 100644 --- a/functions/compose-model-deployment/tests/test_semver.py +++ b/functions/compose-model-deployment/tests/test_semver.py @@ -34,98 +34,345 @@ from function import cel, semver -def _eval(expr: str) -> bool: - """Compile and evaluate a deviceless boolean CEL expression.""" - return cel.Program(expr).matches({}) - - @dataclasses.dataclass -class Case: +class SemverCase: + """A test case for a semver CEL expression.""" + name: str + reason: str expr: str want: bool @dataclasses.dataclass -class ParseErrCase: +class ParseRejectsCase: + """A test case for a version string semver.parse rejects.""" + name: str - input: str + reason: str + s: str + want: str SEMVER_CASES = [ - # parse + doc-comment examples. - Case(name="parse", expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', want=True), - Case(name="parse with prerelease", expr='semver("0.1.0-alpha.1").major() == 0', want=True), + # parse + doc-comment examples. Upstream's parse row returns the version. + # Wrapped to return a bool, it matches CompareEqual below. Both are kept + # to mirror upstream's table. + SemverCase( + name="Parse", + reason="semver() parses 1.2.3 into a version that compares equal to 1.2.3.", + expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', + want=True, + ), + SemverCase( + name="ParsePrerelease", + reason="semver() parses 0.1.0-alpha.1, prerelease and all, as major version 0.", + expr='semver("0.1.0-alpha.1").major() == 0', + want=True, + ), # isSemver strict. - Case(name="isSemver full", expr='isSemver("1.2.3-beta.1+build.1")', want=True), - Case(name="isSemver simple", expr='isSemver("1.0.0")', want=True), - Case(name="isSemver hello", expr='isSemver("hello")', want=False), - Case(name="isSemver empty false", expr='isSemver("")', want=False), - Case(name="isSemver v prefix false", expr='isSemver("v1.0.0")', want=False), - Case(name="isSemver v1.0 false", expr='isSemver("v1.0")', want=False), - Case(name="isSemver leading whitespace false", expr='isSemver(" 1.0.0")', want=False), - Case(name="isSemver inner whitespace false", expr='isSemver("1. 0.0")', want=False), - Case(name="isSemver trailing whitespace false", expr='isSemver("1.0.0 ")', want=False), - Case(name="isSemver leading zeros false", expr='isSemver("01.01.01")', want=False), - Case(name="isSemver major only false", expr='isSemver("1")', want=False), - Case(name="isSemver major minor only false", expr='isSemver("1.1")', want=False), - Case(name="isSemver 200K", expr='isSemver("200K")', want=False), - Case(name="isSemver Mi", expr='isSemver("Mi")', want=False), + SemverCase( + name="IsSemver", + reason="isSemver accepts a version with a prerelease and build metadata.", + expr='isSemver("1.2.3-beta.1+build.1")', + want=True, + ), + SemverCase( + name="IsSemverSimple", + reason="isSemver accepts 1.0.0.", + expr='isSemver("1.0.0")', + want=True, + ), + SemverCase( + name="IsSemverWord", + reason="isSemver rejects the word hello.", + expr='isSemver("hello")', + want=False, + ), + SemverCase( + name="IsSemverEmpty", + reason="isSemver rejects an empty string.", + expr='isSemver("")', + want=False, + ), + SemverCase( + name="IsSemverVPrefix", + reason="isSemver rejects v1.0.0, because of its v prefix.", + expr='isSemver("v1.0.0")', + want=False, + ), + SemverCase( + name="IsSemverShortPrefixed", + reason="isSemver rejects v1.0, which has a v prefix and no patch version.", + expr='isSemver("v1.0")', + want=False, + ), + SemverCase( + name="IsSemverLeadingWhitespace", + reason="isSemver rejects a version with leading whitespace.", + expr='isSemver(" 1.0.0")', + want=False, + ), + SemverCase( + name="IsSemverContainsWhitespace", + reason="isSemver rejects a version with whitespace inside it.", + expr='isSemver("1. 0.0")', + want=False, + ), + SemverCase( + name="IsSemverTrailingWhitespace", + reason="isSemver rejects a version with trailing whitespace.", + expr='isSemver("1.0.0 ")', + want=False, + ), + SemverCase( + name="IsSemverLeadingZeros", + reason="isSemver rejects 01.01.01, because of its leading zeros.", + expr='isSemver("01.01.01")', + want=False, + ), + SemverCase( + name="IsSemverMajorOnly", + reason="isSemver rejects 1, which has no minor or patch version.", + expr='isSemver("1")', + want=False, + ), + SemverCase( + name="IsSemverNoPatch", + reason="isSemver rejects 1.1, which has no patch version.", + expr='isSemver("1.1")', + want=False, + ), + SemverCase( + name="IsSemverQuantity", + reason="isSemver rejects the quantity 200K.", + expr='isSemver("200K")', + want=False, + ), + SemverCase( + name="IsSemverBareSuffix", + reason="isSemver rejects Mi, a bare quantity suffix.", + expr='isSemver("Mi")', + want=False, + ), # isSemver normalize overload. Normalization does NOT trim whitespace. - Case(name="isSemver empty normalize false", expr='isSemver("", true)', want=False), - Case(name="isSemver leading whitespace normalize false", expr='isSemver(" 1.0.0", true)', want=False), - Case(name="isSemver inner whitespace normalize false", expr='isSemver("1. 0.0", true)', want=False), - Case(name="isSemver trailing whitespace normalize false", expr='isSemver("1.0.0 ", true)', want=False), - Case(name="isSemver v prefix normalize true", expr='isSemver("v1.0.0", true)', want=True), - Case(name="isSemver leading zeros normalize true", expr='isSemver("01.01.01", true)', want=True), - Case(name="isSemver major only normalize true", expr='isSemver("1", true)', want=True), - Case(name="isSemver major minor only normalize true", expr='isSemver("1.1", true)', want=True), + SemverCase( + name="NormalizeEmpty", + reason="isSemver with normalize rejects an empty string.", + expr='isSemver("", true)', + want=False, + ), + SemverCase( + name="NormalizeLeadingWhitespace", + reason="isSemver with normalize rejects a version with leading whitespace.", + expr='isSemver(" 1.0.0", true)', + want=False, + ), + SemverCase( + name="NormalizeContainsWhitespace", + reason="isSemver with normalize rejects a version with whitespace inside it.", + expr='isSemver("1. 0.0", true)', + want=False, + ), + SemverCase( + name="NormalizeTrailingWhitespace", + reason="isSemver with normalize rejects a version with trailing whitespace.", + expr='isSemver("1.0.0 ", true)', + want=False, + ), + SemverCase( + name="NormalizeVPrefix", + reason="isSemver with normalize accepts v1.0.0, v prefix and all.", + expr='isSemver("v1.0.0", true)', + want=True, + ), + SemverCase( + name="NormalizeLeadingZeros", + reason="isSemver with normalize accepts 01.01.01, leading zeros and all.", + expr='isSemver("01.01.01", true)', + want=True, + ), + SemverCase( + name="NormalizeMajorOnly", + reason="isSemver with normalize accepts 1, a major version alone.", + expr='isSemver("1", true)', + want=True, + ), + SemverCase( + name="NormalizeNoPatch", + reason="isSemver with normalize accepts 1.1, which has no patch version.", + expr='isSemver("1.1", true)', + want=True, + ), # normalize equality and semver(...) examples. - Case(name="equality normalize", expr='semver("v01.01", true) == semver("1.1.0")', want=True), - Case(name="semver v prefix normalize major", expr='semver("v1.0.0", true).major() == 1', want=True), - Case(name="semver short normalize patch", expr='semver("1.0", true).patch() == 0', want=True), - Case(name="semver leading zeros normalize", expr='semver("01.01.01", true).minor() == 1', want=True), + SemverCase( + name="EqualityNormalize", + reason="v01.01, normalized, equals 1.1.0.", + expr='semver("v01.01", true) == semver("1.1.0")', + want=True, + ), + SemverCase( + name="SemverNormalizeVPrefix", + reason="semver() with normalize parses v1.0.0 as major version 1.", + expr='semver("v1.0.0", true).major() == 1', + want=True, + ), + SemverCase( + name="SemverNormalizeNoPatch", + reason="semver() with normalize parses 1.0 as patch version 0.", + expr='semver("1.0", true).patch() == 0', + want=True, + ), + SemverCase( + name="SemverNormalizeLeadingZeros", + reason="semver() with normalize parses 01.01.01 as minor version 1.", + expr='semver("01.01.01", true).minor() == 1', + want=True, + ), # equality / comparison. - Case(name="equality reflexivity", expr='semver("1.2.3") == semver("1.2.3")', want=True), - Case(name="inequality", expr='semver("1.2.3") == semver("1.0.0")', want=False), - Case(name="less", expr='semver("1.0.0").isLessThan(semver("1.2.3"))', want=True), - Case(name="less false", expr='semver("1.0.0").isLessThan(semver("1.0.0"))', want=False), - Case(name="greater", expr='semver("1.2.3").isGreaterThan(semver("1.0.0"))', want=True), - Case(name="greater false", expr='semver("1.0.0").isGreaterThan(semver("1.0.0"))', want=False), - Case(name="compare equal", expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', want=True), - Case(name="compare less", expr='semver("1.2.3").compareTo(semver("2.0.0")) == -1', want=True), - Case(name="compare greater", expr='semver("1.2.3").compareTo(semver("0.1.2")) == 1', want=True), + SemverCase( + name="EqualityReflexivity", + reason="1.2.3 equals itself.", + expr='semver("1.2.3") == semver("1.2.3")', + want=True, + ), + SemverCase( + name="Inequality", + reason="1.2.3 doesn't equal 1.0.0.", + expr='semver("1.2.3") == semver("1.0.0")', + want=False, + ), + SemverCase( + name="IsLessThan", + reason="1.0.0 is less than 1.2.3.", + expr='semver("1.0.0").isLessThan(semver("1.2.3"))', + want=True, + ), + SemverCase( + name="IsLessThanFalse", + reason="1.0.0 isn't less than itself.", + expr='semver("1.0.0").isLessThan(semver("1.0.0"))', + want=False, + ), + SemverCase( + name="IsGreaterThan", + reason="1.2.3 is greater than 1.0.0.", + expr='semver("1.2.3").isGreaterThan(semver("1.0.0"))', + want=True, + ), + SemverCase( + name="IsGreaterThanFalse", + reason="1.0.0 isn't greater than itself.", + expr='semver("1.0.0").isGreaterThan(semver("1.0.0"))', + want=False, + ), + SemverCase( + name="CompareEqual", + reason="1.2.3 compares equal to itself.", + expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', + want=True, + ), + SemverCase( + name="CompareLess", + reason="1.2.3 compares less than 2.0.0.", + expr='semver("1.2.3").compareTo(semver("2.0.0")) == -1', + want=True, + ), + SemverCase( + name="CompareGreater", + reason="1.2.3 compares greater than 0.1.2.", + expr='semver("1.2.3").compareTo(semver("0.1.2")) == 1', + want=True, + ), # major / minor / patch. - Case(name="major", expr='semver("1.2.3").major() == 1', want=True), - Case(name="minor", expr='semver("1.2.3").minor() == 2', want=True), - Case(name="patch", expr='semver("1.2.3").patch() == 3', want=True), - # A bad version is a runtime error upstream -> non-match here. - Case(name="bad version is non-match", expr='semver("v1.0").major() == 1', want=False), + SemverCase( + name="Major", + reason="1.2.3 has major version 1.", + expr='semver("1.2.3").major() == 1', + want=True, + ), + SemverCase( + name="Minor", + reason="1.2.3 has minor version 2.", + expr='semver("1.2.3").minor() == 2', + want=True, + ), + SemverCase( + name="Patch", + reason="1.2.3 has patch version 3.", + expr='semver("1.2.3").patch() == 3', + want=True, + ), + SemverCase( + name="ParseInvalidVersion", + reason="Reading the major version of v1.0, which isn't valid, doesn't match; upstream raises instead.", + expr='semver("v1.0").major() == 1', + want=False, + ), ] @pytest.mark.parametrize("case", SEMVER_CASES, ids=lambda case: case.name) -def test_semver(case: Case) -> None: +def test_semver(case: SemverCase) -> None: """A semver CEL expression evaluates as it does upstream.""" - assert _eval(case.expr) == case.want + got = cel.Program(case.expr).matches({}) + assert got == case.want, case.reason PARSE_REJECTS_CASES = [ - ParseErrCase(name="v prefix", input="v1.0"), - ParseErrCase(name="major only", input="1"), - ParseErrCase(name="major minor only", input="1.1"), - ParseErrCase(name="leading zeros", input="01.01.01"), - ParseErrCase(name="leading whitespace", input=" 1.0.0"), - ParseErrCase(name="trailing whitespace", input="1.0.0 "), - ParseErrCase(name="empty", input=""), - ParseErrCase(name="word", input="hello"), + ParseRejectsCase( + name="VPrefixNoPatch", + reason="parse rejects v1.0, which has only two parts, before it reads the v prefix.", + s="v1.0", + want=r"no Major\.Minor\.Patch elements found", + ), + ParseRejectsCase( + name="MajorOnly", + reason="parse rejects 1, which has no minor or patch version.", + s="1", + want=r"no Major\.Minor\.Patch elements found", + ), + ParseRejectsCase( + name="NoPatch", + reason="parse rejects 1.1, which has no patch version.", + s="1.1", + want=r"no Major\.Minor\.Patch elements found", + ), + ParseRejectsCase( + name="LeadingZeros", + reason="parse rejects 01.01.01, because its major version has a leading zero.", + s="01.01.01", + want="major number must not contain leading zeroes: '01'", + ), + ParseRejectsCase( + name="LeadingWhitespace", + reason="parse rejects a version with leading whitespace.", + s=" 1.0.0", + want=r"invalid character\(s\) in major number: ' 1'", + ), + ParseRejectsCase( + name="TrailingWhitespace", + reason="parse rejects a version with trailing whitespace.", + s="1.0.0 ", + want=r"invalid character\(s\) in patch number: '0 '", + ), + ParseRejectsCase( + name="Empty", + reason="parse rejects an empty string.", + s="", + want="version string empty", + ), + ParseRejectsCase( + name="Word", + reason="parse rejects the word hello.", + s="hello", + want=r"no Major\.Minor\.Patch elements found", + ), ] @pytest.mark.parametrize("case", PARSE_REJECTS_CASES, ids=lambda case: case.name) -def test_parse_rejects(case: ParseErrCase) -> None: +def test_parse_rejects(case: ParseRejectsCase) -> None: """parse() (strict) rejects what blang/semver Parse rejects.""" - with pytest.raises( - ValueError, match=r"version string empty|no Major\.Minor\.Patch|invalid character|leading zeroes" - ): - semver.parse(case.input) + with pytest.raises(ValueError, match=case.want): + semver.parse(case.s) diff --git a/functions/compose-model-endpoint/tests/test_fn.py b/functions/compose-model-endpoint/tests/test_fn.py index a149608a9..f9d7845d2 100644 --- a/functions/compose-model-endpoint/tests/test_fn.py +++ b/functions/compose-model-endpoint/tests/test_fn.py @@ -15,7 +15,6 @@ """Tests for the compose-model-endpoint function.""" import asyncio -import base64 import dataclasses import json @@ -27,9 +26,7 @@ from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.modelendpoint import v1alpha1 - -_NS = "ml-team" -_NAME = "together-kimi-k2" +from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @dataclasses.dataclass @@ -37,196 +34,212 @@ class Case: """A test case for compose-model-endpoint.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse -def _xr(**spec) -> dict: # noqa: ANN003 - """The ModelEndpoint XR, built from the generated model so a field the XRD - doesn't define can't creep into a test.""" - xr = v1alpha1.ModelEndpoint( - apiVersion="modelplane.ai/v1alpha1", - kind="ModelEndpoint", - metadata={"name": _NAME, "namespace": _NS}, - spec=v1alpha1.Spec(origin="https://api.together.xyz", **spec), - ) - return xr.model_dump(exclude_none=True, mode="json", by_alias=True) - - -def _api_key(secret: str, key: str = "apiKey") -> v1alpha1.Credential: - """An API key credential read from the named Secret.""" - return v1alpha1.Credential( - method="APIKey", apiKey=v1alpha1.ApiKey(secretRef=v1alpha1.SecretRef(name=secret, key=key)) +def _model_endpoint(*, credential_key: str | None) -> fnv1.Resource: + """The together-kimi-k2 XR, with an API key under credential_key of together-api-key, or no credential if None.""" + credential = None + if credential_key is not None: + credential = v1alpha1.Credential( + method="APIKey", + apiKey=v1alpha1.ApiKey(secretRef=v1alpha1.SecretRef(name="together-api-key", key=credential_key)), + ) + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ModelEndpoint( + apiVersion="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + metadata=metav1.ObjectMeta(name="together-kimi-k2", namespace="ml-team"), + spec=v1alpha1.Spec(origin="https://api.together.xyz", credential=credential), + ).model_dump(exclude_none=True, mode="json", by_alias=True) + ) ) -def _secret(name: str, data: dict[str, str]) -> dict: - """A Secret as the API server stores it, values base64 encoded.""" - return { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": name, "namespace": _NS}, - "data": {k: base64.b64encode(v.encode()).decode() for k, v in data.items()}, - } - - -def _credential_requirement(name: str) -> fnv1.Requirements: - return fnv1.Requirements( - resources={"credential": fnv1.ResourceSelector(api_version="v1", kind="Secret", match_name=name, namespace=_NS)} +def _credential_secret(*, key: str) -> fnv1.Resource: + """The together-api-key Secret, as the credential requirement returns it, holding sk-abc under key.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "together-api-key", "namespace": "ml-team"}, + # Base64 encoded, as the API server stores it. c2stYWJj is + # "sk-abc". + "data": {key: "c2stYWJj"}, + } + ) ) -def _response( - *, - reason: str, - status: fnv1.Status, - message: str | None = None, - requirements: fnv1.Requirements | None = None, -) -> fnv1.RunFunctionResponse: - """The whole response. This function composes no resources, so desired - carries only the composite's readiness, which mirrors EndpointReady, and - asserting the whole thing proves it stays that way.""" - ready = fnv1.READY_TRUE if status == fnv1.STATUS_CONDITION_TRUE else fnv1.READY_FALSE - rsp = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=ready)), - context=structpb.Struct(), - conditions=[ - fnv1.Condition(type=fn.CONDITION_TYPE_ENDPOINT_READY, status=status, reason=reason, message=message) - ], - ) - if requirements is not None: - rsp.requirements.CopyFrom(requirements) - if message is not None: - rsp.results.append(fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message=message)) - return rsp +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) +# This function composes no resources, so desired carries only the composite's +# readiness, which mirrors EndpointReady. Comparing the whole response proves it +# stays that way. The desired XR is written inline although every case has it: +# it's a bare fnv1.Resource carrying only readiness, so a helper would only +# rename its constructor. COMPOSE_CASES = [ Case( - name="no credential: usable as soon as it exists", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - ), - want=_response( - reason=fn.CONDITION_REASON_ENDPOINT_USABLE, - status=fnv1.STATUS_CONDITION_TRUE, + name="NoCredential", + reason="A ModelEndpoint with no credential is usable as soon as it exists, without requiring a Secret.", + req=fnv1.RunFunctionRequest(observed=fnv1.State(composite=_model_endpoint(credential_key=None))), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_TRUE)), + context=structpb.Struct(), + conditions=[ + fnv1.Condition(type="EndpointReady", status=fnv1.STATUS_CONDITION_TRUE, reason="EndpointUsable"), + ], ), ), Case( - name="a credential that resolves", + name="CredentialResolved", + reason="A ModelEndpoint whose credential Secret holds its key reports itself usable.", req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) - ), - required_resources={ - "credential": fnv1.Resources( - items=[ - fnv1.Resource( - resource=resource.dict_to_struct(_secret("together-api-key", {"apiKey": "sk-abc"})) - ) - ] - ) - }, + observed=fnv1.State(composite=_model_endpoint(credential_key="apiKey")), + required_resources={"credential": fnv1.Resources(items=[_credential_secret(key="apiKey")])}, ), - want=_response( - reason=fn.CONDITION_REASON_ENDPOINT_USABLE, - status=fnv1.STATUS_CONDITION_TRUE, - requirements=_credential_requirement("together-api-key"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_TRUE)), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "credential": fnv1.ResourceSelector( + api_version="v1", kind="Secret", match_name="together-api-key", namespace="ml-team" + ) + } + ), + conditions=[ + fnv1.Condition(type="EndpointReady", status=fnv1.STATUS_CONDITION_TRUE, reason="EndpointUsable"), + ], ), ), Case( - name="a credential Secret that does not exist", + name="SecretNotFound", + reason="A ModelEndpoint whose credential Secret doesn't exist reports CredentialMissing and isn't ready.", req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) - ), - required_resources={"credential": fnv1.Resources(items=[])}, + observed=fnv1.State(composite=_model_endpoint(credential_key="apiKey")), + required_resources={"credential": fnv1.Resources()}, ), - want=_response( - reason=fn.CONDITION_REASON_CREDENTIAL_MISSING, - status=fnv1.STATUS_CONDITION_FALSE, - message="Secret together-api-key does not exist", - requirements=_credential_requirement("together-api-key"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Secret together-api-key does not exist")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "credential": fnv1.ResourceSelector( + api_version="v1", kind="Secret", match_name="together-api-key", namespace="ml-team" + ) + } + ), + conditions=[ + fnv1.Condition( + type="EndpointReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="CredentialMissing", + message="Secret together-api-key does not exist", + ), + ], ), ), + # A Secret that exists but lacks the key is the likelier mistake, and would + # otherwise surface as a 401 from the provider. Case( - # A Secret that exists but lacks the key is the likelier mistake, - # and would otherwise surface as a 401 from the provider. - name="a credential Secret missing the key", + name="KeyMissing", + reason="A ModelEndpoint whose credential Secret lacks its key reports CredentialMissing and isn't ready.", req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) - ), - required_resources={ - "credential": fnv1.Resources( - items=[ - fnv1.Resource( - resource=resource.dict_to_struct(_secret("together-api-key", {"token": "sk-abc"})) - ) - ] - ) - }, + observed=fnv1.State(composite=_model_endpoint(credential_key="apiKey")), + required_resources={"credential": fnv1.Resources(items=[_credential_secret(key="token")])}, ), - want=_response( - reason=fn.CONDITION_REASON_CREDENTIAL_MISSING, - status=fnv1.STATUS_CONDITION_FALSE, - message="Secret together-api-key has no key apiKey", - requirements=_credential_requirement("together-api-key"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Secret together-api-key has no key apiKey")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "credential": fnv1.ResourceSelector( + api_version="v1", kind="Secret", match_name="together-api-key", namespace="ml-team" + ) + } + ), + conditions=[ + fnv1.Condition( + type="EndpointReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="CredentialMissing", + message="Secret together-api-key has no key apiKey", + ), + ], ), ), Case( - name="a credential under a non-default key", + name="NonDefaultKey", + reason="A ModelEndpoint whose credential names a non-default key reports itself usable when the Secret holds that key.", req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr(credential=_api_key("together-api-key", key="TOGETHER_API_KEY")) + observed=fnv1.State(composite=_model_endpoint(credential_key="TOGETHER_API_KEY")), + required_resources={"credential": fnv1.Resources(items=[_credential_secret(key="TOGETHER_API_KEY")])}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_TRUE)), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "credential": fnv1.ResourceSelector( + api_version="v1", kind="Secret", match_name="together-api-key", namespace="ml-team" ) - ) + } ), - required_resources={ - "credential": fnv1.Resources( - items=[ - fnv1.Resource( - resource=resource.dict_to_struct( - _secret("together-api-key", {"TOGETHER_API_KEY": "sk-abc"}) - ) - ) - ] - ) - }, - ), - want=_response( - reason=fn.CONDITION_REASON_ENDPOINT_USABLE, - status=fnv1.STATUS_CONDITION_TRUE, - requirements=_credential_requirement("together-api-key"), + conditions=[ + fnv1.Condition(type="EndpointReady", status=fnv1.STATUS_CONDITION_TRUE, reason="EndpointUsable"), + ], ), ), Case( - name="an unresolved credential requirement", + name="CredentialUnresolved", + reason="Until Crossplane resolves its credential Secret, a ModelEndpoint requires it and reports WaitingForCredential.", req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) - ), + observed=fnv1.State(composite=_model_endpoint(credential_key="apiKey")), ), - want=_response( - reason=fn.CONDITION_REASON_WAITING_FOR_CREDENTIAL, - status=fnv1.STATUS_CONDITION_FALSE, - message="Waiting for Secret together-api-key to resolve", - requirements=_credential_requirement("together-api-key"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for Secret together-api-key to resolve") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "credential": fnv1.ResourceSelector( + api_version="v1", kind="Secret", match_name="together-api-key", namespace="ml-team" + ) + } + ), + conditions=[ + fnv1.Condition( + type="EndpointReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCredential", + message="Waiting for Secret together-api-key to resolve", + ), + ], ), ), ] -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - @pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction reports whether the endpoint's credential makes it usable.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-model-replica/tests/test_backends.py b/functions/compose-model-replica/tests/test_backends.py index c53c4a1a2..c69872422 100644 --- a/functions/compose-model-replica/tests/test_backends.py +++ b/functions/compose-model-replica/tests/test_backends.py @@ -12,1477 +12,6957 @@ # See the License for the specific language governing permissions and # limitations under the License. -"""Tests for compose-model-replica backends. - -A backend builds the workload (Deployment, LeaderWorkerSet, or PodCliqueSet) and the -ResourceClaimTemplates for one worker engine; the InferencePool, endpoint picker, -and HTTPRoute that front a replica's engines are built by routing.apply. Manifests are -asserted with a `Case` table: each case builds an engine's backend and compares -the composed manifests to a full `want`. Backend selection and serving are -dispatch/behaviour tests below the table. +"""Tests for compose-model-replica's backends and routing. + +A backend builds the workload (Deployment, LeaderWorkerSet, or PodCliqueSet) and +the ResourceClaimTemplates for one engine. routing.apply fronts a replica's +engines with an InferencePool, endpoint picker and HTTPRoute. + +A device request's CEL selector is as compose-model-deployment stamps it. """ +import copy import dataclasses -from typing import Any +import json import pytest -from crossplane.function import resource from function import routing from function.backends import base, grove, llmd, native -from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 from models.ai.modelplane.modelreplica import v1alpha1 from models.io.crossplane.m.kubernetes.object import v1alpha1 as k8sobjv1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -_SERVING = "modelplane.ai/serving" -_ENGINE = "modelplane.ai/engine" -_ROLE = "modelplane.ai/role" -_WORKLOAD = "modelplane.ai/workload" -_CLIQUE_ROLE = "modelplane.ai/clique-role" -_QUEUE_LABEL = "kai.scheduler/queue" -_QUEUE = "modelplane" -_SCHEDULER = "kai-scheduler" - -# A GPU device request (claim: DRA), as compose-model-deployment stamps it. -_GPU_CEL = 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' - - -def _gpu_request(count: int) -> v1alpha1.DeviceRequest: - return v1alpha1.DeviceRequest( - name="gpu", - deviceClassName="gpu.nvidia.com", - count=count, - selectors=[v1alpha1.Selector(cel=_GPU_CEL)], - ) +@dataclasses.dataclass +class BuildCase: + """A test case for a backend's build.""" -def _standalone_engine( - name: str = "main", - *, - copies: int = 1, - args: list[str] | None = None, - command: list[str] | None = None, - device_requests: list[v1alpha1.DeviceRequest] | None = None, -) -> v1alpha1.Engine: - """A single Standalone-member engine.""" - container = v1alpha1.Container( - name="engine", - image="vllm/vllm-openai:latest", - args=args if args is not None else ["--model=Qwen/Qwen3-0.6B"], - ) - if command is not None: - container.command = command - return v1alpha1.Engine( - name=name, - copies=copies, - members=[ - v1alpha1.Member( - role="Standalone", - nodePoolName="frontier", - deviceRequests=device_requests if device_requests is not None else [_gpu_request(1)], - template=v1alpha1.Template(spec=v1alpha1.Spec(containers=[container])), - ), - ], - ) + name: str + reason: str + backend: base.Backend + replica: v1alpha1.ModelReplica + provider_config: str + serving_label: str + stack: str + want: dict[str, dict] -def _gang_engine( - name: str = "main", - *, - copies: int = 1, - nodes: int = 1, - leader_args: list[str] | None = None, - leader_command: list[str] | None = None, - worker_args: list[str] | None = None, - worker_command: list[str] | None = None, - leader_device_requests: list[v1alpha1.DeviceRequest] | None = None, - leader_pool: str = "frontier", -) -> v1alpha1.Engine: - """A Leader + Worker engine. - - The members carry their own pool pins and device requests, defaulting to a - homogeneous gang on one pool. leader_device_requests=[] makes the leader - claimless (a coordinator-only leader); leader_pool moves it to another - pool. - """ +@dataclasses.dataclass +class SelectBackendCase: + """A test case for base.select_backend.""" - def member( - role: str, - nodes: int | None, - args: list[str] | None, - command: list[str] | None, - device_requests: list[v1alpha1.DeviceRequest], - pool: str, - ) -> v1alpha1.Member: - container = v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - if args is not None: - container.args = args - if command is not None: - container.command = command - kwargs: dict[str, Any] = { - "role": role, - "nodePoolName": pool, - "template": v1alpha1.Template(spec=v1alpha1.Spec(containers=[container])), - } - if device_requests: - kwargs["deviceRequests"] = device_requests - if nodes is not None: - kwargs["worker"] = v1alpha1.Worker(nodes=nodes) - return v1alpha1.Member(**kwargs) - - leader_requests = leader_device_requests if leader_device_requests is not None else [_gpu_request(8)] - return v1alpha1.Engine( - name=name, - copies=copies, - members=[ - member("Leader", None, leader_args, leader_command, leader_requests, leader_pool), - member("Worker", nodes, worker_args, worker_command, [_gpu_request(8)], "frontier"), - ], - ) + name: str + reason: str + engine: v1alpha1.Engine + stack: str + want: str -def _replica( - name: str = "r", *, namespace: str = "ml-team", engines: list[v1alpha1.Engine] | None = None -) -> v1alpha1.ModelReplica: - if engines is None: - engines = [_standalone_engine()] - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name=name, namespace=namespace), - spec=v1alpha1.SpecModel(clusterName="cluster-a", engines=engines), - ) +@dataclasses.dataclass +class CacheMountsCase: + """A test case for base.cache_mounts.""" + name: str + reason: str + replica: v1alpha1.ModelReplica + want: tuple[list[dict], list[dict]] -# The composed workload name for the default replica "r" / engine "main": -# engine-qualified so a multi-engine replica's workloads don't collide. -_WORKLOAD_NAME = resource.child_name("r", "main") -# The Grove PodCliqueSet name for the same replica/engine, budgeted tighter -# than _WORKLOAD_NAME per base.grove_pcs_name. -_GROVE_PCS_NAME = base.grove_pcs_name(_replica(), _gang_engine()) +@dataclasses.dataclass +class CacheEnvCase: + """A test case for base.cache_env.""" + name: str + reason: str + replica: v1alpha1.ModelReplica + want: list[dict] -def _claim_template(count: int, *, replica: str = "r", engine: str = "main", role: str = "standalone") -> dict: - """The ResourceClaimTemplate manifest a member's device requests produce.""" - return { - "apiVersion": "resource.k8s.io/v1", - "kind": "ResourceClaimTemplate", - "metadata": {"name": resource.child_name(replica, engine, role, "devices"), "namespace": "mp-ml-team-51733"}, - "spec": { - "spec": { - "devices": { - "requests": [ - { - "name": "gpu", - "exactly": { - "deviceClassName": "gpu.nvidia.com", - "count": count, - "selectors": [{"cel": {"expression": _GPU_CEL}}], - }, - } - ] - } - } - }, - } +@dataclasses.dataclass +class ApplyCase: + """A test case for routing.apply.""" -_CLUSTER = icv1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta(name="cluster-a"), - spec=icv1alpha1.Spec( - cluster=icv1alpha1.Cluster( - source="Existing", existing=icv1alpha1.Existing(secretRef=icv1alpha1.SecretRef(name="k")) - ) - ), - status=icv1alpha1.Status(providerConfigRef=icv1alpha1.ProviderConfigRef(name="cluster-a-pc")), -) + name: str + reason: str + composed: dict[str, dict] + replica: v1alpha1.ModelReplica + provider_config: str + want: dict[str, dict] -_PC = "cluster-a-pc" +@dataclasses.dataclass +class KvBlockSizeCase: + """A test case for routing._kv_block_size.""" -_NATIVE_WANT = { - "model-serving-main": { - "apiVersion": "apps/v1", - "kind": "Deployment", - "metadata": {"name": _WORKLOAD_NAME, "namespace": "mp-ml-team-51733"}, - "spec": { - "replicas": 1, - "selector": {"matchLabels": {_WORKLOAD: _WORKLOAD_NAME}}, - "template": { - "metadata": { - "labels": {_ENGINE: "main", _ROLE: "Standalone", _SERVING: "r", _WORKLOAD: _WORKLOAD_NAME} - }, - "spec": { - "containers": [ - { - "name": "engine", - "image": "vllm/vllm-openai:latest", - "args": ["--model=Qwen/Qwen3-0.6B"], - "ports": [{"name": "http", "containerPort": 8000}], - "resources": {"claims": [{"name": "devices"}]}, - "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], - "readinessProbe": { - "httpGet": {"path": "/health", "port": 8000}, - "initialDelaySeconds": 30, - "periodSeconds": 10, - "timeoutSeconds": 5, - }, - } - ], - "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], - "nodeSelector": {"modelplane.ai/pool": "frontier"}, - "resourceClaims": [ - { - "name": "devices", - "resourceClaimTemplateName": resource.child_name("r", "main", "standalone", "devices"), - } - ], - "tolerations": [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}], - }, - }, - }, - }, - "resource-claim-main-standalone": _claim_template(1), -} + name: str + reason: str + engine_args: list[str] + want: int -def _claims(role: str) -> list[dict]: - """The pod-level claim referencing a member's ResourceClaimTemplate.""" - return [ - { - "name": "devices", - "resourceClaimTemplateName": resource.child_name("r", "main", role, "devices"), - } - ] +@dataclasses.dataclass +class DisaggregatedConfigCase: + """A test case for routing._disaggregated_epp_config_yaml.""" + name: str + reason: str + block_size: int + want: str -def _clique(manifest: dict, name: str) -> dict: - """The named clique from a PodCliqueSet manifest.""" - return next(c for c in manifest["spec"]["template"]["cliques"] if c["name"] == name) +@dataclasses.dataclass +class RemoteNamespaceCase: + """A test case for base.remote_namespace.""" -def _pcs(leader_container: dict, worker_container: dict, *, worker_replicas: int = 1, copies: int = 1) -> dict: - node_selector = {"modelplane.ai/pool": "frontier"} - tolerations = [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}] + name: str + reason: str + replica: v1alpha1.ModelReplica + want: str - def pod_spec(container: dict, role: str) -> dict: - return { - "containers": [container], - "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], - "schedulerName": _SCHEDULER, - "nodeSelector": node_selector, - "resourceClaims": _claims(role), - "tolerations": tolerations, - } +def _route() -> dict: + """The Object composing the replica's HTTPRoute to its InferencePool.""" return { - "apiVersion": "grove.io/v1alpha1", - "kind": "PodCliqueSet", - "metadata": {"name": _GROVE_PCS_NAME, "namespace": "mp-ml-team-51733"}, "spec": { - "replicas": 1, - "template": { - "cliqueStartupType": "CliqueStartupTypeExplicit", - "terminationDelay": "4h", - "headlessServiceConfig": {"publishNotReadyAddresses": True}, - "cliques": [ - { - "name": "leader", - "labels": { - _ENGINE: "main", - _ROLE: "Leader", - _SERVING: "r", - _QUEUE_LABEL: _QUEUE, - _CLIQUE_ROLE: "leader", - }, - "spec": { - "roleName": "leader", - "replicas": 1, - "minAvailable": 1, - "podSpec": pod_spec(leader_container, "leader"), - }, - }, - { - "name": "worker", - "labels": {_ENGINE: "main", _ROLE: "Worker", _QUEUE_LABEL: _QUEUE}, - "spec": { - "roleName": "worker", - "replicas": worker_replicas, - "minAvailable": worker_replicas, - "podSpec": pod_spec(worker_container, "worker"), - }, + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "HTTPRoute", + "metadata": {"name": "r", "namespace": "mp-ml-team-51733"}, + "spec": { + "parentRefs": [{"name": "cluster-gateway", "namespace": "modelplane-system"}], + "rules": [ + { + "matches": [{"path": {"type": "PathPrefix", "value": "/ml-team/r/"}}], + # No request timeout, so long token streams + # aren't severed. + "timeouts": {"request": "0s"}, + "filters": [ + { + "type": "URLRewrite", + "urlRewrite": { + "path": {"type": "ReplacePrefixMatch", "replacePrefixMatch": "/"} + }, + } + ], + "backendRefs": [ + { + "group": "inference.networking.k8s.io", + "kind": "InferencePool", + "name": "r-pool", + } + ], + } + ], }, - ], - "podCliqueScalingGroups": [ - { - "name": "gang", - "cliqueNames": ["leader", "worker"], - "replicas": copies, - # 1 regardless of copies, so a wedged gang doesn't take - # the healthy ones down with it. - "minAvailable": 1, - } - ], + } }, - }, + } } -def _engine( - *, serving: bool, args: list[str] | None = None, command: list[str] | None = None, env: list[dict] | None = None -) -> dict[str, Any]: - c: dict[str, Any] = { - "name": "engine", - "image": "vllm/vllm-openai:latest", - "resources": {"claims": [{"name": "devices"}]}, - "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], - } - if command is not None: - c["command"] = command - if args is not None: - c["args"] = args - # A container carries an env only when the test gives it one. The Grove - # backend always injects a leader-address alias (see grove.py); callers - # composing a Grove _GROVE_WANT container pass it explicitly. - if env is not None: - c["env"] = env - if serving: - c["ports"] = [{"name": "http", "containerPort": 8000}] - c["readinessProbe"] = { - "httpGet": {"path": "/health", "port": 8000}, - "initialDelaySeconds": 30, - "periodSeconds": 10, - "timeoutSeconds": 5, +def _inference_pool() -> dict: + """The Object composing the InferencePool that fronts the replica's serving pods.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "inference.networking.k8s.io/v1", + "kind": "InferencePool", + "metadata": {"name": "r-pool", "namespace": "mp-ml-team-51733"}, + "spec": { + "selector": {"matchLabels": {"modelplane.ai/serving": "r"}}, + "targetPorts": [{"number": 8000}], + "endpointPickerRef": { + "name": "r-epp", + "port": {"number": 9002}, + "failureMode": "FailOpen", + }, + }, + } + }, } - return c - - -# A multi-node engine with verbatim leader/worker commands - no flag injection, -# no bootstrap. The follower addresses the leader through -# $(MODELPLANE_LEADER_ADDRESS). -_LEADER_CMD = [ - "/bin/sh", - "-c", - "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B " - "--tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", -] -_WORKER_CMD = ["/bin/sh", "-c", "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block"] -_GROVE_WANT = { - "model-serving-main": _pcs( - _engine(serving=True, command=_LEADER_CMD, env=[base.grove_leader_address_env()]), - _engine(serving=False, command=_WORKER_CMD, env=[base.grove_leader_address_env()]), - ), - "resource-claim-main-leader": _claim_template(8, role="leader"), - "resource-claim-main-worker": _claim_template(8, role="worker"), -} - - -@dataclasses.dataclass -class Case: - name: str - backend: base.Backend - engine: v1alpha1.Engine - want: dict - stack: str = "Standard" - + } -MANIFESTS_CASES = [ - Case( - name="native Standalone engine composes a Deployment", - backend=native.NativeBackend(), - engine=_standalone_engine(), - want=_NATIVE_WANT, - ), - Case( - name="Grove Leader/Worker engine composes a PodCliqueSet, commands verbatim", - backend=grove.GroveBackend(), - engine=_gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD), - want=_GROVE_WANT, - stack="Dynamo", - ), -] +def _epp(*, labels: dict, config_checksum: str) -> dict: + """The Object composing the endpoint picker's Deployment, its pod labeled and rolled by its config's checksum. -@pytest.mark.parametrize("case", MANIFESTS_CASES, ids=lambda case: case.name) -def test_manifests(case: Case) -> None: - """A backend composes an engine's manifests.""" - replica = _replica(engines=[case.engine]) - out = case.backend.build(replica, case.engine, _PC, base.serving_label(replica), case.stack) - got = {key: obj.spec.forProvider.manifest for key, obj in out.items()} - assert got == case.want - - -def test_leader_address_env_injected_but_not_rank() -> None: - """The Grove backend injects a leader address alias, but no rank.""" - # The Grove backend injects MODELPLANE_LEADER_ADDRESS (aliasing Grove's - # own GROVE_PCSG_* vars) but not MODELPLANE_RANK: Grove exposes no - # group-wide pod index yet (grove#755, open), so a gang engine's - # command computes its own rank from GROVE_PCLQ_POD_INDEX directly - # (see grove.py and the multinode example). - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - # Spelled out rather than compared against grove_leader_address_env(), - # which would pass whatever that function returned. The PCSG vars are - # what make the address vary per gang; the PCS-scoped ones are - # identical across gangs and would silently point every copy at gang - # 0's leader. - want = { - "name": "MODELPLANE_LEADER_ADDRESS", - "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", - } - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - assert container["env"] == [want] - - -def test_user_env_passed_through() -> None: - """A Grove member's own env follows the leader address alias.""" - # A member's own env passes through verbatim, after the leader-address - # alias (see test_leader_address_env_injected_but_not_rank). - engine = _gang_engine( - leader_command=_LEADER_CMD, - worker_command=_WORKER_CMD, - ) - spec = engine.members[0].template.spec - assert spec is not None - spec.containers[0].env = [v1alpha1.EnvItem(name="HF_TOKEN", value="x")] - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader")["spec"]["podSpec"] - env = leader["containers"][0]["env"] - assert env == [base.grove_leader_address_env(), {"name": "HF_TOKEN", "value": "x"}] - - -def test_fieldref_env_passes_through() -> None: - """A Grove member's pod-field env survives into the composed manifest.""" - # A pod-field env (e.g. VLLM_HOST_IP from status.podIP, which multi-NIC - # RDMA nodes need so the engine binds the right interface — #141) survives - # model_dump into the composed manifest. - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - spec = engine.members[0].template.spec - assert spec is not None - spec.containers[0].env = [ - v1alpha1.EnvItem( - name="VLLM_HOST_IP", - valueFrom=v1alpha1.ValueFrom(fieldRef=v1alpha1.FieldRef(fieldPath="status.podIP")), - ) - ] - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader")["spec"]["podSpec"] - env = leader["containers"][0]["env"] - assert env == [ - base.grove_leader_address_env(), - {"name": "VLLM_HOST_IP", "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}}, - ] - - -def test_member_metadata_propagates_to_native_pod_template() -> None: - """A Standalone member's template metadata lands on the Deployment's pod template.""" - # A Standalone member's template.metadata labels and annotations land - # on the Deployment's pod template, merged with the managed labels - # (#378). - engine = _standalone_engine() - engine.members[0].template.metadata = v1alpha1.Metadata( - labels={"example.com/role": "standalone"}, - annotations={"example.com/config": "standalone"}, - ) - replica = _replica(engines=[engine]) - out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - meta = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["metadata"] - assert meta["labels"] == { - "example.com/role": "standalone", - _ENGINE: "main", - _ROLE: "Standalone", - _SERVING: "r", - _WORKLOAD: _WORKLOAD_NAME, + The picker serves its metrics on a port named http, which the collector's + engine job scrapes, and with --metrics-endpoint-auth=false. By default it + authenticates callers by TokenReview, which needs a ClusterRole the picker's + namespaced ServiceAccount can't hold, so every scrape would be rejected. + --secure-serving stays on: it's the ext-proc gRPC server Envoy calls. + """ + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": "r-epp", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": 1, + "selector": {"matchLabels": {"app": "r-epp"}}, + "template": { + "metadata": { + "labels": labels, + "annotations": {"modelplane.ai/epp-config-checksum": config_checksum}, + }, + "spec": { + "serviceAccountName": "r-epp", + "containers": [ + { + "name": "epp", + "image": "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0", + "args": [ + "--pool-name=r-pool", + "--pool-namespace=mp-ml-team-51733", + "--pool-group=inference.networking.k8s.io", + "--config-file=/config/epp-config.yaml", + "--grpc-port=9002", + "--metrics-port=9090", + "--metrics-endpoint-auth=false", + ], + "ports": [ + {"name": "grpc", "containerPort": 9002}, + {"name": "grpc-health", "containerPort": 9003}, + {"name": "http", "containerPort": 9090}, + ], + "volumeMounts": [{"name": "config", "mountPath": "/config"}], + } + ], + "volumes": [{"name": "config", "configMap": {"name": "r-epp"}}], + }, + }, + }, + } + }, + } } - assert meta["annotations"] == {"example.com/config": "standalone"} -def test_member_metadata_propagates_to_cliques_independently() -> None: - """Each Grove member's template metadata lands on its own clique only.""" - # Leader metadata lands on the leader clique and worker metadata on the - # worker clique; neither leaks into the other. Grove propagates a - # clique's labels and annotations to its pods (#378). - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - engine.members[0].template.metadata = v1alpha1.Metadata( - labels={"example.com/role": "leader"}, annotations={"example.com/config": "leader"} - ) - engine.members[1].template.metadata = v1alpha1.Metadata( - labels={"example.com/role": "worker"}, annotations={"example.com/config": "worker"} - ) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader") - assert leader["labels"] == { - "example.com/role": "leader", - _ENGINE: "main", - _ROLE: "Leader", - _SERVING: "r", - _QUEUE_LABEL: _QUEUE, - _CLIQUE_ROLE: "leader", +def _epp_config(*, config: str) -> dict: + """The Object composing the ConfigMap that holds the endpoint picker's config.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "r-epp", "namespace": "mp-ml-team-51733"}, + "data": {"epp-config.yaml": config}, + } + }, + } } - assert leader["annotations"] == {"example.com/config": "leader"} - worker = _clique(manifest, "worker") - assert worker["labels"] == {"example.com/role": "worker", _ENGINE: "main", _ROLE: "Worker", _QUEUE_LABEL: _QUEUE} - assert worker["annotations"] == {"example.com/config": "worker"} - - -def test_worker_without_metadata_composes_only_managed_labels() -> None: - """A Grove worker with no template metadata carries only the managed labels.""" - # A worker member with no template.metadata composes a worker clique - # carrying only the managed queue label and no annotations key. - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - worker = _clique(manifest, "worker") - assert worker["labels"] == {_ENGINE: "main", _ROLE: "Worker", _QUEUE_LABEL: _QUEUE} - assert "annotations" not in worker - - -def _names(out: dict[str, k8sobjv1alpha1.Object]) -> set[str]: - """The names of the manifests a backend composed.""" - return {o.spec.forProvider.manifest["metadata"]["name"] for o in out.values()} - - -def test_co_located_replicas_get_distinct_names() -> None: - """Two replicas on one cluster compose distinct resource names.""" - # Two replicas of one deployment on the same cluster must produce - # distinct resource names on the remote cluster. - a = _replica("dep-clusterA") - b = _replica("dep-clusterB") - out_a = native.NativeBackend().build(a, a.spec.engines[0], _PC, base.serving_label(a), "Standard") - out_b = native.NativeBackend().build(b, b.spec.engines[0], _PC, base.serving_label(b), "Standard") - assert _names(out_a) & _names(out_b) == set() - - -def test_multi_engine_qualifies_workload_names() -> None: - """Each engine of a multi-engine replica composes distinctly named resources.""" - # A replica with two engines names each engine's workload distinctly so - # they don't collide on the remote cluster. - engines = [_standalone_engine("prefill"), _standalone_engine("decode")] - replica = _replica(engines=engines) - names = set() - for g in engines: - out = native.NativeBackend().build(replica, g, _PC, base.serving_label(replica), "Standard") - names |= _names(out) - assert len(names) == 4 # 2 deployments + 2 claim templates - - -@pytest.mark.parametrize( - ("backend", "engine", "stack", "want_cel"), - [ - pytest.param(native.NativeBackend(), _standalone_engine(), "Standard", base.AVAILABLE_CEL, id="native"), - pytest.param( - grove.GroveBackend(), - _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD), - "Dynamo", - base.GROVE_AVAILABLE_CEL, - id="grove", - ), - ], -) -def test_workload_readiness_policies(backend: base.Backend, engine: v1alpha1.Engine, stack: str, want_cel: str) -> None: - """A workload's readiness derives from its status, and a claim template's from its creation.""" - # A Deployment reports readiness from its Available condition; a - # PodCliqueSet publishes no such condition, so it's derived from its - # replica counters instead (base.GROVE_AVAILABLE_CEL). Either way the - # claim templates are ready on create. - replica = _replica(engines=[engine]) - out = backend.build(replica, engine, _PC, base.serving_label(replica), stack) - serving = out["model-serving-main"].spec.readiness - assert serving is not None - assert serving.policy == "DeriveFromCelQuery" - assert serving.celQuery == want_cel - for key, obj in out.items(): - if key.startswith("resource-claim"): - readiness = obj.spec.readiness - assert readiness is not None - assert readiness.policy == "SuccessfulCreate" - - -def test_multiple_device_requests_single_container_claim() -> None: - """Several device requests compose one container claim and one template carrying them all.""" - # resources.claims is a list-map keyed on name alone, so N device - # requests must NOT produce N container claims all named "devices". The - # container references the whole pod claim once; the template carries all - # requests. - engine = _standalone_engine( - device_requests=[ - v1alpha1.DeviceRequest(name="gpu", deviceClassName="gpu.nvidia.com", count=8), - v1alpha1.DeviceRequest(name="nic", deviceClassName="nic.nvidia.com", count=8), - ], - ) - replica = _replica(engines=[engine]) - out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - pod = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"] - claims = pod["containers"][0]["resources"]["claims"] - assert claims == [{"name": "devices"}] - assert pod["resourceClaims"][0]["name"] == "devices" - template = out["resource-claim-main-standalone"].spec.forProvider.manifest - template_requests = template["spec"]["spec"]["devices"]["requests"] - assert [r["name"] for r in template_requests] == ["gpu", "nic"] - claim_readiness = out["resource-claim-main-standalone"].spec.readiness - assert claim_readiness is not None - assert claim_readiness.policy == "SuccessfulCreate" - - -def test_claimless_leader_gets_no_claim() -> None: - """A Grove leader with no device requests composes no claim, but still pins and tolerates.""" - # A coordinator-only leader (e.g. a vLLM DP head running - # --data-parallel-size-local=0) carries no deviceRequests. Its pod must - # get no resourceClaims, its container no resources.claims, and no - # leader ResourceClaimTemplate must be composed - only the worker's. - # It still pins to its pool and tolerates the GPU taint. - engine = _gang_engine( - leader_command=_LEADER_CMD, - worker_command=_WORKER_CMD, - leader_device_requests=[], - ) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - - assert "resource-claim-main-leader" not in out - assert "resource-claim-main-worker" in out - - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader")["spec"]["podSpec"] - assert "resourceClaims" not in leader - assert "resources" not in leader["containers"][0] - assert leader["nodeSelector"] == {"modelplane.ai/pool": "frontier"} - assert leader["tolerations"] == [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}] - - worker = _clique(manifest, "worker")["spec"]["podSpec"] - assert worker["resourceClaims"] == _claims("worker") - assert worker["containers"][0]["resources"] == {"claims": [{"name": "devices"}]} - - -def test_members_pin_to_their_own_pools() -> None: - """Each Grove member's pods pin to that member's own pool.""" - # The scheduler may split a gang across pools when no single pool - # satisfies every member. Each member's pods must pin to that member's - # pool, not a shared engine-wide one. - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD, leader_pool="head") - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - assert _clique(manifest, "leader")["spec"]["podSpec"]["nodeSelector"] == {"modelplane.ai/pool": "head"} - assert _clique(manifest, "worker")["spec"]["podSpec"]["nodeSelector"] == {"modelplane.ai/pool": "frontier"} - - -# The LeaderWorkerSet backend for a Leader/Worker gang engine. - -_LWS_ROLE = "modelplane.ai/lws-role" - - -def _llmd_lws(engine: v1alpha1.Engine, replica: v1alpha1.ModelReplica) -> dict: - """The LeaderWorkerSet manifest the llm-d backend composes for engine.""" - out = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - return out["model-serving-main"].spec.forProvider.manifest - - -def test_llmd_leader_worker_set_shape() -> None: - """The llm-d backend composes a LeaderWorkerSet of copies gangs, each the leader plus its workers.""" - engine = _gang_engine(nodes=3, copies=2) - replica = _replica(engines=[engine]) - manifest = _llmd_lws(engine, replica) - assert manifest["apiVersion"] == "leaderworkerset.x-k8s.io/v1" - assert manifest["kind"] == "LeaderWorkerSet" - assert manifest["metadata"] == {"name": _WORKLOAD_NAME, "namespace": "mp-ml-team-51733"} - assert manifest["spec"]["replicas"] == 2 - # Gang size is the leader plus the worker's node count. - assert manifest["spec"]["leaderWorkerTemplate"]["size"] == 4 - - -def test_llmd_only_leader_carries_serving_label() -> None: - """Only the LeaderWorkerSet's leader carries the serving label.""" - engine = _gang_engine() - replica = _replica(engines=[engine]) - lwt = _llmd_lws(engine, replica)["spec"]["leaderWorkerTemplate"] - leader_labels = lwt["leaderTemplate"]["metadata"]["labels"] - assert leader_labels[_SERVING] == "r" - assert leader_labels[_LWS_ROLE] == "leader" - # The worker followers never serve, so they carry no serving label and - # the replica's Service can't route to them. They do carry the - # telemetry identity: a worker holds GPUs, and its metrics are the - # deployment's. - worker_labels = lwt["workerTemplate"]["metadata"]["labels"] - assert _SERVING not in worker_labels - assert worker_labels == {_ENGINE: "main", _ROLE: "Worker"} - - -def test_llmd_leader_address_and_rank_env_injected() -> None: - """Every LeaderWorkerSet container leads with the leader address and rank aliases.""" - # Every gang container leads with the backend-neutral coordination vars - # aliasing LWS_LEADER_ADDRESS / LWS_WORKER_INDEX. - engine = _gang_engine() - replica = _replica(engines=[engine]) - lwt = _llmd_lws(engine, replica)["spec"]["leaderWorkerTemplate"] - for tmpl in (lwt["leaderTemplate"], lwt["workerTemplate"]): - env = tmpl["spec"]["containers"][0]["env"] - assert env[0] == {"name": "MODELPLANE_LEADER_ADDRESS", "value": "$(LWS_LEADER_ADDRESS)"} - assert env[1] == {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"} - - -def test_llmd_no_modelexpress_env_even_with_a_cache() -> None: - """The llm-d backend injects no ModelExpress env, even for a replica with a cache.""" - # The llm-d (Standard) backend never injects ModelExpress env: that P2P - # wiring is the Grove (Dynamo) backend's, gated on the cluster stack. - engine = _gang_engine() - replica = v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="c"), - engines=[engine], - ), - ) - lwt = _llmd_lws(engine, replica)["spec"]["leaderWorkerTemplate"] - for tmpl in (lwt["leaderTemplate"], lwt["workerTemplate"]): - container = tmpl["spec"]["containers"][0] - env_names = [e["name"] for e in container["env"]] - # HF_HUB_CACHE is the cache's own env (every stack); the MX bundle - # is not. - assert env_names == ["MODELPLANE_LEADER_ADDRESS", "MODELPLANE_RANK", "HF_HUB_CACHE"] - assert "MX_SERVER_ADDRESS" not in env_names - assert "securityContext" not in container - - -def test_llmd_workload_readiness_uses_available_cel() -> None: - """The LeaderWorkerSet's readiness derives from its Available condition.""" - engine = _gang_engine() - replica = _replica(engines=[engine]) - out = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - readiness = out["model-serving-main"].spec.readiness - assert readiness is not None - assert readiness.policy == "DeriveFromCelQuery" - assert readiness.celQuery == base.AVAILABLE_CEL - - -def test_select_backend_standalone_engine_is_native() -> None: - """A Standalone engine selects the native backend.""" - # A Standalone engine is native regardless of the cluster's stack. - assert base.select_backend(_standalone_engine(), "Standard") == base.NATIVE - assert base.select_backend(_standalone_engine(), "Dynamo") == base.NATIVE - - -def test_select_backend_leader_worker_engine_is_llmd() -> None: - """A Leader/Worker engine on a Standard cluster selects the llm-d backend.""" - assert base.select_backend(_gang_engine(), "Standard") == base.LLMD -def test_select_backend_leader_worker_engine_is_grove() -> None: - """A Leader/Worker engine on a Dynamo cluster selects the Grove backend.""" - assert base.select_backend(_gang_engine(), "Dynamo") == base.GROVE +def _epp_role() -> dict: + """The Object composing the endpoint picker's Role.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "Role", + "metadata": {"name": "r-epp", "namespace": "mp-ml-team-51733"}, + "rules": [ + {"apiGroups": [""], "resources": ["pods"], "verbs": ["get", "watch", "list"]}, + { + "apiGroups": ["inference.networking.k8s.io"], + "resources": ["inferencepools"], + "verbs": ["get", "watch", "list"], + }, + { + "apiGroups": ["inference.networking.x-k8s.io"], + "resources": ["inferenceobjectives"], + "verbs": ["get", "watch", "list"], + }, + ], + } + }, + } + } -def _cache_replica( - *, cache: str | None = None, args: list[str] | None = None, command: list[str] | None = None -) -> v1alpha1.ModelReplica: - """A replica with one Standalone engine, referencing cache if one's given.""" - engine = _standalone_engine(args=args or [], command=command) - modelcache = v1alpha1.ModelCacheRef(name=cache) if cache else None - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(namespace="ml-team"), - spec=v1alpha1.SpecModel(clusterName="c", modelCacheRef=modelcache, engines=[engine]), - ) +def _epp_role_binding() -> dict: + """The Object composing the endpoint picker's RoleBinding.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "RoleBinding", + "metadata": {"name": "r-epp", "namespace": "mp-ml-team-51733"}, + "subjects": [{"kind": "ServiceAccount", "name": "r-epp", "namespace": "mp-ml-team-51733"}], + "roleRef": {"apiGroup": "rbac.authorization.k8s.io", "kind": "Role", "name": "r-epp"}, + } + }, + } + } -def test_no_cache_no_mounts() -> None: - """A replica with no cache mounts nothing.""" - volumes, mounts = base.cache_mounts(_cache_replica()) - assert (volumes, mounts) == ([], []) +def _epp_service_account() -> dict: + """The Object composing the endpoint picker's ServiceAccount.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ServiceAccount", + "metadata": {"name": "r-epp", "namespace": "mp-ml-team-51733"}, + } + }, + } + } -def test_cache_adds_volume_and_mount() -> None: - """A replica with a cache mounts the cache's PVC.""" - volumes, mounts = base.cache_mounts(_cache_replica(cache="qwen")) - assert volumes == [{"name": "model-cache", "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}}] - assert mounts == [{"name": "model-cache", "mountPath": "/mnt/models"}] +def _epp_service() -> dict: + """The Object composing the endpoint picker's Service.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Service", + "metadata": {"name": "r-epp", "namespace": "mp-ml-team-51733"}, + "spec": { + "selector": {"app": "r-epp"}, + "ports": [{"name": "grpc-ext-proc", "port": 9002, "targetPort": 9002, "appProtocol": "http2"}], + }, + } + }, + } + } -def test_cache_env_points_huggingface_at_the_mount() -> None: - """A replica with a cache points HF_HUB_CACHE at the mount.""" - # The cache is staged in HuggingFace's cache layout, so pointing - # HF_HUB_CACHE at the mount is what lets an engine's own --model= - # resolve against it instead of pulling from HuggingFace (#407). - assert base.cache_env(_cache_replica(cache="qwen")) == [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}] +def _deployment( + *, + name: str, + claim_template_name: str, + replicas: int, + pod_metadata: dict, + containers: list[dict], + volumes: list[dict], +) -> dict: + """The Object composing a Standalone engine's Deployment.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": name, "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": replicas, + "selector": {"matchLabels": {"modelplane.ai/workload": name}}, + "template": { + "metadata": pod_metadata, + "spec": { + "containers": containers, + "volumes": volumes, + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": claim_template_name, + } + ], + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ], + }, + }, + }, + } + }, + } + } -def test_cache_env_empty_without_cache() -> None: - """A replica with no cache gets no cache env.""" - assert base.cache_env(_cache_replica()) == [] +def _leader_worker_set( + *, + replicas: int, + size: int, + leader_containers: list[dict], + leader_volumes: list[dict], + worker_containers: list[dict], + worker_volumes: list[dict], +) -> dict: + """The Object composing the main engine's LeaderWorkerSet.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "leaderworkerset.x-k8s.io/v1", + "kind": "LeaderWorkerSet", + "metadata": {"name": "r-main-bb4e3", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": replicas, + "leaderWorkerTemplate": { + "size": size, + "leaderTemplate": { + "metadata": { + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Leader", + "modelplane.ai/serving": "r", + "modelplane.ai/lws-role": "leader", + } + }, + "spec": { + "containers": leader_containers, + "volumes": leader_volumes, + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ], + }, + }, + "workerTemplate": { + "metadata": { + "labels": {"modelplane.ai/engine": "main", "modelplane.ai/role": "Worker"} + }, + "spec": { + "containers": worker_containers, + "volumes": worker_volumes, + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ], + }, + }, + }, + }, + } + }, + } + } -def test_cache_env_sets_no_offline_flag() -> None: - """A replica with a cache doesn't set HF_HUB_OFFLINE.""" - # HF_HUB_OFFLINE would break an engine that fetches a *different* repo - # at startup (kimi-k2's separately-gated tokenizer), and resolution - # doesn't need it. - names = {e["name"] for e in base.cache_env(_cache_replica(cache="qwen"))} - assert "HF_HUB_OFFLINE" not in names +def _pod_clique_set(*, name: str, cliques: list[dict]) -> dict: + """The Object composing a Leader/Worker engine's PodCliqueSet, from its leader and worker cliques.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status) && has(object.status.observedGeneration) && object.status.observedGeneration == object.metadata.generation && object.spec.replicas > 0 && has(object.status.availableReplicas) && object.status.availableReplicas >= object.spec.replicas", + }, + "forProvider": { + "manifest": { + "apiVersion": "grove.io/v1alpha1", + "kind": "PodCliqueSet", + "metadata": {"name": name, "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": 1, + "template": { + "cliqueStartupType": "CliqueStartupTypeExplicit", + "terminationDelay": "4h", + "headlessServiceConfig": {"publishNotReadyAddresses": True}, + "cliques": cliques, + "podCliqueScalingGroups": [ + { + "name": "gang", + "cliqueNames": ["leader", "worker"], + "replicas": 1, + # 1 whatever the copies, so a wedged gang + # doesn't take the healthy ones down with it. + "minAvailable": 1, + } + ], + }, + }, + } + }, + } + } -def _native_cache_replica() -> v1alpha1.ModelReplica: - """A replica with one Standalone engine, referencing the qwen cache.""" - engine = _standalone_engine(args=[]) - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), - engines=[engine], - ), - ) +def _claim_template(*, name: str, count: int) -> dict: + """The Object composing a member's ResourceClaimTemplate, for count GPUs of at least 80Gi.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": name, "namespace": "mp-ml-team-51733"}, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": count, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ] + } + } + }, + } + }, + } + } -def test_native_cache_mounts_pvc_and_sets_cache_env() -> None: - """The native backend mounts a cache and points HF_HUB_CACHE at it, injecting no --model.""" - # A cache contributes a volume, a mount, and the HF_HUB_CACHE that makes - # the engine's own --model= resolve against it. Modelplane injects - # no --model of its own: naming the model is the command's job. - replica = _native_cache_replica() - out = native.NativeBackend().build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Standard") - dep = out["model-serving-main"].spec.forProvider.manifest - pod = dep["spec"]["template"]["spec"] - vol_names = {v["name"] for v in pod["volumes"]} - assert "model-cache" in vol_names - container = pod["containers"][0] - assert {"name": "model-cache", "mountPath": "/mnt/models"} in container["volumeMounts"] - assert {"name": "HF_HUB_CACHE", "value": "/mnt/models"} in container["env"] - assert container["args"] == [] - - -def test_native_cache_user_env_comes_after_cache_env() -> None: - """A Standalone member's own env follows the cache env.""" - # Kubernetes expands $(VAR) left to right, so Modelplane's own entries - # must precede the user's for a user entry to reference them. - replica = _native_cache_replica() - engine = replica.spec.engines[0] - spec = engine.members[0].template.spec - assert spec is not None - spec.containers[0].env = [v1alpha1.EnvItem(name="HF_TOKEN", value="x")] - out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] - assert container["env"] == [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}, {"name": "HF_TOKEN", "value": "x"}] - - -def _grove_cache_replica( +def _replica( *, - leader_command: list[str] | None = None, - worker_command: list[str] | None = None, - leader_args: list[str] | None = None, - worker_args: list[str] | None = None, + name: str | None, + namespace: str, + cluster_name: str, + labels: dict[str, str] | None, + model_cache_ref: v1alpha1.ModelCacheRef | None, + serving: v1alpha1.Serving | None, + engines: list[v1alpha1.Engine], ) -> v1alpha1.ModelReplica: - """A replica with one Leader/Worker engine, referencing the kimi cache.""" - engine = _gang_engine( - leader_command=leader_command, - worker_command=worker_command, - leader_args=leader_args, - worker_args=worker_args, - ) + """A ModelReplica.""" return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), + metadata=metav1.ObjectMeta(name=name, namespace=namespace, labels=labels), spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="kimi"), - engines=[engine], + clusterName=cluster_name, + modelCacheRef=model_cache_ref, + serving=serving, + engines=engines, ), ) -def test_grove_cache_both_cliques_mount_cache() -> None: - """The Grove backend mounts a cache on both cliques.""" - replica = _grove_cache_replica(leader_args=[], worker_command=["/bin/sh", "-c", "join"]) - manifest = ( - grove.GroveBackend() - .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] - .spec.forProvider.manifest - ) - for clique_name in ("leader", "worker"): - pod = _clique(manifest, clique_name)["spec"]["podSpec"] - assert "model-cache" in {v["name"] for v in pod["volumes"]} - assert {"name": "model-cache", "mountPath": "/mnt/models"} in pod["containers"][0]["volumeMounts"] - - -def test_grove_cache_sets_cache_env_on_every_clique_and_injects_no_model() -> None: - """The Grove backend points both cliques' HF_HUB_CACHE at a cache, injecting no --model.""" - # A cache gives both cliques HF_HUB_CACHE so their own --model= - # resolves against the mount; Modelplane adds no --model itself. - replica = _grove_cache_replica(leader_args=[], worker_command=["/bin/sh", "-c", "join"]) - manifest = ( - grove.GroveBackend() - .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] - .spec.forProvider.manifest - ) - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - assert {"name": "HF_HUB_CACHE", "value": "/mnt/models"} in container["env"] - assert "--model=/mnt/models" not in container.get("args", []) - - -def test_grove_cache_command_engine_mounts_cache_without_injecting_model() -> None: - """A Grove member with its own command mounts a cache and keeps its command verbatim.""" - # A member with its own command keeps it verbatim and gets no injected - # --model (it points at the cache with its own flag). - leader_cmd = [ - "/bin/sh", - "-c", - "python3 -m sglang.launch_server --model-path /mnt/models --tp 16", - ] - replica = _grove_cache_replica(leader_command=leader_cmd, worker_command=["/bin/sh", "-c", "join"]) - manifest = ( - grove.GroveBackend() - .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] - .spec.forProvider.manifest - ) - leader = _clique(manifest, "leader")["spec"]["podSpec"]["containers"][0] - assert {"name": "model-cache", "mountPath": "/mnt/models"} in leader["volumeMounts"] - assert leader["command"] == leader_cmd - - -# serving.mode: PrefillDecode routing layers an InferencePool + endpoint -# picker over two engines, role-labels them, and sidecars decode — no unified -# Service. Mirrors how fn.py composes engines then calls routing.apply. - - -def _disaggregated_apply() -> dict[str, k8sobjv1alpha1.Object]: - """Routing for a PrefillDecode replica of two native engines.""" - prefill = _standalone_engine(name="prefill") - prefill.phase = "Prefill" - decode = _standalone_engine(name="decode") - decode.phase = "Decode" - replica = _replica(engines=[prefill, decode]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = {} - for engine in replica.spec.engines: - composed.update(native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard")) - return routing.apply(composed, replica, _PC) - - -def _serving_pod(out: dict[str, k8sobjv1alpha1.Object], engine_name: str) -> dict: - """The pod template of an engine's Deployment.""" - return out[f"model-serving-{engine_name}"].spec.forProvider.manifest["spec"]["template"] - - -def test_disaggregated_replaces_unified_service_with_pool_and_epp() -> None: - """PrefillDecode routing fronts the engines with an InferencePool and endpoint picker.""" - out = _disaggregated_apply() - assert "inference-pool" in out - assert "epp" in out - assert "epp-config" in out - pool = out["inference-pool"].spec.forProvider.manifest - assert pool["kind"] == "InferencePool" - assert pool["spec"]["endpointPickerRef"]["name"] == "r-epp" - - -def test_disaggregated_the_picker_is_scrapeable_and_attributed() -> None: - """A built-in MetricMapping renames the picker's scheduling latency. - - Nothing can match it unless something scrapes the picker, and the - collector's engine job keeps a pod on two things: the deployment - label, and a container port named `http`. Without both, the mapping - is config that matches nothing for the life of the fleet. - - Its metrics endpoint authenticates callers by TokenReview by default, - which needs a ClusterRole the picker's namespaced ServiceAccount - cannot hold, so every scrape would be rejected. --secure-serving is - left alone: that one is the ext-proc gRPC server Envoy calls. - """ - replica = _replica() - replica.metadata = metav1.ObjectMeta( - name="r", - namespace="ml-team", - labels={base.LABEL_DEPLOYMENT: "qwen3-8b", "modelplane.ai/replica-index": "2"}, - ) - replica.spec.serving = v1alpha1.Serving(mode="Unified") - composed = {} - for engine in replica.spec.engines: - composed.update(native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard")) - template = routing.apply(composed, replica, _PC)["epp"].spec.forProvider.manifest["spec"]["template"] - - labels = template["metadata"]["labels"] - assert labels[base.LABEL_DEPLOYMENT] == "qwen3-8b" - assert labels[base.LABEL_REPLICA] == "2" - assert labels[base.LABEL_ROLE] == "picker" - # Still the Deployment's own selector label, which must not move. - assert labels["app"] == "r-epp" - - container = next(c for c in template["spec"]["containers"] if c["name"] == "epp") - assert {"name": base.ENGINE_PORT_NAME, "containerPort": 9090} in container["ports"] - assert "--metrics-port=9090" in container["args"] - assert "--metrics-endpoint-auth=false" in container["args"] - assert "--secure-serving=false" not in container["args"] - - -def test_disaggregated_injects_nixl_plumbing() -> None: - """Both PrefillDecode engines get the NIXL plumbing the schema can't express.""" - # The plumbing is a Memory /dev/shm and VLLM_NIXL_SIDE_CHANNEL_HOST = pod IP. - out = _disaggregated_apply() - for role in ("prefill", "decode"): - pod = _serving_pod(out, role)["spec"] - assert any(v.get("emptyDir", {}).get("medium") == "Memory" for v in pod["volumes"]), ( - f"{role} missing Memory /dev/shm volume" - ) - engine = next(c for c in pod["containers"] if c["name"] == "engine") - assert "/dev/shm" in [m["mountPath"] for m in engine["volumeMounts"]] - host = next((e for e in engine["env"] if e["name"] == "VLLM_NIXL_SIDE_CHANNEL_HOST"), None) - assert host is not None, f"{role} missing VLLM_NIXL_SIDE_CHANNEL_HOST" - assert host["valueFrom"]["fieldRef"]["fieldPath"] == "status.podIP" - assert "VLLM_NIXL_SIDE_CHANNEL_PORT" in [e["name"] for e in engine["env"]] - - -def test_disaggregated_epp_config_arms_the_pd_decider() -> None: - """PrefillDecode silently serves decode-only unless the PD decider is armed.""" - # Selective prefix-based-pd-decider needs all of: nonCachedTokens > 0 (0 = - # disabled), the approx-prefix-cache-producer plugin that populates the - # attribute it reads, and that producer pinned to autoTune: false (the - # true default never populates). And it must NOT carry the prepareDataPlugins - # feature gate, which the v0.8.0 EPP image rejects and crashloops on. - cfg = _disaggregated_apply()["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - assert "prefix-based-pd-decider" in cfg - assert "nonCachedTokens: 16" in cfg - assert "approx-prefix-cache-producer" in cfg - assert "autoTune: false" in cfg - assert "nonCachedTokens: 0" not in cfg - assert "prepareDataPlugins" not in cfg - - -def test_disaggregated_epp_and_sidecar_images_and_config_group_are_pinned() -> None: - """Lock the picker and sidecar images and the EndpointPickerConfig API group.""" - # Nothing else asserts these, so a wrong tag/registry path or a stale config - # group passes CI and only surfaces as an EPP/sidecar crashloop at deploy. - # These are deliberate literals, not routing._* constants: comparing to the - # constant would be tautological (it can't catch a typo in the constant), and - # a literal forces a bump to show up here and be reviewed. - out = _disaggregated_apply() - epp = out["epp"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"] - assert next(c["image"] for c in epp if c["name"] == "epp") == "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0" - sidecar = next(c for c in _serving_pod(out, "decode")["spec"]["containers"] if c["name"] == "pd-sidecar") - assert sidecar["image"] == "ghcr.io/llm-d/llm-d-router-disagg-sidecar:v0.9.0" - cfg = out["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - assert "apiVersion: llm-d.ai/v1alpha1" in cfg - - -def test_disaggregated_epp_role_watches_inferenceobjectives() -> None: - """The picker watches InferenceObjectives (GIE x-k8s.io group); the Role must allow it.""" - rules = _disaggregated_apply()["epp-role"].spec.forProvider.manifest["rules"] - assert any( - "inference.networking.x-k8s.io" in r["apiGroups"] and "inferenceobjectives" in r["resources"] for r in rules - ), f"EPP Role missing inferenceobjectives watch: {rules}" - - -def test_disaggregated_decode_port_follows_user_arg() -> None: - """The sidecar and the decode container port track the user's --port, not a hardcoded one.""" - prefill = _standalone_engine(name="prefill") - prefill.phase = "Prefill" - decode = _standalone_engine(name="decode", args=["--model=m", "--port=9000"]) - decode.phase = "Decode" - replica = _replica(engines=[prefill, decode]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = {} - for e in replica.spec.engines: - composed.update(native.NativeBackend().build(replica, e, _PC, base.serving_label(replica), "Standard")) - out = routing.apply(composed, replica, _PC) - containers = _serving_pod(out, "decode")["spec"]["containers"] - engine = next(c for c in containers if c["name"] == "engine") - sidecar = next(c for c in containers if c["name"] == "pd-sidecar") - assert engine["ports"][0]["containerPort"] == 9000 - assert "--vllm-port=9000" in sidecar["args"] - assert sidecar["ports"][0]["containerPort"] == 8000 - - -def test_disaggregated_engines_role_labeled() -> None: - """PrefillDecode routing labels each engine's pods with its role.""" - out = _disaggregated_apply() - assert _serving_pod(out, "prefill")["metadata"]["labels"]["llm-d.ai/role"] == "prefill" - decode_labels = _serving_pod(out, "decode")["metadata"]["labels"] - assert decode_labels["llm-d.ai/role"] == "decode" - assert decode_labels["app"] == "r" - - -def test_disaggregated_decode_gets_sidecar_and_moves_engine_port() -> None: - """The decode engine gets the pd-sidecar on the serving port, and moves to another.""" - out = _disaggregated_apply() - containers = _serving_pod(out, "decode")["spec"]["containers"] - names = [c["name"] for c in containers] - assert names == ["engine", "pd-sidecar"] - engine = next(c for c in containers if c["name"] == "engine") - assert engine["ports"][0]["containerPort"] == 8001 - assert engine["readinessProbe"]["timeoutSeconds"] == 5 - sidecar = next(c for c in containers if c["name"] == "pd-sidecar") - assert sidecar["ports"][0]["containerPort"] == 8000 - assert sidecar["readinessProbe"]["timeoutSeconds"] == 5 - assert "--secure-proxy=false" in sidecar["args"] - - -def test_disaggregated_a_decode_engine_keeps_a_scrapeable_port() -> None: - """The collector's engine job keeps a pod on the port named `http`. - - Moving the decode engine off 8000 for the sidecar drops the name with - it, and the sidecar takes the port unnamed because it serves inference - rather than /metrics. A decode pod with no named port anywhere is one - nothing scrapes, so a disaggregated deployment reports half its - engines and the shortfall looks like idle capacity. - """ - containers = _serving_pod(_disaggregated_apply(), "decode")["spec"]["containers"] - engine = next(c for c in containers if c["name"] == "engine") - assert engine["ports"] == [{"name": base.ENGINE_PORT_NAME, "containerPort": 8001}] - - -def test_disaggregated_prefill_has_no_sidecar() -> None: - """The prefill engine gets no sidecar.""" - containers = _serving_pod(_disaggregated_apply(), "prefill")["spec"]["containers"] - assert [c["name"] for c in containers] == ["engine"] - - -def test_disaggregated_route_targets_inference_pool() -> None: - """PrefillDecode routing points the HTTPRoute at the InferencePool, with no request timeout.""" - route = _disaggregated_apply()[base.ROUTE_KEY].spec.forProvider.manifest - rule = route["spec"]["rules"][0] - ref = rule["backendRefs"][0] - assert ref["kind"] == "InferencePool" - assert ref["name"] == "r-pool" - # Disable the request timeout so long token streams aren't severed. - assert rule["timeouts"]["request"] == "0s" - - -def test_disaggregated_selects_engines_by_phase_not_name() -> None: - """Roles come from each engine's phase, not its name.""" - decode = _standalone_engine(name="alpha") - decode.phase = "Decode" - prefill = _standalone_engine(name="beta") - prefill.phase = "Prefill" - replica = _replica(engines=[decode, prefill]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = {} - for e in replica.spec.engines: - composed.update(native.NativeBackend().build(replica, e, _PC, base.serving_label(replica), "Standard")) - out = routing.apply(composed, replica, _PC) - # alpha is Decode -> sidecar; beta is Prefill -> none, despite their names. - assert [c["name"] for c in _serving_pod(out, "alpha")["spec"]["containers"]] == ["engine", "pd-sidecar"] - assert [c["name"] for c in _serving_pod(out, "beta")["spec"]["containers"]] == ["engine"] - assert _serving_pod(out, "alpha")["metadata"]["labels"]["llm-d.ai/role"] == "decode" - assert _serving_pod(out, "beta")["metadata"]["labels"]["llm-d.ai/role"] == "prefill" - - -def test_disaggregated_decode_can_be_a_grove_gang() -> None: - """PrefillDecode routing decorates a Grove decode gang's leader clique, and leaves its worker alone.""" - # A PrefillDecode engine can itself be a Leader/Worker gang, so routing - # must decorate a Grove PodCliqueSet's leader clique - role label, serving - # label, pd-sidecar, NIXL plumbing - exactly like a Deployment's pod - # template. Exercises the _serving_pod_templates normalization that lets - # one routing layer decorate both workload shapes. - prefill = _standalone_engine(name="prefill") - prefill.phase = "Prefill" - decode = _gang_engine(name="decode", leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - decode.phase = "Decode" - replica = _replica(engines=[prefill, decode]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = { - **native.NativeBackend().build(replica, prefill, _PC, base.serving_label(replica), "Standard"), - **grove.GroveBackend().build(replica, decode, _PC, base.serving_label(replica), "Dynamo"), - } - out = routing.apply(composed, replica, _PC) - - manifest = out["model-serving-decode"].spec.forProvider.manifest - leader_clique = _clique(manifest, "leader") - assert leader_clique["labels"]["llm-d.ai/role"] == "decode" - assert leader_clique["labels"]["app"] == "r" - leader = leader_clique["spec"]["podSpec"] - assert [c["name"] for c in leader["containers"]] == ["engine", "pd-sidecar"] - assert any(v.get("emptyDir", {}).get("medium") == "Memory" for v in leader["volumes"]), ( - "leader clique missing Memory /dev/shm volume for NIXL" - ) - engine = next(c for c in leader["containers"] if c["name"] == "engine") - assert "VLLM_NIXL_SIDE_CHANNEL_HOST" in [e["name"] for e in engine["env"]] - - # The worker clique never serves; routing must not touch it at all - - # its labels stay exactly what the Grove backend composed (just the - # queue label), with no role or serving label added. - worker_clique = _clique(manifest, "worker") - worker = worker_clique["spec"]["podSpec"] - assert [c["name"] for c in worker["containers"]] == ["engine"] - assert worker_clique["labels"] == {_ENGINE: "decode", _ROLE: "Worker", _QUEUE_LABEL: _QUEUE} - - -# Unified serving (or no serving block) fronts the pods with an -# InferencePool + endpoint picker in place of a plain Service, so requests -# route by prefix cache and load rather than round-robin - one pod or many. -# Mirrors how fn.py composes engines then calls routing.apply. - - -def _unified_apply(copies: int = 1) -> dict[str, k8sobjv1alpha1.Object]: - """Routing for a Unified replica of one native engine.""" - engine = _standalone_engine(copies=copies) - replica = _replica(engines=[engine]) - composed = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - return routing.apply(composed, replica, _PC) - - -def test_unified_fronts_with_pool_and_epp() -> None: - """Unified routing fronts the engine with an InferencePool and endpoint picker.""" - out = _unified_apply() - assert "inference-pool" in out - assert "epp" in out - assert "epp-config" in out - pool = out["inference-pool"].spec.forProvider.manifest - assert pool["kind"] == "InferencePool" - assert pool["spec"]["endpointPickerRef"]["name"] == "r-epp" - - -@pytest.mark.parametrize("copies", [1, 2]) -def test_unified_single_pod_also_pools(copies: int) -> None: - """A single serving pod gets a pool too, as several do.""" - # A single serving pod has nothing to pick between, but still gets the - # pool. Always fronting with one avoids swapping a Service for a pool when a - # second pod appears - a swap that would drop in-flight requests. - out = _unified_apply(copies=copies) - assert "inference-pool" in out - assert "epp" in out - - -def test_unified_fronts_a_leader_worker_set() -> None: - """Unified routing fronts a LeaderWorkerSet.""" - # A Standard multi-node engine composes a LeaderWorkerSet, and unified - # routing must handle that shape too: it reads the engine args for the KV - # block size through _serving_pod_templates, which has to normalize a - # LeaderWorkerSet's leaderTemplate alongside a Deployment's pod template - # and a Grove PodCliqueSet's leader clique. Regression for a shape - # normalization that only knew Deployment and PodCliqueSet and raised - # KeyError on a LeaderWorkerSet. - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _replica(engines=[engine]) - composed = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - out = routing.apply(composed, replica, _PC) - assert "inference-pool" in out - assert out["model-serving-main"].spec.forProvider.manifest["kind"] == "LeaderWorkerSet" - - -def test_unified_pool_selects_pods_by_the_serving_label() -> None: - """The pool selects the pods by the serving label they already carry, so no relabeling is needed.""" - pool = _unified_apply()["inference-pool"].spec.forProvider.manifest - assert pool["spec"]["selector"]["matchLabels"] == {base.LABEL_SERVING: "r"} - - -def test_unified_route_targets_inference_pool() -> None: - """Unified routing points the HTTPRoute at the InferencePool.""" - route = _unified_apply()[base.ROUTE_KEY].spec.forProvider.manifest - ref = route["spec"]["rules"][0]["backendRefs"][0] - assert ref["kind"] == "InferencePool" - assert ref["name"] == "r-pool" - - -def test_unified_epp_config_is_unified_not_disaggregated() -> None: - """The unified picker scores by prefix cache and queue depth, with no prefill/decode split.""" - # It scores in a single profile, and still needs the - # approx-prefix-cache-producer that feeds the prefix-cache scorer. - cfg = _unified_apply()["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - assert "prefix-cache-scorer" in cfg - assert "queue-scorer" in cfg - assert "approx-prefix-cache-producer" in cfg - assert "prefill" not in cfg - assert "decider" not in cfg - - -def test_unified_epp_image_and_config_group_are_pinned() -> None: - """Lock the picker image and the EndpointPickerConfig API group for the unified path too.""" - # A deliberate literal (not routing._EPP_IMAGE) so a wrong - # tag/registry or a stale config group is caught in review, not as a - # deploy-time crashloop. Unified has no sidecar, so only the EPP is checked. - out = _unified_apply() - epp = out["epp"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"] - assert next(c["image"] for c in epp if c["name"] == "epp") == "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0" - cfg = out["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - assert "apiVersion: llm-d.ai/v1alpha1" in cfg - - -def test_unified_epp_pod_carries_config_checksum() -> None: - """The EPP pod template carries a sha256 of its config, so a config change rolls the pod.""" - # The EPP reads its config once at startup, so a config change must roll - # the pod. The pod template carries a sha256 of the rendered config to drive - # that rollout. - template = _unified_apply()["epp"].spec.forProvider.manifest["spec"]["template"] - checksum = template["metadata"]["annotations"]["modelplane.ai/epp-config-checksum"] - assert len(checksum) == 64 - - -# On a Dynamo cluster the native (Standalone) and Grove (Leader/Worker) -# backends inject the ModelExpress P2P env (MX_SERVER_ADDRESS/MODEL_EXPRESS_URL/ -# MX_MODEL_REVISION/MX_P2P_METADATA/POD_*) and the IPC_LOCK security context -# into every engine container of a replica that references a cache. The env is -# inert unless the engine command opts in with --load-format modelexpress. It's -# gated on the cluster's Dynamo stack: on Standard neither backend injects it -# (the portable engine command falls back), and the llm-d backend never does. -# -# HF_HUB_CACHE is deliberately NOT in this set: it's the cache's own env, on -# every stack (see base.cache_env), and ModelExpress reads it only as a -# fallback for its cache root. Keeping it out here is what makes these -# assertions fail if it ever leaks back into modelexpress_env as a duplicate. - -_MODELEXPRESS_ENV_NAMES = { - "MX_SERVER_ADDRESS", - "MODEL_EXPRESS_URL", - "MX_MODEL_REVISION", - "MX_P2P_METADATA", - "POD_NAME", - "POD_UID", - "POD_NAMESPACE", -} -# What a cache-referencing engine carries on Dynamo: the cache's env plus -# the MX bundle, and nothing else. -_CACHE_ENV_NAME = "HF_HUB_CACHE" - - -def _modelexpress_replica(*, cache: bool = True, engines: list[v1alpha1.Engine] | None = None) -> v1alpha1.ModelReplica: - """A replica of engines, referencing the qwen cache unless cache is False.""" - engines = engines if engines is not None else [_standalone_engine(args=[])] - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen") if cache else None, - engines=engines, - ), - ) +def _to_dicts(*, objects: dict[str, k8sobjv1alpha1.Object]) -> dict[str, dict]: + """objects as dicts of only the fields the code set.""" + return {key: obj.model_dump(exclude_unset=True, by_alias=True) for key, obj in objects.items()} -def test_grove_gang_gets_modelexpress_env_on_both_cliques() -> None: - """A cached Grove gang on Dynamo gets the ModelExpress env and IPC_LOCK on both cliques.""" - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _modelexpress_replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - # Grove also gets the leader-address alias, unconditional on a cache, - # ahead of the cache env and the ModelExpress bundle. - want_env_names = _MODELEXPRESS_ENV_NAMES | {base.LEADER_ADDRESS_ENV, _CACHE_ENV_NAME} - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - env_names = {e["name"] for e in container["env"]} - assert env_names == want_env_names, f"{clique_name}: {env_names}" - assert container["env"][0] == base.grove_leader_address_env() - server_env = next(e for e in container["env"] if e["name"] == "MX_SERVER_ADDRESS") - # The per-cluster shared server's well-known Service, qualified by - # its namespace because the engine runs in its team's namespace. - assert server_env["value"] == "modelexpress-server.default.svc:8001" - mxurl_env = next(e for e in container["env"] if e["name"] == "MODEL_EXPRESS_URL") - assert mxurl_env["value"] == server_env["value"] - assert container["securityContext"] == {"capabilities": {"add": ["IPC_LOCK"]}} - - -def test_grove_gang_without_cache_gets_no_modelexpress_env() -> None: - """A Grove gang with no cache gets only the leader address alias, and no security context.""" - # No cache means no ModelExpress env or security context, but the - # leader-address alias is unconditional (it doesn't depend on a cache). - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _modelexpress_replica(cache=False, engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - assert container["env"] == [base.grove_leader_address_env()] - assert "securityContext" not in container - - -def test_native_engine_gets_modelexpress_env_on_dynamo() -> None: - """A cached Standalone engine on Dynamo gets the ModelExpress env and IPC_LOCK.""" - # A Standalone engine on a Dynamo cluster with a cache is as valid a P2P - # peer set as a gang, so it gets the full ModelExpress env and the - # IPC_LOCK security context on its engine container. - replica = _modelexpress_replica() - out = native.NativeBackend().build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo") - container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] - env = {e["name"]: e for e in container["env"]} - assert set(env) == _MODELEXPRESS_ENV_NAMES | {_CACHE_ENV_NAME} - assert env["MX_SERVER_ADDRESS"]["value"] == "modelexpress-server.default.svc:8001" - assert env["MX_P2P_METADATA"]["value"] == "1" - assert env["HF_HUB_CACHE"]["value"] == "/mnt/models" - # Isolates this cache's P2P source identity, qualified by the - # Modelplane namespace (like cache_pvc_name) so two namespaces' caches - # of the same name can't collide at the cluster's one shared server. - assert env["MX_MODEL_REVISION"]["value"] == base.cache_pvc_name("ml-team", "qwen") - for name, field in ( - ("POD_NAME", "metadata.name"), - ("POD_UID", "metadata.uid"), - ("POD_NAMESPACE", "metadata.namespace"), - ): - assert env[name]["valueFrom"]["fieldRef"]["fieldPath"] == field - assert container["securityContext"] == {"capabilities": {"add": ["IPC_LOCK"]}} - - -def test_native_engine_gets_no_modelexpress_env_on_standard() -> None: - """A cached Standalone engine on Standard gets only the cache env, and no security context.""" - # The same cached Standalone engine on a Standard cluster gets no - # ModelExpress env and no security context: the portable engine command - # falls back. It keeps the cache's own HF_HUB_CACHE, which is not part - # of the ModelExpress bundle and applies on every stack. - replica = _modelexpress_replica() - out = native.NativeBackend().build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Standard") - container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] - assert container["env"] == [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}] - assert "securityContext" not in container - assert container["args"] == [] - - -# The EPP prefix-cache producer's blockSizeTokens is derived best-effort -# from the engine flags (#179) so it matches the engine's KV block size. - - -def test_kv_block_size_defaults_to_16_when_absent() -> None: - """The KV block size defaults to 16 when no flag sets it.""" - assert routing._kv_block_size([]) == 16 - assert routing._kv_block_size(["--model=/mnt/models"]) == 16 - - -def test_kv_block_size_reads_vllm_block_size() -> None: - """The KV block size comes from vLLM's --block-size.""" - assert routing._kv_block_size(["--block-size", "32"]) == 32 - assert routing._kv_block_size(["--model=/m", "--block-size=8"]) == 8 - - -def test_kv_block_size_reads_sglang_page_size() -> None: - """The KV block size comes from SGLang's --page-size.""" - assert routing._kv_block_size(["--page-size=64"]) == 64 - - -def test_kv_block_size_non_integer_falls_back_to_default() -> None: - """A non-integer block size falls back to 16.""" - assert routing._kv_block_size(["--block-size", "auto"]) == 16 - - -def test_kv_block_size_rendered_config_uses_block_size() -> None: - """The rendered EPP config carries the block size in place of its placeholder.""" - cfg = routing._disaggregated_epp_config_yaml(32) - assert "blockSizeTokens: 32" in cfg - assert "BLOCK_SIZE_TOKENS" not in cfg - - -# The mirrored namespace a replica's objects land in. The expected names are -# spelled out, because compose-inference-cluster creates the namespace and -# compose-model-route and compose-model-cache land objects in it by the same -# derivation, and all four must agree. +def _sorted(*, objects: dict[str, dict]) -> dict[str, dict]: + """objects with their keys sorted, so pytest's diff of two lines them up.""" + return json.loads(json.dumps(objects, sort_keys=True)) -REMOTE_NAMESPACE_CASES = [ - pytest.param("ml-team", "mp-ml-team-51733", id="a short namespace keeps its name, prefixed and hashed"), - pytest.param( - # 63 is the longest a namespace can be, so mp- plus it can't be - # used as is. It's truncated to leave room for the hash. - "a" * 63, - "mp-" + "a" * 54 + "-38bfb", - id="the longest valid namespace still yields a valid one", + +BUILD_CASES = [ + # A Deployment reports readiness from its Available condition, and a claim + # template is ready once it's created. The engine and role labels are what + # the collector attributes the pod's metrics to. + BuildCase( + name="StandaloneEngine", + reason=( + "A Standalone engine composes a Deployment, its pod labeled with the engine and its role, and a claim " + "template for its member's device request." + ), + backend=native.NativeBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _claim_template(name="r-main-standalone-devices-f456f", count=1), + }, + ), + # See #378. + BuildCase( + name="StandaloneMemberMetadata", + reason=( + "A Standalone member's template labels and annotations land on its Deployment's pod template, merged with " + "the managed labels." + ), + backend=native.NativeBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + metadata=v1alpha1.Metadata( + labels={"example.com/role": "standalone"}, + annotations={"example.com/config": "standalone"}, + ), + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ), + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "example.com/role": "standalone", + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + }, + "annotations": {"example.com/config": "standalone"}, + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _claim_template(name="r-main-standalone-devices-f456f", count=1), + }, + ), + # Two replicas of one deployment on the same cluster, this one and + # dep-clusterB, must compose distinct resource names there. + BuildCase( + name="CoLocatedReplicaA", + reason="A replica named dep-clusterA qualifies its resource names with its own name.", + backend=native.NativeBackend(), + replica=_replica( + name="dep-clusterA", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="dep-clusterA", + stack="Standard", + want={ + "model-serving-main": _deployment( + name="dep-clusterA-main-3d1d5", + claim_template_name="dep-clusterA-main-standalone-devices-145eb", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "dep-clusterA", + "modelplane.ai/workload": "dep-clusterA-main-3d1d5", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _claim_template( + name="dep-clusterA-main-standalone-devices-145eb", count=1 + ), + }, + ), + BuildCase( + name="CoLocatedReplicaB", + reason=( + "A replica named dep-clusterB qualifies its resource names with its own name, so they don't collide with " + "dep-clusterA's." + ), + backend=native.NativeBackend(), + replica=_replica( + name="dep-clusterB", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="dep-clusterB", + stack="Standard", + want={ + "model-serving-main": _deployment( + name="dep-clusterB-main-d6c52", + claim_template_name="dep-clusterB-main-standalone-devices-5a8a8", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "dep-clusterB", + "modelplane.ai/workload": "dep-clusterB-main-d6c52", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _claim_template( + name="dep-clusterB-main-standalone-devices-5a8a8", count=1 + ), + }, + ), + # Qualifying the names by engine keeps a multi-engine replica's from + # colliding on the remote cluster. + BuildCase( + name="MultiEngineReplica", + reason=( + "A replica with two engines named prefill and decode composes a Deployment and a claim template for each, " + "named for its engine." + ), + backend=native.NativeBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="prefill", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + v1alpha1.Engine( + name="decode", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-prefill": _deployment( + name="r-prefill-d90b0", + claim_template_name="r-prefill-standalone-devices-f62af", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "prefill", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-prefill-d90b0", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-prefill-standalone": _claim_template(name="r-prefill-standalone-devices-f62af", count=1), + "model-serving-decode": _deployment( + name="r-decode-4b27b", + claim_template_name="r-decode-standalone-devices-63392", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "decode", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-decode-4b27b", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-decode-standalone": _claim_template(name="r-decode-standalone-devices-63392", count=1), + }, + ), + # resources.claims is a list-map keyed on name alone, so N device requests + # mustn't compose N container claims all named "devices". + BuildCase( + name="SeveralDeviceRequests", + reason=( + "A member requesting GPUs and NICs composes one container claim on the pod's claim, and a claim template " + "carrying both requests." + ), + backend=native.NativeBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest(name="gpu", deviceClassName="gpu.nvidia.com", count=8), + v1alpha1.DeviceRequest(name="nic", deviceClassName="nic.nvidia.com", count=8), + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + # Written out because it's the only claim template that asks for + # more than GPUs, which is all _claim_template builds. + "resource-claim-main-standalone": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": "r-main-standalone-devices-f456f", "namespace": "mp-ml-team-51733"}, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": {"deviceClassName": "gpu.nvidia.com", "count": 8}, + }, + { + "name": "nic", + "exactly": {"deviceClassName": "nic.nvidia.com", "count": 8}, + }, + ] + } + } + }, + } + }, + } + }, + }, + ), + # HF_HUB_CACHE makes the engine's own --model= resolve against the + # cache. Modelplane injects no --model of its own: naming the model is the + # command's job. + # + # HF_HUB_CACHE isn't part of the ModelExpress bundle, and applies on every + # stack. + BuildCase( + name="StandaloneCacheOnStandard", + reason=( + "A cached Standalone engine on Standard mounts the cache's PVC and points HF_HUB_CACHE at it, with no " + "ModelExpress env or security context." + ), + backend=native.NativeBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=v1alpha1.ModelCacheRef(name="qwen"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest", args=[]) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": [], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}], + } + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}, + }, + ], + ), + "resource-claim-main-standalone": _claim_template(name="r-main-standalone-devices-f456f", count=1), + }, + ), + # Kubernetes expands $(VAR) left to right, so Modelplane's own entries must + # precede the user's for a user entry to reference them. + BuildCase( + name="StandaloneEnvWithCache", + reason="A cached Standalone member's own env follows the cache env in its container.", + backend=native.NativeBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=v1alpha1.ModelCacheRef(name="qwen"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=[], + env=[v1alpha1.EnvItem(name="HF_TOKEN", value="x")], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": [], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + {"name": "HF_TOKEN", "value": "x"}, + ], + } + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}, + }, + ], + ), + "resource-claim-main-standalone": _claim_template(name="r-main-standalone-devices-f456f", count=1), + }, + ), + # A Standalone engine on a Dynamo cluster with a cache is as valid a P2P + # peer set as a gang. MX_SERVER_ADDRESS is the per-cluster shared server's + # well-known Service, qualified by its namespace because the engine runs in + # its team's. MX_MODEL_REVISION isolates this cache's P2P source identity, + # qualified by the Modelplane namespace like the cache's PVC name, so two + # namespaces' caches of the same name can't collide at the cluster's one + # shared server. + # + # HF_HUB_CACHE appears once. It's the cache's own env, and ModelExpress + # reads it only as a fallback for its cache root. + BuildCase( + name="StandaloneCacheOnDynamo", + reason=( + "A cached Standalone engine on Dynamo gets the ModelExpress env and the IPC_LOCK security context " + "alongside the cache env." + ), + backend=native.NativeBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=v1alpha1.ModelCacheRef(name="qwen"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest", args=[]) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": [], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_SERVER_ADDRESS", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MODEL_EXPRESS_URL", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MX_MODEL_REVISION", + "value": "modelcache-ml-team-qwen-17db2", + }, + {"name": "MX_P2P_METADATA", "value": "1"}, + { + "name": "POD_NAME", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.name"}}, + }, + { + "name": "POD_UID", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.uid"}}, + }, + { + "name": "POD_NAMESPACE", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, + }, + ], + "securityContext": {"capabilities": {"add": ["IPC_LOCK"]}}, + } + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}, + }, + ], + ), + "resource-claim-main-standalone": _claim_template(name="r-main-standalone-devices-f456f", count=1), + }, + ), + # There's no flag injection or bootstrap. The worker addresses the leader + # through $(MODELPLANE_LEADER_ADDRESS), which concatenates Grove's PCSG vars + # because they vary per gang. The PCS-scoped ones are identical across + # gangs, and would point every copy at gang 0's leader. There's no + # MODELPLANE_RANK: Grove exposes no group-wide pod index yet (grove#755), so + # a gang engine's command computes its own rank from GROVE_PCLQ_POD_INDEX. + # + # A PodCliqueSet publishes no Available condition, so its readiness derives + # from its replica counters. A worker with no template metadata carries only + # the managed labels, and with no cache there's no ModelExpress env or + # security context. + BuildCase( + name="GroveGang", + reason=( + "A Leader/Worker engine on Grove composes a PodCliqueSet with its members' commands verbatim, and a claim " + "template for each member." + ), + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _pod_clique_set( + name="r-main-bb4e3", + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Leader", + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Worker", + "kai.scheduler/queue": "modelplane", + }, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ], + ), + "resource-claim-main-leader": _claim_template(name="r-main-leader-devices-f58c6", count=8), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + }, + ), + BuildCase( + name="GroveMemberEnv", + reason="A Grove leader's own env follows the leader address alias in its container.", + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + env=[v1alpha1.EnvItem(name="HF_TOKEN", value="x")], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _pod_clique_set( + name="r-main-bb4e3", + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Leader", + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + {"name": "HF_TOKEN", "value": "x"}, + ], + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Worker", + "kai.scheduler/queue": "modelplane", + }, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ], + ), + "resource-claim-main-leader": _claim_template(name="r-main-leader-devices-f58c6", count=8), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + }, + ), + # Multi-NIC RDMA nodes need VLLM_HOST_IP from status.podIP so the engine + # binds the right interface (#141). + BuildCase( + name="GroveFieldRefEnv", + reason="A Grove leader's env drawn from a pod field passes through unchanged.", + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + env=[ + v1alpha1.EnvItem( + name="VLLM_HOST_IP", + valueFrom=v1alpha1.ValueFrom( + fieldRef=v1alpha1.FieldRef(fieldPath="status.podIP") + ), + ) + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _pod_clique_set( + name="r-main-bb4e3", + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Leader", + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + { + "name": "VLLM_HOST_IP", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + ], + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Worker", + "kai.scheduler/queue": "modelplane", + }, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ], + ), + "resource-claim-main-leader": _claim_template(name="r-main-leader-devices-f58c6", count=8), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + }, + ), + # Grove propagates a clique's labels and annotations to its pods (#378). + BuildCase( + name="GroveMemberMetadata", + reason=( + "Each Grove member's template labels and annotations land on its own clique alone, merged with the managed " + "labels." + ), + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + metadata=v1alpha1.Metadata( + labels={"example.com/role": "leader"}, + annotations={"example.com/config": "leader"}, + ), + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ), + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + metadata=v1alpha1.Metadata( + labels={"example.com/role": "worker"}, + annotations={"example.com/config": "worker"}, + ), + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ), + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _pod_clique_set( + name="r-main-bb4e3", + cliques=[ + { + "name": "leader", + "labels": { + "example.com/role": "leader", + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Leader", + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "annotations": {"example.com/config": "leader"}, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": { + "example.com/role": "worker", + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Worker", + "kai.scheduler/queue": "modelplane", + }, + "annotations": {"example.com/config": "worker"}, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ], + ), + "resource-claim-main-leader": _claim_template(name="r-main-leader-devices-f58c6", count=8), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + }, + ), + # A coordinator-only leader, like a vLLM DP head running + # --data-parallel-size-local=0, has no deviceRequests. + BuildCase( + name="ClaimlessGroveLeader", + reason=( + "A Grove leader with no device requests composes no claim or claim template, but still pins to its pool " + "and tolerates the GPU taint." + ), + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _pod_clique_set( + name="r-main-bb4e3", + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Leader", + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Worker", + "kai.scheduler/queue": "modelplane", + }, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ], + ), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + }, + ), + # The scheduler may split a gang across pools when no single pool satisfies + # every member, so each member's pods pin to that member's pool rather than + # an engine-wide one. + BuildCase( + name="GroveMembersSplitPools", + reason="A Grove leader on the head pool and worker on the frontier pool each pin to their own pool.", + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="head", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _pod_clique_set( + name="r-main-bb4e3", + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Leader", + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "head"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Worker", + "kai.scheduler/queue": "modelplane", + }, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ], + ), + "resource-claim-main-leader": _claim_template(name="r-main-leader-devices-f58c6", count=8), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + }, + ), + # Pointing HF_HUB_CACHE at the mount lets each clique's own --model= + # resolve against it. + BuildCase( + name="GroveCacheBareLeader", + reason=( + "A cached Grove gang whose leader has no command or args mounts the cache on both cliques and points " + "HF_HUB_CACHE at it, injecting no --model." + ), + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=v1alpha1.ModelCacheRef(name="kimi"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest", args=[]) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=["/bin/sh", "-c", "join"], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _pod_clique_set( + name="r-main-bb4e3", + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Leader", + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_SERVER_ADDRESS", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MODEL_EXPRESS_URL", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MX_MODEL_REVISION", + "value": "modelcache-ml-team-kimi-aa322", + }, + {"name": "MX_P2P_METADATA", "value": "1"}, + { + "name": "POD_NAME", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.name"}}, + }, + { + "name": "POD_UID", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.uid"}}, + }, + { + "name": "POD_NAMESPACE", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, + }, + ], + "securityContext": {"capabilities": {"add": ["IPC_LOCK"]}}, + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-kimi-aa322"}, + }, + ], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Worker", + "kai.scheduler/queue": "modelplane", + }, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "command": ["/bin/sh", "-c", "join"], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_SERVER_ADDRESS", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MODEL_EXPRESS_URL", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MX_MODEL_REVISION", + "value": "modelcache-ml-team-kimi-aa322", + }, + {"name": "MX_P2P_METADATA", "value": "1"}, + { + "name": "POD_NAME", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.name"}}, + }, + { + "name": "POD_UID", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.uid"}}, + }, + { + "name": "POD_NAMESPACE", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, + }, + ], + "securityContext": {"capabilities": {"add": ["IPC_LOCK"]}}, + } + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-kimi-aa322"}, + }, + ], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ], + ), + "resource-claim-main-leader": _claim_template(name="r-main-leader-devices-f58c6", count=8), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + }, + ), + BuildCase( + name="GroveCacheLeaderCommand", + reason=( + "A cached Grove gang whose leader has its own command mounts the cache, keeping the command verbatim with " + "no --model injected." + ), + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=v1alpha1.ModelCacheRef(name="kimi"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "python3 -m sglang.launch_server --model-path /mnt/models --tp 16", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=["/bin/sh", "-c", "join"], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _pod_clique_set( + name="r-main-bb4e3", + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Leader", + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "python3 -m sglang.launch_server --model-path /mnt/models --tp 16", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_SERVER_ADDRESS", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MODEL_EXPRESS_URL", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MX_MODEL_REVISION", + "value": "modelcache-ml-team-kimi-aa322", + }, + {"name": "MX_P2P_METADATA", "value": "1"}, + { + "name": "POD_NAME", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.name"}}, + }, + { + "name": "POD_UID", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.uid"}}, + }, + { + "name": "POD_NAMESPACE", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, + }, + ], + "securityContext": {"capabilities": {"add": ["IPC_LOCK"]}}, + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-kimi-aa322"}, + }, + ], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Worker", + "kai.scheduler/queue": "modelplane", + }, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "command": ["/bin/sh", "-c", "join"], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_SERVER_ADDRESS", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MODEL_EXPRESS_URL", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MX_MODEL_REVISION", + "value": "modelcache-ml-team-kimi-aa322", + }, + {"name": "MX_P2P_METADATA", "value": "1"}, + { + "name": "POD_NAME", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.name"}}, + }, + { + "name": "POD_UID", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.uid"}}, + }, + { + "name": "POD_NAMESPACE", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, + }, + ], + "securityContext": {"capabilities": {"add": ["IPC_LOCK"]}}, + } + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-kimi-aa322"}, + }, + ], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ], + ), + "resource-claim-main-leader": _claim_template(name="r-main-leader-devices-f58c6", count=8), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + }, + ), + # Each pod publishes itself as a source independently, so the worker needs + # the ModelExpress env too. + BuildCase( + name="GroveModelExpressEnv", + reason=( + "A cached Grove gang on Dynamo gets the ModelExpress env on both cliques, after the leader address alias " + "and the cache env." + ), + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=v1alpha1.ModelCacheRef(name="qwen"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _pod_clique_set( + name="r-main-bb4e3", + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Leader", + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_SERVER_ADDRESS", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MODEL_EXPRESS_URL", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MX_MODEL_REVISION", + "value": "modelcache-ml-team-qwen-17db2", + }, + {"name": "MX_P2P_METADATA", "value": "1"}, + { + "name": "POD_NAME", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.name"}}, + }, + { + "name": "POD_UID", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.uid"}}, + }, + { + "name": "POD_NAMESPACE", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, + }, + ], + "securityContext": {"capabilities": {"add": ["IPC_LOCK"]}}, + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}, + }, + ], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Worker", + "kai.scheduler/queue": "modelplane", + }, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_SERVER_ADDRESS", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MODEL_EXPRESS_URL", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MX_MODEL_REVISION", + "value": "modelcache-ml-team-qwen-17db2", + }, + {"name": "MX_P2P_METADATA", "value": "1"}, + { + "name": "POD_NAME", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.name"}}, + }, + { + "name": "POD_UID", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.uid"}}, + }, + { + "name": "POD_NAMESPACE", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, + }, + ], + "securityContext": {"capabilities": {"add": ["IPC_LOCK"]}}, + } + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}, + }, + ], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ], + ), + "resource-claim-main-leader": _claim_template(name="r-main-leader-devices-f58c6", count=8), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + }, + ), + # The worker followers never serve, so without the serving label the + # replica's Service can't route to them. They do carry the telemetry + # identity: a worker holds GPUs, and its metrics are the deployment's. Every + # gang container leads with the backend-neutral coordination vars aliasing + # LWS_LEADER_ADDRESS and LWS_WORKER_INDEX. A LeaderWorkerSet reports + # readiness from its Available condition. + BuildCase( + name="LLMDGang", + reason=( + "A Leader/Worker engine on llm-d composes a LeaderWorkerSet whose leader alone carries the serving label, " + "and whose worker carries only the engine and its role." + ), + backend=llmd.LLMDBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _leader_worker_set( + replicas=1, + size=2, + leader_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + leader_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + worker_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + } + ], + worker_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-leader": _claim_template(name="r-main-leader-devices-f58c6", count=8), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + }, + ), + # Modelplane is unopinionated about the engine: each member's command passes + # through verbatim, so a launch convention Modelplane has never heard of + # still works. + BuildCase( + name="LLMDGangCommands", + reason=( + "A Leader/Worker engine on llm-d whose leader and worker each set a command composes a LeaderWorkerSet " + "running both commands verbatim." + ), + backend=llmd.LLMDBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _leader_worker_set( + replicas=1, + size=2, + leader_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + leader_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + worker_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + } + ], + worker_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-leader": _claim_template(name="r-main-leader-devices-f58c6", count=8), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + }, + ), + BuildCase( + name="LLMDMultiNodeWorker", + reason="Two copies of a gang with a three-node worker compose a LeaderWorkerSet of two replicas of size four.", + backend=llmd.LLMDBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=2, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=3), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _leader_worker_set( + replicas=2, + size=4, + leader_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + leader_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + worker_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + } + ], + worker_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-leader": _claim_template(name="r-main-leader-devices-f58c6", count=8), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + }, + ), + # Only the native and Grove backends wire ModelExpress, and only on Dynamo, + # which never selects llm-d. HF_HUB_CACHE is the cache's own env, on every + # stack. + BuildCase( + name="LLMDWithCache", + reason="A cached Leader/Worker engine on llm-d gets the cache env but no ModelExpress env.", + backend=llmd.LLMDBackend(), + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=v1alpha1.ModelCacheRef(name="c"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _leader_worker_set( + replicas=1, + size=2, + leader_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + ], + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + leader_volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-c-c5cc6"}, + }, + ], + worker_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + ], + } + ], + worker_volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-c-c5cc6"}, + }, + ], + ), + "resource-claim-main-leader": _claim_template(name="r-main-leader-devices-f58c6", count=8), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + }, + ), +] + + +@pytest.mark.parametrize("case", BUILD_CASES, ids=lambda case: case.name) +def test_build(case: BuildCase) -> None: + """A backend composes an engine's workload, and the claim templates its members need.""" + # build composes one engine, so build each of the replica's engines in turn. + got: dict[str, k8sobjv1alpha1.Object] = {} + for engine in case.replica.spec.engines: + got.update(case.backend.build(case.replica, engine, case.provider_config, case.serving_label, case.stack)) + assert _sorted(objects=_to_dicts(objects=got)) == _sorted(objects=case.want), case.reason + + +SELECT_BACKEND_CASES = [ + SelectBackendCase( + name="StandaloneOnStandard", + reason="A Standalone engine on Standard selects the native backend.", + engine=v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", image="vllm/vllm-openai:latest", args=["--model=Qwen/Qwen3-0.6B"] + ) + ] + ) + ), + ) + ], + ), + stack="Standard", + want="native", + ), + SelectBackendCase( + name="StandaloneOnDynamo", + reason="A Standalone engine on Dynamo selects the native backend.", + engine=v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", image="vllm/vllm-openai:latest", args=["--model=Qwen/Qwen3-0.6B"] + ) + ] + ) + ), + ) + ], + ), + stack="Dynamo", + want="native", + ), + SelectBackendCase( + name="GangOnStandard", + reason="A Leader/Worker engine on Standard selects the llm-d backend.", + engine=v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + ], + ), + stack="Standard", + want="llmd", + ), + SelectBackendCase( + name="GangOnDynamo", + reason="A Leader/Worker engine on Dynamo selects the Grove backend.", + engine=v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + ], + ), + stack="Dynamo", + want="grove", + ), +] + + +@pytest.mark.parametrize("case", SELECT_BACKEND_CASES, ids=lambda case: case.name) +def test_select_backend(case: SelectBackendCase) -> None: + """An engine's member roles and its cluster's stack select its backend.""" + assert base.select_backend(case.engine, case.stack) == case.want, case.reason + + +# cache_mounts reads only a replica's namespace and cache, so the replicas here +# have no name and a placeholder cluster, c. +CACHE_MOUNTS_CASES = [ + CacheMountsCase( + name="NoCache", + reason="A replica with no cache gets no volume or mount.", + replica=_replica( + name=None, + namespace="ml-team", + cluster_name="c", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=[], + ) + ] + ) + ), + ) + ], + ) + ], + ), + want=([], []), + ), + CacheMountsCase( + name="WithCache", + reason="A replica with a cache gets a volume for its PVC and a mount for it.", + replica=_replica( + name=None, + namespace="ml-team", + cluster_name="c", + labels=None, + model_cache_ref=v1alpha1.ModelCacheRef(name="qwen"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=[], + ) + ] + ) + ), + ) + ], + ) + ], + ), + want=( + [{"name": "model-cache", "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}}], + [{"name": "model-cache", "mountPath": "/mnt/models"}], + ), + ), +] + + +@pytest.mark.parametrize("case", CACHE_MOUNTS_CASES, ids=lambda case: case.name) +def test_cache_mounts(case: CacheMountsCase) -> None: + """A replica's cache contributes a volume and a mount.""" + assert base.cache_mounts(case.replica) == case.want, case.reason + + +# cache_env reads only whether a replica has a cache, so the replicas here have +# no name and a placeholder cluster, c. +CACHE_ENV_CASES = [ + CacheEnvCase( + name="NoCache", + reason="A replica with no cache gets no env.", + replica=_replica( + name=None, + namespace="ml-team", + cluster_name="c", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=[], + ) + ] + ) + ), + ) + ], + ) + ], + ), + want=[], + ), + # The cache is staged in HuggingFace's cache layout, so pointing + # HF_HUB_CACHE at the mount is what lets an engine's own --model= + # resolve against it instead of pulling from HuggingFace (#407). There's no + # HF_HUB_OFFLINE: it would break an engine that fetches a different repo at + # startup, like kimi-k2's separately gated tokenizer, and resolution doesn't + # need it. + CacheEnvCase( + name="WithCache", + reason="A replica with a cache gets HF_HUB_CACHE pointing at the mount, and no HF_HUB_OFFLINE.", + replica=_replica( + name=None, + namespace="ml-team", + cluster_name="c", + labels=None, + model_cache_ref=v1alpha1.ModelCacheRef(name="qwen"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=[], + ) + ] + ) + ), + ) + ], + ) + ], + ), + want=[{"name": "HF_HUB_CACHE", "value": "/mnt/models"}], + ), +] + + +@pytest.mark.parametrize("case", CACHE_ENV_CASES, ids=lambda case: case.name) +def test_cache_env(case: CacheEnvCase) -> None: + """A replica's cache contributes the env that resolves a model against it.""" + assert base.cache_env(case.replica) == case.want, case.reason + + +APPLY_CASES = [ + # Both engines get the NIXL plumbing the schema can't express: a Memory + # /dev/shm, and VLLM_NIXL_SIDE_CHANNEL_HOST set to the pod IP. The decode + # engine moves to port 8001, behind the pd-sidecar on 8000, and keeps the + # port name http: the collector's engine job keeps a pod on that name, so a + # decode pod with no named port is one nothing scrapes. The sidecar's port + # is unnamed, because it serves inference rather than /metrics. + # + # PrefillDecode silently serves decode-only unless the picker's config arms + # the prefix-based PD decider. That needs nonCachedTokens > 0, the + # approx-prefix-cache-producer that populates the attribute it reads, pinned + # to autoTune: false, and no prepareDataPlugins feature gate, which the + # v0.8.0 EPP image rejects and crashloops on. The picker watches + # InferenceObjectives, so its Role must allow that. + # + # A wrong picker or sidecar image tag, or a stale EndpointPickerConfig API + # group, would otherwise only surface as a crashloop at deploy. Pinning them + # here means a bump shows up to be reviewed. + ApplyCase( + name="PrefillDecode", + reason=( + "PrefillDecode role-labels its prefill and decode engines, sidecars decode, and fronts both with an " + "InferencePool and endpoint picker." + ), + composed={ + "model-serving-prefill": _deployment( + name="r-prefill-d90b0", + claim_template_name="r-prefill-standalone-devices-f62af", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "prefill", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-prefill-d90b0", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-prefill-standalone": _claim_template(name="r-prefill-standalone-devices-f62af", count=1), + "model-serving-decode": _deployment( + name="r-decode-4b27b", + claim_template_name="r-decode-standalone-devices-63392", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "decode", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-decode-4b27b", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-decode-standalone": _claim_template(name="r-decode-standalone-devices-63392", count=1), + }, + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=v1alpha1.Serving(mode="PrefillDecode"), + engines=[ + v1alpha1.Engine( + name="prefill", + phase="Prefill", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + v1alpha1.Engine( + name="decode", + phase="Decode", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + ], + ), + provider_config="cluster-a-pc", + want={ + "model-serving-prefill": _deployment( + name="r-prefill-d90b0", + claim_template_name="r-prefill-standalone-devices-f62af", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "prefill", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-prefill-d90b0", + "llm-d.ai/role": "prefill", + "llm-d.ai/inference-serving": "true", + "app": "r", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + {"name": "VLLM_NIXL_SIDE_CHANNEL_PORT", "value": "5557"}, + ], + } + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + ), + "resource-claim-prefill-standalone": _claim_template(name="r-prefill-standalone-devices-f62af", count=1), + "model-serving-decode": _deployment( + name="r-decode-4b27b", + claim_template_name="r-decode-standalone-devices-63392", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "decode", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-decode-4b27b", + "llm-d.ai/role": "decode", + "llm-d.ai/inference-serving": "true", + "app": "r", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8001}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8001}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + {"name": "VLLM_NIXL_SIDE_CHANNEL_PORT", "value": "5557"}, + ], + }, + { + "name": "pd-sidecar", + "image": "ghcr.io/llm-d/llm-d-router-disagg-sidecar:v0.9.0", + "args": [ + "--secure-proxy=false", + "--kv-connector=nixlv2", + "--vllm-port=8001", + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + }, + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + ), + "resource-claim-decode-standalone": _claim_template(name="r-decode-standalone-devices-63392", count=1), + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: disagg-headers-handler\n" + "- type: queue-scorer\n" + "- type: prefill-filter\n" + "- type: decode-filter\n" + "- type: max-score-picker\n" + "- type: prefix-based-pd-decider\n" + " parameters:\n" + " nonCachedTokens: 16\n" + "- type: disagg-profile-handler\n" + " parameters:\n" + " deciders:\n" + " prefill: prefix-based-pd-decider\n" + "schedulingProfiles:\n" + "- name: prefill\n" + " plugins:\n" + " - pluginRef: prefill-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + "- name: decode\n" + " plugins:\n" + " - pluginRef: decode-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp( + labels={"app": "r-epp", "modelplane.ai/role": "picker"}, + config_checksum="f715f3024f37e7042d42628f10cbb04c7da78b96dc65953dac0f289ef2c7ef98", + ), + "epp-service": _epp_service(), + }, + ), + ApplyCase( + name="DecodeUserPort", + reason="A decode engine run with --port=9000 serves on that port, and its sidecar forwards to it there.", + composed={ + "model-serving-prefill": _deployment( + name="r-prefill-d90b0", + claim_template_name="r-prefill-standalone-devices-f62af", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "prefill", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-prefill-d90b0", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-prefill-standalone": _claim_template(name="r-prefill-standalone-devices-f62af", count=1), + "model-serving-decode": _deployment( + name="r-decode-4b27b", + claim_template_name="r-decode-standalone-devices-63392", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "decode", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-decode-4b27b", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=m", "--port=9000"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-decode-standalone": _claim_template(name="r-decode-standalone-devices-63392", count=1), + }, + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=v1alpha1.Serving(mode="PrefillDecode"), + engines=[ + v1alpha1.Engine( + name="prefill", + phase="Prefill", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + v1alpha1.Engine( + name="decode", + phase="Decode", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=m", "--port=9000"], + ) + ] + ) + ), + ) + ], + ), + ], + ), + provider_config="cluster-a-pc", + want={ + "model-serving-prefill": _deployment( + name="r-prefill-d90b0", + claim_template_name="r-prefill-standalone-devices-f62af", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "prefill", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-prefill-d90b0", + "llm-d.ai/role": "prefill", + "llm-d.ai/inference-serving": "true", + "app": "r", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + {"name": "VLLM_NIXL_SIDE_CHANNEL_PORT", "value": "5557"}, + ], + } + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + ), + "resource-claim-prefill-standalone": _claim_template(name="r-prefill-standalone-devices-f62af", count=1), + "model-serving-decode": _deployment( + name="r-decode-4b27b", + claim_template_name="r-decode-standalone-devices-63392", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "decode", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-decode-4b27b", + "llm-d.ai/role": "decode", + "llm-d.ai/inference-serving": "true", + "app": "r", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=m", "--port=9000"], + "ports": [{"name": "http", "containerPort": 9000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 9000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + {"name": "VLLM_NIXL_SIDE_CHANNEL_PORT", "value": "5557"}, + ], + }, + { + "name": "pd-sidecar", + "image": "ghcr.io/llm-d/llm-d-router-disagg-sidecar:v0.9.0", + "args": [ + "--secure-proxy=false", + "--kv-connector=nixlv2", + "--vllm-port=9000", + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + }, + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + ), + "resource-claim-decode-standalone": _claim_template(name="r-decode-standalone-devices-63392", count=1), + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: disagg-headers-handler\n" + "- type: queue-scorer\n" + "- type: prefill-filter\n" + "- type: decode-filter\n" + "- type: max-score-picker\n" + "- type: prefix-based-pd-decider\n" + " parameters:\n" + " nonCachedTokens: 16\n" + "- type: disagg-profile-handler\n" + " parameters:\n" + " deciders:\n" + " prefill: prefix-based-pd-decider\n" + "schedulingProfiles:\n" + "- name: prefill\n" + " plugins:\n" + " - pluginRef: prefill-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + "- name: decode\n" + " plugins:\n" + " - pluginRef: decode-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp( + labels={"app": "r-epp", "modelplane.ai/role": "picker"}, + config_checksum="f715f3024f37e7042d42628f10cbb04c7da78b96dc65953dac0f289ef2c7ef98", + ), + "epp-service": _epp_service(), + }, + ), + ApplyCase( + name="EngineNamesUnlikePhases", + reason=( + "Engines named alpha and beta take their PrefillDecode roles from their phases, so alpha, the Decode " + "engine, gets the sidecar." + ), + composed={ + "model-serving-alpha": _deployment( + name="r-alpha-b36ce", + claim_template_name="r-alpha-standalone-devices-9ec1c", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "alpha", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-alpha-b36ce", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-alpha-standalone": _claim_template(name="r-alpha-standalone-devices-9ec1c", count=1), + "model-serving-beta": _deployment( + name="r-beta-52d85", + claim_template_name="r-beta-standalone-devices-9d8b1", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "beta", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-beta-52d85", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-beta-standalone": _claim_template(name="r-beta-standalone-devices-9d8b1", count=1), + }, + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=v1alpha1.Serving(mode="PrefillDecode"), + engines=[ + v1alpha1.Engine( + name="alpha", + phase="Decode", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + v1alpha1.Engine( + name="beta", + phase="Prefill", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + ], + ), + provider_config="cluster-a-pc", + want={ + "model-serving-alpha": _deployment( + name="r-alpha-b36ce", + claim_template_name="r-alpha-standalone-devices-9ec1c", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "alpha", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-alpha-b36ce", + "llm-d.ai/role": "decode", + "llm-d.ai/inference-serving": "true", + "app": "r", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8001}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8001}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + {"name": "VLLM_NIXL_SIDE_CHANNEL_PORT", "value": "5557"}, + ], + }, + { + "name": "pd-sidecar", + "image": "ghcr.io/llm-d/llm-d-router-disagg-sidecar:v0.9.0", + "args": [ + "--secure-proxy=false", + "--kv-connector=nixlv2", + "--vllm-port=8001", + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + }, + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + ), + "resource-claim-alpha-standalone": _claim_template(name="r-alpha-standalone-devices-9ec1c", count=1), + "model-serving-beta": _deployment( + name="r-beta-52d85", + claim_template_name="r-beta-standalone-devices-9d8b1", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "beta", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-beta-52d85", + "llm-d.ai/role": "prefill", + "llm-d.ai/inference-serving": "true", + "app": "r", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + {"name": "VLLM_NIXL_SIDE_CHANNEL_PORT", "value": "5557"}, + ], + } + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + ), + "resource-claim-beta-standalone": _claim_template(name="r-beta-standalone-devices-9d8b1", count=1), + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: disagg-headers-handler\n" + "- type: queue-scorer\n" + "- type: prefill-filter\n" + "- type: decode-filter\n" + "- type: max-score-picker\n" + "- type: prefix-based-pd-decider\n" + " parameters:\n" + " nonCachedTokens: 16\n" + "- type: disagg-profile-handler\n" + " parameters:\n" + " deciders:\n" + " prefill: prefix-based-pd-decider\n" + "schedulingProfiles:\n" + "- name: prefill\n" + " plugins:\n" + " - pluginRef: prefill-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + "- name: decode\n" + " plugins:\n" + " - pluginRef: decode-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp( + labels={"app": "r-epp", "modelplane.ai/role": "picker"}, + config_checksum="f715f3024f37e7042d42628f10cbb04c7da78b96dc65953dac0f289ef2c7ef98", + ), + "epp-service": _epp_service(), + }, + ), + # The leader clique gets the llm-d role, inference-serving and app labels, + # the pd-sidecar and the NIXL plumbing. The worker clique never serves. + ApplyCase( + name="GroveDecodeGang", + reason=( + "A Grove decode gang has its leader clique decorated like a Deployment's pod template, and its worker " + "clique left as the backend composed it." + ), + composed={ + "model-serving-prefill": _deployment( + name="r-prefill-d90b0", + claim_template_name="r-prefill-standalone-devices-f62af", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "prefill", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-prefill-d90b0", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-prefill-standalone": _claim_template(name="r-prefill-standalone-devices-f62af", count=1), + "model-serving-decode": _pod_clique_set( + name="r-decode-4b27b", + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/engine": "decode", + "modelplane.ai/role": "Leader", + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-decode-leader-devices-d2e70", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": { + "modelplane.ai/engine": "decode", + "modelplane.ai/role": "Worker", + "kai.scheduler/queue": "modelplane", + }, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-decode-worker-devices-08d44", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ], + ), + "resource-claim-decode-leader": _claim_template(name="r-decode-leader-devices-d2e70", count=8), + "resource-claim-decode-worker": _claim_template(name="r-decode-worker-devices-08d44", count=8), + }, + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=v1alpha1.Serving(mode="PrefillDecode"), + engines=[ + v1alpha1.Engine( + name="prefill", + phase="Prefill", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + v1alpha1.Engine( + name="decode", + phase="Decode", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ), + ], + ), + provider_config="cluster-a-pc", + want={ + "model-serving-prefill": _deployment( + name="r-prefill-d90b0", + claim_template_name="r-prefill-standalone-devices-f62af", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "prefill", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-prefill-d90b0", + "llm-d.ai/role": "prefill", + "llm-d.ai/inference-serving": "true", + "app": "r", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + {"name": "VLLM_NIXL_SIDE_CHANNEL_PORT", "value": "5557"}, + ], + } + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + ), + "resource-claim-prefill-standalone": _claim_template(name="r-prefill-standalone-devices-f62af", count=1), + "model-serving-decode": _pod_clique_set( + name="r-decode-4b27b", + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/engine": "decode", + "modelplane.ai/role": "Leader", + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + "llm-d.ai/role": "decode", + "llm-d.ai/inference-serving": "true", + "app": "r", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + { + "name": "VLLM_NIXL_SIDE_CHANNEL_PORT", + "value": "5557", + }, + ], + "ports": [{"name": "http", "containerPort": 8001}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8001}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + }, + { + "name": "pd-sidecar", + "image": "ghcr.io/llm-d/llm-d-router-disagg-sidecar:v0.9.0", + "args": [ + "--secure-proxy=false", + "--kv-connector=nixlv2", + "--vllm-port=8001", + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + }, + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-decode-leader-devices-d2e70", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": { + "modelplane.ai/engine": "decode", + "modelplane.ai/role": "Worker", + "kai.scheduler/queue": "modelplane", + }, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-decode-worker-devices-08d44", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ], + ), + "resource-claim-decode-leader": _claim_template(name="r-decode-leader-devices-d2e70", count=8), + "resource-claim-decode-worker": _claim_template(name="r-decode-worker-devices-08d44", count=8), + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: disagg-headers-handler\n" + "- type: queue-scorer\n" + "- type: prefill-filter\n" + "- type: decode-filter\n" + "- type: max-score-picker\n" + "- type: prefix-based-pd-decider\n" + " parameters:\n" + " nonCachedTokens: 16\n" + "- type: disagg-profile-handler\n" + " parameters:\n" + " deciders:\n" + " prefill: prefix-based-pd-decider\n" + "schedulingProfiles:\n" + "- name: prefill\n" + " plugins:\n" + " - pluginRef: prefill-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + "- name: decode\n" + " plugins:\n" + " - pluginRef: decode-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp( + labels={"app": "r-epp", "modelplane.ai/role": "picker"}, + config_checksum="f715f3024f37e7042d42628f10cbb04c7da78b96dc65953dac0f289ef2c7ef98", + ), + "epp-service": _epp_service(), + }, + ), + # A replica with no serving block is served Unified. + # + # A single serving pod has nothing to pick between, but still gets the pool. + # Always fronting with one avoids swapping a Service for a pool when a + # second pod appears, a swap that would drop in-flight requests. The pool + # selects the pods by the serving label they already carry. + # + # The picker scores by prefix cache and queue depth in a single profile, + # with no prefill/decode split, fed by the approx-prefix-cache-producer. It + # reads its config once at startup, so its pod template carries a sha256 of + # the config, and a config change rolls the pod. Its image and the + # EndpointPickerConfig API group are pinned here too, so a wrong tag or a + # stale group fails here rather than crashlooping at deploy. + ApplyCase( + name="UnifiedOnePod", + reason=( + "A replica of one Standalone pod with no serving block is fronted by an InferencePool and endpoint picker." + ), + composed={ + "model-serving-main": _deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _claim_template(name="r-main-standalone-devices-f456f", count=1), + }, + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + want={ + "model-serving-main": _deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _claim_template(name="r-main-standalone-devices-f456f", count=1), + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: queue-scorer\n" + "- type: max-score-picker\n" + "schedulingProfiles:\n" + "- name: default\n" + " plugins:\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp( + labels={"app": "r-epp", "modelplane.ai/role": "picker"}, + config_checksum="20c1dfea3fc4ad41e335cc74edbeb1e8689a607bc4bf7395849ffd2cf0bb2ae1", + ), + "epp-service": _epp_service(), + }, + ), + # A built-in MetricMapping renames the picker's scheduling latency, and + # nothing can match it unless something scrapes the picker. The collector's + # engine job keeps a pod on two things: the deployment label, which comes + # from this replica's, and a container port named http, which every picker + # has. The app label is still the Deployment's own selector, which must not + # move. + ApplyCase( + name="UnifiedLabeledReplica", + reason="A Unified replica labeled with its deployment and index gets an endpoint picker labeled with both.", + composed={ + "model-serving-main": _deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/deployment": "qwen3-8b", + "modelplane.ai/replica": "2", + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _claim_template(name="r-main-standalone-devices-f456f", count=1), + }, + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels={"modelplane.ai/deployment": "qwen3-8b", "modelplane.ai/replica-index": "2"}, + model_cache_ref=None, + serving=v1alpha1.Serving(mode="Unified"), + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + want={ + "model-serving-main": _deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/deployment": "qwen3-8b", + "modelplane.ai/replica": "2", + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _claim_template(name="r-main-standalone-devices-f456f", count=1), + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: queue-scorer\n" + "- type: max-score-picker\n" + "schedulingProfiles:\n" + "- name: default\n" + " plugins:\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp( + labels={ + "app": "r-epp", + "modelplane.ai/role": "picker", + "modelplane.ai/deployment": "qwen3-8b", + "modelplane.ai/replica": "2", + }, + config_checksum="20c1dfea3fc4ad41e335cc74edbeb1e8689a607bc4bf7395849ffd2cf0bb2ae1", + ), + "epp-service": _epp_service(), + }, + ), + ApplyCase( + name="UnifiedSeveralPods", + reason=( + "A replica of two Standalone pods with no serving block is fronted by an InferencePool and endpoint picker." + ), + composed={ + "model-serving-main": _deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=2, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _claim_template(name="r-main-standalone-devices-f456f", count=1), + }, + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=2, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + want={ + "model-serving-main": _deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=2, + pod_metadata={ + "labels": { + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _claim_template(name="r-main-standalone-devices-f456f", count=1), + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: queue-scorer\n" + "- type: max-score-picker\n" + "schedulingProfiles:\n" + "- name: default\n" + " plugins:\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp( + labels={"app": "r-epp", "modelplane.ai/role": "picker"}, + config_checksum="20c1dfea3fc4ad41e335cc74edbeb1e8689a607bc4bf7395849ffd2cf0bb2ae1", + ), + "epp-service": _epp_service(), + }, + ), + # Unified routing reads the engine args for the KV block size through + # _serving_pod_templates, which normalizes a LeaderWorkerSet's + # leaderTemplate alongside a Deployment's pod template and a Grove + # PodCliqueSet's leader clique. A regression case for a normalization that + # only knew Deployment and PodCliqueSet, and raised KeyError on a + # LeaderWorkerSet. + ApplyCase( + name="UnifiedLeaderWorkerSet", + reason="A LeaderWorkerSet with no serving block is fronted by an InferencePool and endpoint picker.", + composed={ + "model-serving-main": _leader_worker_set( + replicas=1, + size=2, + leader_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + leader_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + worker_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + } + ], + worker_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-leader": _claim_template(name="r-main-leader-devices-f58c6", count=8), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + }, + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + want={ + "model-serving-main": _leader_worker_set( + replicas=1, + size=2, + leader_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + "ports": [{"name": "http", "containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + leader_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + worker_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + } + ], + worker_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-leader": _claim_template(name="r-main-leader-devices-f58c6", count=8), + "resource-claim-main-worker": _claim_template(name="r-main-worker-devices-99b8a", count=8), + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: queue-scorer\n" + "- type: max-score-picker\n" + "schedulingProfiles:\n" + "- name: default\n" + " plugins:\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp( + labels={"app": "r-epp", "modelplane.ai/role": "picker"}, + config_checksum="20c1dfea3fc4ad41e335cc74edbeb1e8689a607bc4bf7395849ffd2cf0bb2ae1", + ), + "epp-service": _epp_service(), + }, + ), +] + + +@pytest.mark.parametrize("case", APPLY_CASES, ids=lambda case: case.name) +def test_apply(case: ApplyCase) -> None: + """routing.apply fronts a replica's engines with the routing its serving mode selects.""" + composed = {key: k8sobjv1alpha1.Object.model_validate(copy.deepcopy(obj)) for key, obj in case.composed.items()} + got = routing.apply(composed, case.replica, case.provider_config) + assert _sorted(objects=_to_dicts(objects=got)) == _sorted(objects=case.want), case.reason + + +# The EPP prefix-cache producer's blockSizeTokens is derived best-effort from +# the engine flags (#179), so it matches the engine's KV block size. +KV_BLOCK_SIZE_CASES = [ + KvBlockSizeCase( + name="NoArgs", + reason="With no engine args the block size defaults to 16.", + engine_args=[], + want=16, + ), + KvBlockSizeCase( + name="NoBlockSizeFlag", + reason="With no block size flag the block size defaults to 16.", + engine_args=["--model=/mnt/models"], + want=16, + ), + KvBlockSizeCase( + name="VLLMBlockSize", + reason="vLLM's --block-size followed by 32 sets the block size.", + engine_args=["--block-size", "32"], + want=32, + ), + KvBlockSizeCase( + name="VLLMBlockSizeEquals", + reason="vLLM's --block-size=8 sets the block size.", + engine_args=["--model=/m", "--block-size=8"], + want=8, + ), + KvBlockSizeCase( + name="SGLangPageSize", + reason="SGLang's --page-size=64 sets the block size.", + engine_args=["--page-size=64"], + want=64, + ), + KvBlockSizeCase( + name="NonIntegerBlockSize", + reason="A block size of auto falls back to 16.", + engine_args=["--block-size", "auto"], + want=16, + ), +] + + +@pytest.mark.parametrize("case", KV_BLOCK_SIZE_CASES, ids=lambda case: case.name) +def test_kv_block_size(case: KvBlockSizeCase) -> None: + """The KV block size comes from the engine's flags.""" + assert routing._kv_block_size(case.engine_args) == case.want, case.reason + + +DISAGGREGATED_CONFIG_CASES = [ + DisaggregatedConfigCase( + name="BlockSize32", + reason="A block size of 32 renders in place of the config's placeholder.", + block_size=32, + want=( + "apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 32\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: disagg-headers-handler\n" + "- type: queue-scorer\n" + "- type: prefill-filter\n" + "- type: decode-filter\n" + "- type: max-score-picker\n" + "- type: prefix-based-pd-decider\n" + " parameters:\n" + " nonCachedTokens: 16\n" + "- type: disagg-profile-handler\n" + " parameters:\n" + " deciders:\n" + " prefill: prefix-based-pd-decider\n" + "schedulingProfiles:\n" + "- name: prefill\n" + " plugins:\n" + " - pluginRef: prefill-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + "- name: decode\n" + " plugins:\n" + " - pluginRef: decode-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + ), +] + + +@pytest.mark.parametrize("case", DISAGGREGATED_CONFIG_CASES, ids=lambda case: case.name) +def test_disaggregated_config(case: DisaggregatedConfigCase) -> None: + """The disaggregated EPP config renders with the engine's KV block size.""" + assert routing._disaggregated_epp_config_yaml(case.block_size) == case.want, case.reason + + +# compose-inference-cluster creates the namespace a replica's objects land in, +# and compose-model-route and compose-model-cache land objects in it by the same +# derivation, so all four must agree on it. +REMOTE_NAMESPACE_CASES = [ + RemoteNamespaceCase( + name="ShortNamespace", + reason="A short namespace keeps its name, prefixed with mp- and suffixed with a hash.", + replica=_replica( + name="r", + namespace="ml-team", + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + want="mp-ml-team-51733", + ), + # 63 is the longest a namespace can be, so mp- plus it can't be used as + # is. The lengths are the point, so the names are written as repeats rather + # than 63-character literals. + RemoteNamespaceCase( + name="LongestNamespace", + reason="A 63-character namespace is truncated to leave room for the hash, yielding a 63-character name.", + replica=_replica( + name="r", + namespace="a" * 63, + cluster_name="cluster-a", + labels=None, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + want="mp-" + "a" * 54 + "-38bfb", ), ] -@pytest.mark.parametrize(("namespace", "want"), REMOTE_NAMESPACE_CASES) -def test_remote_namespace(namespace: str, want: str) -> None: +@pytest.mark.parametrize("case", REMOTE_NAMESPACE_CASES, ids=lambda case: case.name) +def test_remote_namespace(case: RemoteNamespaceCase) -> None: """A replica's objects land in a namespace mirroring its own.""" - got = base.remote_namespace(_replica(namespace=namespace)) - assert got == want - assert len(got) <= 63 + assert base.remote_namespace(case.replica) == case.want, case.reason diff --git a/functions/compose-model-replica/tests/test_fn.py b/functions/compose-model-replica/tests/test_fn.py index f78010a1c..88efbe8c5 100644 --- a/functions/compose-model-replica/tests/test_fn.py +++ b/functions/compose-model-replica/tests/test_fn.py @@ -12,7 +12,11 @@ # See the License for the specific language governing permissions and # limitations under the License. -"""Tests for the compose-model-replica function.""" +"""Tests for the compose-model-replica function. + +Every resource the function composes on the remote cluster lands in +mp-ml-team-51733, the namespace mirroring the replica's. +""" import asyncio import dataclasses @@ -28,59 +32,22 @@ from models.ai.modelplane.modelreplica import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -# A GPU device request CEL selector, as compose-model-deployment stamps it. -_GPU_CEL = 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' - -# Unified routing fronts the serving pods with an InferencePool + endpoint -# picker; their manifests are asserted in detail in test_backends. Here we -# only check the function wired the whole set in (and dropped the plain -# Service), then drop their manifests so the golden covers the dispatch, -# wiring and readiness the function itself owns. -_ROUTING_KEYS = { - "inference-pool", - "epp", - "epp-config", - "epp-role", - "epp-rolebinding", - "epp-serviceaccount", - "epp-service", -} - @dataclasses.dataclass class Case: """A test case for compose-model-replica.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse -def _observed_object(*, ready: bool) -> fnv1.Resource: - """A composed provider-kubernetes Object as observed back, with the Ready - condition its readiness policy derives.""" - return fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True" if ready else "False", - "reason": "Available" if ready else "Unavailable", - "lastTransitionTime": "2025-01-01T00:00:00Z", - }, - ], - }, - } - ), - ) - +def _model_replica() -> fnv1.Resource: + """The ModelReplica XR test-replica in ml-team, with one Standalone engine. -def _compose_cases() -> list[Case]: - """The compose cases. Later cases are built from earlier ones.""" + The device request's CEL selector is as compose-model-deployment stamps it. + """ xr = v1alpha1.ModelReplica( metadata=metav1.ObjectMeta( name="test-replica", @@ -105,7 +72,11 @@ def _compose_cases() -> list[Case]: name="gpu", deviceClassName="gpu.nvidia.com", count=1, - selectors=[v1alpha1.Selector(cel=_GPU_CEL)], + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], ), ], template=v1alpha1.Template( @@ -124,450 +95,559 @@ def _compose_cases() -> list[Case]: ), ], ), - ).model_dump(exclude_none=True, mode="json") - - cluster_requirement = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceCluster", - match_name="cluster-a", ) + return fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json", by_alias=True))) - # Case 1: cluster resolved with providerConfigRef — composes native - # Deployment. First reconcile: none of the composed resources are in - # observed yet, so none are marked ready (the function only asserts - # readiness for a resource it can see in observed state). - req1 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), - ), - ) - req1.required_resources["cluster"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceCluster", - "metadata": {"name": "cluster-a"}, - "spec": { - "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k"}}}, - }, - "status": { - "providerConfigRef": {"name": "cluster-a-pc"}, - "gateway": {"address": "10.0.0.1"}, - }, - } - ) - ) - ) - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - resources={ - "model-serving-main": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", +def _cluster(*, provider_config_ref: str | None) -> fnv1.Resource: + """The InferenceCluster cluster-a, with no status until it reports a providerConfigRef.""" + cluster = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "cluster-a"}, + "spec": { + "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k"}}}, + }, + } + if provider_config_ref is not None: + cluster["status"] = { + "providerConfigRef": {"name": provider_config_ref}, + "gateway": {"address": "10.0.0.1"}, + } + return fnv1.Resource(resource=resource.dict_to_struct(cluster)) + + +def _deployment(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the engine's Deployment. + + Its pod carries the labels the collector attributes the engine's metrics by: + the deployment, the engine and the member's role. + """ + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": { + "name": "test-replica-main-6b608", + "namespace": "mp-ml-team-51733", + }, "spec": { - "providerConfigRef": { - "kind": "ClusterProviderConfig", - "name": "cluster-a-pc", - }, - "readiness": { - "policy": "DeriveFromCelQuery", - "celQuery": ( - "has(object.status.conditions) && " - "object.status.conditions.exists(" - 'c, c.type == "Available" && c.status == "True")' - ), - }, - "forProvider": { - "manifest": { - "apiVersion": "apps/v1", - "kind": "Deployment", - "metadata": { - "name": resource.child_name("test-replica", "main"), - "namespace": "mp-ml-team-51733", - }, - "spec": { - "replicas": 1, - "selector": { - "matchLabels": { - "modelplane.ai/workload": resource.child_name( - "test-replica", "main" - ), - }, - }, - "template": { - "metadata": { - "labels": { - "modelplane.ai/deployment": "my-deployment", - "modelplane.ai/engine": "main", - "modelplane.ai/role": "Standalone", - "modelplane.ai/serving": "test-replica", - "modelplane.ai/workload": resource.child_name( - "test-replica", "main" - ), - }, - }, - "spec": { - "containers": [ - { - "name": "engine", - "image": "vllm/vllm-openai:latest", - "args": ["--model=Qwen/Qwen3-0.6B"], - "ports": [{"name": "http", "containerPort": 8000}], - "resources": {"claims": [{"name": "devices"}]}, - "volumeMounts": [ - {"name": "dshm", "mountPath": "/dev/shm"}, - ], - "readinessProbe": { - "httpGet": {"path": "/health", "port": 8000}, - "initialDelaySeconds": 30, - "periodSeconds": 10, - "timeoutSeconds": 5, - }, - }, - ], - "volumes": [ - {"name": "dshm", "emptyDir": {"medium": "Memory"}}, - ], - "nodeSelector": {"modelplane.ai/pool": "frontier"}, - "resourceClaims": [ - { - "name": "devices", - "resourceClaimTemplateName": resource.child_name( - "test-replica", "main", "standalone", "devices" - ), - }, - ], - "tolerations": [ - { - "key": "nvidia.com/gpu", - "operator": "Exists", - "effect": "NoSchedule", - }, - ], + "replicas": 1, + "selector": {"matchLabels": {"modelplane.ai/workload": "test-replica-main-6b608"}}, + "template": { + "metadata": { + "labels": { + "modelplane.ai/deployment": "my-deployment", + "modelplane.ai/engine": "main", + "modelplane.ai/role": "Standalone", + "modelplane.ai/serving": "test-replica", + "modelplane.ai/workload": "test-replica-main-6b608", + } + }, + "spec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"name": "http", "containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, }, - }, - }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "test-replica-main-standalone-devices-e609d", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], }, }, }, } - ), - ), - "model-route": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", + }, + }, + } + ), + ready=ready, + ) + + +def _route(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the HTTPRoute to the InferencePool.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "HTTPRoute", + "metadata": {"name": "test-replica", "namespace": "mp-ml-team-51733"}, "spec": { - "providerConfigRef": { - "kind": "ClusterProviderConfig", - "name": "cluster-a-pc", - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "gateway.networking.k8s.io/v1", - "kind": "HTTPRoute", - "metadata": { - "name": "test-replica", - "namespace": "mp-ml-team-51733", - }, - "spec": { - "parentRefs": [ - { - "name": "cluster-gateway", - "namespace": "modelplane-system", + "parentRefs": [{"name": "cluster-gateway", "namespace": "modelplane-system"}], + "rules": [ + { + "matches": [ + { + "path": { + "type": "PathPrefix", + "value": "/ml-team/test-replica/", + } + } + ], + "timeouts": {"request": "0s"}, + "filters": [ + { + "type": "URLRewrite", + "urlRewrite": { + "path": { + "type": "ReplacePrefixMatch", + "replacePrefixMatch": "/", + } }, - ], - "rules": [ - { - "matches": [ - { - "path": { - "type": "PathPrefix", - "value": "/ml-team/test-replica/", - }, - }, - ], - "timeouts": {"request": "0s"}, - "filters": [ - { - "type": "URLRewrite", - "urlRewrite": { - "path": { - "type": "ReplacePrefixMatch", - "replacePrefixMatch": "/", - }, - }, - }, - ], - "backendRefs": [ + } + ], + "backendRefs": [ + { + "group": "inference.networking.k8s.io", + "kind": "InferencePool", + "name": "test-replica-pool", + } + ], + } + ], + }, + } + }, + }, + } + ), + ready=ready, + ) + + +def _claim_template(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the engine's ResourceClaimTemplate.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": { + "name": "test-replica-main-standalone-devices-e609d", + "namespace": "mp-ml-team-51733", + }, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ { - "group": "inference.networking.k8s.io", - "kind": "InferencePool", - "name": "test-replica-pool", - }, + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } ], }, - ], - }, - }, - }, + } + ] + } + } }, } - ), - ), - "resource-claim-main-standalone": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", + }, + }, + } + ), + ready=ready, + ) + + +def _inference_pool(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the InferencePool fronting the engine.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "inference.networking.k8s.io/v1", + "kind": "InferencePool", + "metadata": {"name": "test-replica-pool", "namespace": "mp-ml-team-51733"}, "spec": { - "providerConfigRef": { - "kind": "ClusterProviderConfig", - "name": "cluster-a-pc", + "selector": {"matchLabels": {"modelplane.ai/serving": "test-replica"}}, + "targetPorts": [{"number": 8000}], + "endpointPickerRef": { + "name": "test-replica-epp", + "port": {"number": 9002}, + "failureMode": "FailOpen", }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "resource.k8s.io/v1", - "kind": "ResourceClaimTemplate", - "metadata": { - "name": resource.child_name( - "test-replica", "main", "standalone", "devices" - ), - "namespace": "mp-ml-team-51733", + }, + } + }, + }, + } + ), + ready=ready, + ) + + +def _epp(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the endpoint picker's Deployment. + + Its pod carries the labels the collector attributes the picker's metrics by: + the deployment and the picker role. + """ + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": "test-replica-epp", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": 1, + "selector": {"matchLabels": {"app": "test-replica-epp"}}, + "template": { + "metadata": { + "labels": { + "app": "test-replica-epp", + "modelplane.ai/role": "picker", + "modelplane.ai/deployment": "my-deployment", }, - "spec": { - "spec": { - "devices": { - "requests": [ - { - "name": "gpu", - "exactly": { - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "selectors": [ - {"cel": {"expression": _GPU_CEL}}, - ], - }, - }, - ], - }, - }, + "annotations": { + "modelplane.ai/epp-config-checksum": "20c1dfea3fc4ad41e335cc74edbeb1e8689a607bc4bf7395849ffd2cf0bb2ae1" }, }, + "spec": { + "serviceAccountName": "test-replica-epp", + "containers": [ + { + "name": "epp", + "image": "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0", + "args": [ + "--pool-name=test-replica-pool", + "--pool-namespace=mp-ml-team-51733", + "--pool-group=inference.networking.k8s.io", + "--config-file=/config/epp-config.yaml", + "--grpc-port=9002", + "--metrics-port=9090", + "--metrics-endpoint-auth=false", + ], + "ports": [ + {"name": "grpc", "containerPort": 9002}, + {"name": "grpc-health", "containerPort": 9003}, + {"name": "http", "containerPort": 9090}, + ], + "volumeMounts": [{"name": "config", "mountPath": "/config"}], + } + ], + "volumes": [ + { + "name": "config", + "configMap": {"name": "test-replica-epp"}, + } + ], + }, }, }, } - ), - ), - }, + }, + }, + } ), - conditions=[ - fnv1.Condition( - type="ModelAccepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Deploying", - ), - fnv1.Condition( - type="ModelReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForModel", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Composing vllm/vllm-openai:latest on cluster-a", - ), - ], - context=structpb.Struct(), + ready=ready, ) - want1.requirements.resources["cluster"].CopyFrom(cluster_requirement) - # Case 2: cluster not resolved — early return with conditions. - req2 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), + +def _epp_config(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the endpoint picker's ConfigMap.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "test-replica-epp", "namespace": "mp-ml-team-51733"}, + "data": { + "epp-config.yaml": ( + "apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: queue-scorer\n" + "- type: max-score-picker\n" + "schedulingProfiles:\n" + "- name: default\n" + " plugins:\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ) + }, + } + }, + }, + } ), + ready=ready, ) - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - # Nothing is composed while waiting, so the XR is marked not ready - # rather than left to aggregate to trivially ready. - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - conditions=[ - fnv1.Condition( - type="ModelAccepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - fnv1.Condition( - type="ModelReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForModel", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for cluster to be resolved", - ), - ], - context=structpb.Struct(), - ) - want2.requirements.resources["cluster"].CopyFrom(cluster_requirement) - # Case 3: cluster resolved but no providerConfigRef — early return. - req3 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), +def _epp_role(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the endpoint picker's Role.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "Role", + "metadata": {"name": "test-replica-epp", "namespace": "mp-ml-team-51733"}, + "rules": [ + { + "apiGroups": [""], + "resources": ["pods"], + "verbs": ["get", "watch", "list"], + }, + { + "apiGroups": ["inference.networking.k8s.io"], + "resources": ["inferencepools"], + "verbs": ["get", "watch", "list"], + }, + { + "apiGroups": ["inference.networking.x-k8s.io"], + "resources": ["inferenceobjectives"], + "verbs": ["get", "watch", "list"], + }, + ], + } + }, + }, + } ), + ready=ready, ) - req3.required_resources["cluster"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceCluster", - "metadata": {"name": "cluster-a"}, - "spec": { - "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k"}}}, + + +def _epp_role_binding(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the endpoint picker's RoleBinding.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "RoleBinding", + "metadata": {"name": "test-replica-epp", "namespace": "mp-ml-team-51733"}, + "subjects": [ + { + "kind": "ServiceAccount", + "name": "test-replica-epp", + "namespace": "mp-ml-team-51733", + } + ], + "roleRef": { + "apiGroup": "rbac.authorization.k8s.io", + "kind": "Role", + "name": "test-replica-epp", + }, + } }, - } - ) - ) + }, + } + ), + ready=ready, ) - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - conditions=[ - fnv1.Condition( - type="ModelAccepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - fnv1.Condition( - type="ModelReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForModel", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for cluster providerConfigRef", - ), - ], - context=structpb.Struct(), + +def _epp_service_account(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the endpoint picker's ServiceAccount.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ServiceAccount", + "metadata": {"name": "test-replica-epp", "namespace": "mp-ml-team-51733"}, + } + }, + }, + } + ), + ready=ready, ) - want3.requirements.resources["cluster"].CopyFrom(cluster_requirement) - - # The routing objects' manifests are dropped from the golden (see - # _ROUTING_KEYS). - for key in _ROUTING_KEYS: - want1.desired.resources[key].CopyFrom(fnv1.Resource()) - - # Case 4: the resources from case 1 now exist in observed, and the - # workload Object reports Available (so its derived Ready is True). The - # function marks each observed resource ready once its Object reports - # Ready: the workload because it's serving and the rest because existing - # is being ready for them. Built from case 1, mutating only what the - # observed-ready transition changes: the three ready flags, the - # acceptance/readiness conditions, and the dropped first-reconcile event. - req4 = fnv1.RunFunctionRequest() - req4.CopyFrom(req1) - # The workload Object as provider-kubernetes observes it back: applied - # (atProvider.manifest populated) and Available (its derived Ready=True). - req4.observed.resources["model-serving-main"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": {"forProvider": {"manifest": {"kind": "Deployment"}}}, - "status": { - "atProvider": {"manifest": {"kind": "Deployment"}}, - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2025-01-01T00:00:00Z", + + +def _epp_service(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the endpoint picker's Service.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Service", + "metadata": {"name": "test-replica-epp", "namespace": "mp-ml-team-51733"}, + "spec": { + "selector": {"app": "test-replica-epp"}, + "ports": [ + { + "name": "grpc-ext-proc", + "port": 9002, + "targetPort": 9002, + "appProtocol": "http2", + } + ], }, - ], + } }, - } - ), - ) + }, + } + ), + ready=ready, ) - # The other two have no runtime readiness to wait on, so under their - # SuccessfulCreate policy provider-kubernetes reports them Ready once - # applied. (The InferencePool + endpoint picker resources aren't - # observed here, so they stay unready.) - for key in ("model-route", "resource-claim-main-standalone"): - req4.observed.resources[key].CopyFrom(_observed_object(ready=True)) - - want4 = fnv1.RunFunctionResponse() - want4.CopyFrom(want1) - for key in ("model-serving-main", "model-route", "resource-claim-main-standalone"): - want4.desired.resources[key].ready = fnv1.READY_TRUE - del want4.conditions[:] - want4.conditions.extend( - [ - fnv1.Condition(type="ModelAccepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Accepted"), - fnv1.Condition(type="ModelReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Serving"), - ] + + +def _observed_deployment() -> fnv1.Resource: + """The engine's Deployment Object as observed back, applied and Available, so its derived Ready is True.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": {"forProvider": {"manifest": {"kind": "Deployment"}}}, + "status": { + "atProvider": {"manifest": {"kind": "Deployment"}}, + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2025-01-01T00:00:00Z", + }, + ], + }, + } + ), + ) + + +# One helper plays nine roles: the observed route, claim template, InferencePool +# and six endpoint picker objects. The function reads only an observed Object's +# Ready condition, so all nine have the same shape, and writing out the seven +# that appear in only two cases inline would bury those cases. +def _observed_object(*, ready: bool) -> fnv1.Resource: + """A composed Object as observed back, with the Ready condition its readiness policy derives.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True" if ready else "False", + "reason": "Available" if ready else "Unavailable", + "lastTransitionTime": "2025-01-01T00:00:00Z", + }, + ], + }, + } + ), ) - # The "Composing ..." event fires only the first reconcile (model-serving - # not yet observed), so it's gone now. - del want4.results[:] - - # Case 5: everything from case 4 plus the routing objects is observed, - # but the endpoint picker's Service failed to apply, say because its - # name was invalid, so its Object isn't Ready. Being observed isn't - # being applied, so it stays unready and holds the replica unready with - # it. Built from case 4, mutating only the routing objects' ready flags. - req5 = fnv1.RunFunctionRequest() - req5.CopyFrom(req4) - for key in _ROUTING_KEYS: - req5.observed.resources[key].CopyFrom(_observed_object(ready=key != "epp-service")) - - want5 = fnv1.RunFunctionResponse() - want5.CopyFrom(want4) - for key in _ROUTING_KEYS - {"epp-service"}: - want5.desired.resources[key].ready = fnv1.READY_TRUE - - # Case 6: as case 5, but everything applied and the endpoint picker's - # Deployment isn't Available yet, so its Object's CEL-derived Ready is - # False. The gateway fails closed without a picker, so the replica stays - # unready with it. - req6 = fnv1.RunFunctionRequest() - req6.CopyFrom(req4) - for key in _ROUTING_KEYS: - req6.observed.resources[key].CopyFrom(_observed_object(ready=key != "epp")) - - want6 = fnv1.RunFunctionResponse() - want6.CopyFrom(want4) - for key in _ROUTING_KEYS - {"epp"}: - want6.desired.resources[key].ready = fnv1.READY_TRUE - - return [ - Case(name="cluster ready composes native Deployment", req=req1, want=want1), - Case(name="cluster not resolved returns waiting conditions", req=req2, want=want2), - Case(name="cluster without providerConfigRef returns waiting conditions", req=req3, want=want3), - Case(name="observed resources are marked ready", req=req4, want=want4), - Case(name="an object that failed to apply stays unready", req=req5, want=want5), - Case(name="an unavailable endpoint picker stays unready", req=req6, want=want6), - ] def _to_dict(msg: message.Message) -> dict: @@ -575,19 +655,324 @@ def _to_dict(msg: message.Message) -> dict: return json.loads(json_format.MessageToJson(msg, sort_keys=True)) -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +COMPOSE_CASES = [ + # The function only asserts readiness for a resource it can see in observed + # state. + Case( + name="ClusterResolved", + reason=( + "With its cluster resolved, a first reconcile composes a native Deployment fronted by an InferencePool and " + "endpoint picker, and marks none of it ready." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_model_replica()), + required_resources={"cluster": fnv1.Resources(items=[_cluster(provider_config_ref="cluster-a-pc")])}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + resources={ + "model-serving-main": _deployment(ready=fnv1.READY_UNSPECIFIED), + "model-route": _route(ready=fnv1.READY_UNSPECIFIED), + "resource-claim-main-standalone": _claim_template(ready=fnv1.READY_UNSPECIFIED), + "inference-pool": _inference_pool(ready=fnv1.READY_UNSPECIFIED), + "epp": _epp(ready=fnv1.READY_UNSPECIFIED), + "epp-config": _epp_config(ready=fnv1.READY_UNSPECIFIED), + "epp-role": _epp_role(ready=fnv1.READY_UNSPECIFIED), + "epp-rolebinding": _epp_role_binding(ready=fnv1.READY_UNSPECIFIED), + "epp-serviceaccount": _epp_service_account(ready=fnv1.READY_UNSPECIFIED), + "epp-service": _epp_service(ready=fnv1.READY_UNSPECIFIED), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Composing vllm/vllm-openai:latest on cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "cluster": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_name="cluster-a", + ), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelAccepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Deploying", + ), + fnv1.Condition( + type="ModelReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForModel", + ), + ], + ), + ), + # Nothing is composed while waiting, so the XR is marked not ready rather + # than left to aggregate to trivially ready. + Case( + name="ClusterUnresolved", + reason="Before its cluster resolves, a replica composes nothing and reports waiting, with the XR not ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_model_replica()), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for cluster to be resolved", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "cluster": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_name="cluster-a", + ), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelAccepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + fnv1.Condition( + type="ModelReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForModel", + ), + ], + ), + ), + Case( + name="NoProviderConfigRef", + reason=( + "A replica whose cluster has no providerConfigRef yet composes nothing and reports waiting, with the XR " + "not ready." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_model_replica()), + required_resources={"cluster": fnv1.Resources(items=[_cluster(provider_config_ref=None)])}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for cluster providerConfigRef", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "cluster": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_name="cluster-a", + ), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelAccepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + fnv1.Condition( + type="ModelReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForModel", + ), + ], + ), + ), + # The route and claim template have no runtime readiness to wait on, so + # their SuccessfulCreate Objects report Ready once applied. The "Composing + # ..." event fires only on the first reconcile, before the workload is + # observed, so there are no results. + Case( + name="ResourcesObserved", + reason=( + "With the workload observed Available and the route and claim template observed Ready, each is marked " + "ready, while the unobserved routing objects stay unready." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_replica(), + resources={ + "model-serving-main": _observed_deployment(), + "model-route": _observed_object(ready=True), + "resource-claim-main-standalone": _observed_object(ready=True), + }, + ), + required_resources={"cluster": fnv1.Resources(items=[_cluster(provider_config_ref="cluster-a-pc")])}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + resources={ + "model-serving-main": _deployment(ready=fnv1.READY_TRUE), + "model-route": _route(ready=fnv1.READY_TRUE), + "resource-claim-main-standalone": _claim_template(ready=fnv1.READY_TRUE), + "inference-pool": _inference_pool(ready=fnv1.READY_UNSPECIFIED), + "epp": _epp(ready=fnv1.READY_UNSPECIFIED), + "epp-config": _epp_config(ready=fnv1.READY_UNSPECIFIED), + "epp-role": _epp_role(ready=fnv1.READY_UNSPECIFIED), + "epp-rolebinding": _epp_role_binding(ready=fnv1.READY_UNSPECIFIED), + "epp-serviceaccount": _epp_service_account(ready=fnv1.READY_UNSPECIFIED), + "epp-service": _epp_service(ready=fnv1.READY_UNSPECIFIED), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "cluster": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_name="cluster-a", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ModelAccepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Accepted"), + fnv1.Condition(type="ModelReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Serving"), + ], + ), + ), + # An observed Object that isn't Ready is how a Service that failed to apply + # looks, say because its name was invalid. Being observed isn't being + # applied, and Crossplane holds the XR unready with it. + Case( + name="PickerServiceUnready", + reason=( + "With everything observed but the endpoint picker's Service Object not Ready, that Object stays unready." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_replica(), + resources={ + "model-serving-main": _observed_deployment(), + "model-route": _observed_object(ready=True), + "resource-claim-main-standalone": _observed_object(ready=True), + "inference-pool": _observed_object(ready=True), + "epp": _observed_object(ready=True), + "epp-config": _observed_object(ready=True), + "epp-role": _observed_object(ready=True), + "epp-rolebinding": _observed_object(ready=True), + "epp-serviceaccount": _observed_object(ready=True), + "epp-service": _observed_object(ready=False), + }, + ), + required_resources={"cluster": fnv1.Resources(items=[_cluster(provider_config_ref="cluster-a-pc")])}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + resources={ + "model-serving-main": _deployment(ready=fnv1.READY_TRUE), + "model-route": _route(ready=fnv1.READY_TRUE), + "resource-claim-main-standalone": _claim_template(ready=fnv1.READY_TRUE), + "inference-pool": _inference_pool(ready=fnv1.READY_TRUE), + "epp": _epp(ready=fnv1.READY_TRUE), + "epp-config": _epp_config(ready=fnv1.READY_TRUE), + "epp-role": _epp_role(ready=fnv1.READY_TRUE), + "epp-rolebinding": _epp_role_binding(ready=fnv1.READY_TRUE), + "epp-serviceaccount": _epp_service_account(ready=fnv1.READY_TRUE), + "epp-service": _epp_service(ready=fnv1.READY_UNSPECIFIED), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "cluster": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_name="cluster-a", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ModelAccepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Accepted"), + fnv1.Condition(type="ModelReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Serving"), + ], + ), + ), + # The Object's CEL-derived Ready is False. Crossplane holds the XR unready + # with it, because the gateway fails closed without a picker, whatever the + # InferencePool's failureMode says. + Case( + name="PickerUnavailable", + reason=( + "With everything observed but the endpoint picker's Object reporting Ready False, as it does until its " + "Deployment is Available, that Object stays unready." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_replica(), + resources={ + "model-serving-main": _observed_deployment(), + "model-route": _observed_object(ready=True), + "resource-claim-main-standalone": _observed_object(ready=True), + "inference-pool": _observed_object(ready=True), + "epp": _observed_object(ready=False), + "epp-config": _observed_object(ready=True), + "epp-role": _observed_object(ready=True), + "epp-rolebinding": _observed_object(ready=True), + "epp-serviceaccount": _observed_object(ready=True), + "epp-service": _observed_object(ready=True), + }, + ), + required_resources={"cluster": fnv1.Resources(items=[_cluster(provider_config_ref="cluster-a-pc")])}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + resources={ + "model-serving-main": _deployment(ready=fnv1.READY_TRUE), + "model-route": _route(ready=fnv1.READY_TRUE), + "resource-claim-main-standalone": _claim_template(ready=fnv1.READY_TRUE), + "inference-pool": _inference_pool(ready=fnv1.READY_TRUE), + "epp": _epp(ready=fnv1.READY_UNSPECIFIED), + "epp-config": _epp_config(ready=fnv1.READY_TRUE), + "epp-role": _epp_role(ready=fnv1.READY_TRUE), + "epp-rolebinding": _epp_role_binding(ready=fnv1.READY_TRUE), + "epp-serviceaccount": _epp_service_account(ready=fnv1.READY_TRUE), + "epp-service": _epp_service(ready=fnv1.READY_TRUE), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "cluster": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_name="cluster-a", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ModelAccepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Accepted"), + fnv1.Condition(type="ModelReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Serving"), + ], + ), + ), +] + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: - """The function dispatches to a backend to compose serving resources on a remote cluster.""" + """RunFunction dispatches to a backend to compose serving resources on a remote cluster.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - got_dict = _to_dict(got) - resources = got_dict.get("desired", {}).get("resources", {}) - if "model-serving-main" in resources: - assert set(resources) >= _ROUTING_KEYS - assert "model-service" not in resources - # The routing objects land in the mirrored namespace too, - # before they're dropped from the golden below. - for key in _ROUTING_KEYS: - manifest = resources[key]["resource"]["spec"]["forProvider"]["manifest"] - assert manifest["metadata"]["namespace"] == "mp-ml-team-51733", key - del resources[key]["resource"] - assert got_dict == _to_dict(case.want) + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-model-route/tests/test_fn.py b/functions/compose-model-route/tests/test_fn.py index c19c64d14..c287990f5 100644 --- a/functions/compose-model-route/tests/test_fn.py +++ b/functions/compose-model-route/tests/test_fn.py @@ -15,7 +15,6 @@ """Tests for the compose-model-route function.""" import asyncio -import base64 import dataclasses import json @@ -26,23 +25,8 @@ from google.protobuf import duration_pb2 as durationpb from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb -from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 -from models.ai.modelplane.inferencegateway import v1alpha1 as igv1alpha1 -from models.ai.modelplane.modelendpoint import v1alpha1 as mev1alpha1 from models.ai.modelplane.modelroute import v1alpha1 - -_NS = "ml-team" -_SVC = "assistant" -_MODEL = f"{_NS}/{_SVC}" -_GW = "eu" -_CLUSTER_CA = "-----BEGIN CERTIFICATE-----\ncluster\n-----END CERTIFICATE-----\n" -_CLIENT_CA = "-----BEGIN CERTIFICATE-----\nclient\n-----END CERTIFICATE-----\n" - - -def _be(ep: str) -> str: - """A composed backend object's name: child_name of the ModelRoute's own name - (service-gateway) and the endpoint's.""" - return resource.child_name(f"{_SVC}-{_GW}", ep) +from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @dataclasses.dataclass @@ -50,204 +34,483 @@ class Case: """A test case for compose-model-route.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse -def _entry( - label: str, *, name: str | None = None, priority: int | None = None, weight: int | None = None -) -> v1alpha1.Endpoint: - kwargs = {} - if priority is not None: - kwargs["priority"] = priority - if weight is not None: - kwargs["weight"] = weight - return v1alpha1.Endpoint( - name=name or label, - selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": label}), - **kwargs, +def _model_route(*, endpoints: list[v1alpha1.Endpoint]) -> fnv1.Resource: + """The ModelRoute XR pinning ml-team's assistant service to gateway eu.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ModelRoute( + apiVersion="modelplane.ai/v1alpha1", + kind="ModelRoute", + metadata=metav1.ObjectMeta(name="assistant-eu", namespace="ml-team"), + spec=v1alpha1.Spec( + gatewayName="eu", + serviceName="assistant", + endpoints=endpoints, + timeouts=v1alpha1.Timeouts(request="600s", idle="0s"), + ), + ).model_dump(exclude_none=True, mode="json", by_alias=True) + ) ) -def _route_xr(entries: list[v1alpha1.Endpoint], *, gateway: str = _GW) -> dict: - xr = v1alpha1.ModelRoute( - apiVersion="modelplane.ai/v1alpha1", - kind="ModelRoute", - metadata={"name": f"{_SVC}-{gateway}", "namespace": _NS}, - spec=v1alpha1.Spec( - gatewayName=gateway, - serviceName=_SVC, - endpoints=entries, - timeouts=v1alpha1.Timeouts(request="600s", idle="0s"), - ), +def _desired_model_route(*, address: str | None, total_endpoints: int, ready_endpoints: int) -> fnv1.Resource: + """The desired ModelRoute XR, not yet ready, reporting its model, its gateway's address and its endpoint counts.""" + status: dict = {"model": "ml-team/assistant"} + if address is not None: + status["address"] = address + status["endpoints"] = {"total": total_endpoints, "ready": ready_endpoints} + return fnv1.Resource(resource=resource.dict_to_struct({"status": status}), ready=fnv1.READY_FALSE) + + +def _inference_gateway(*, tls: bool, address: str, client_ca_published: bool) -> fnv1.Resource: + """The InferenceGateway eu on cluster gw-eu, as the gateway requirement returns it.""" + spec: dict = {"clusterName": "gw-eu"} + if tls: + spec["tls"] = {"certificateRefs": [{"name": "eu-tls"}]} + status: dict = {"address": address} + if client_ca_published: + status["clientCACertificate"] = "-----BEGIN CERTIFICATE-----\nclient\n-----END CERTIFICATE-----\n" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceGateway", + "metadata": {"name": "eu"}, + "spec": spec, + "status": status, + } + ) ) - return xr.model_dump(exclude_none=True, mode="json", by_alias=True) -def _endpoint( - name: str, +def _cluster(*, gateway_ca_published: bool) -> fnv1.Resource: + """The InferenceCluster gw-eu the gateway runs on, as the clusters requirement returns it.""" + status: dict = {"providerConfigRef": {"name": "gw-eu-pc"}} + if gateway_ca_published: + status["gateway"] = {"caCertificate": "-----BEGIN CERTIFICATE-----\ncluster\n-----END CERTIFICATE-----\n"} + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "gw-eu"}, + "spec": { + "cluster": { + "source": "Existing", + "existing": {"secretRef": {"name": "gw-eu-kubeconfig", "key": "kubeconfig"}}, + }, + "stack": "Standard", + }, + "status": status, + } + ) + ) + + +def _composed_endpoint(*, model: str | None) -> fnv1.Resource: + """The ready ModelEndpoint self, which Modelplane composed on gw-eu, as an endpoints requirement returns it.""" + spec: dict = {"origin": "https://gw-eu.example.com"} + if model is not None: + spec["model"] = model + spec["api"] = {"schema": "OpenAI", "prefix": "/v1"} + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": { + "name": "self", + "namespace": "ml-team", + "labels": {"modelplane.ai/cluster": "gw-eu", "modelplane.ai/deployment": "d"}, + }, + "spec": spec, + "status": { + "conditions": [ + { + "type": "EndpointReady", + "status": "True", + "reason": "EndpointUsable", + "lastTransitionTime": "2026-06-08T00:00:00Z", + } + ] + }, + } + ) + ) + + +def _third_party_endpoint( *, + name: str, origin: str, - model: str | None = None, - api: mev1alpha1.Api | None = None, - credential: str | None = None, - ready: bool = True, - composed: bool = False, -) -> dict: - ep = mev1alpha1.ModelEndpoint( - apiVersion="modelplane.ai/v1alpha1", - kind="ModelEndpoint", - metadata={"name": name, "namespace": _NS}, - spec=mev1alpha1.Spec( - origin=origin, - **({"model": model} if model else {}), - **({"api": api} if api else {}), - **( - { - "credential": mev1alpha1.Credential( - method="APIKey", apiKey=mev1alpha1.ApiKey(secretRef=mev1alpha1.SecretRef(name=credential)) - ) - } - if credential - else {} - ), - ), + model: str | None, + schema: str, + api_key_secret: str | None, + ready: bool, +) -> fnv1.Resource: + """A ModelEndpoint without the cluster label, so third-party, EndpointUsable if ready and CredentialMissing if not.""" + spec: dict = {"origin": origin} + if model is not None: + spec["model"] = model + spec["api"] = {"schema": schema, "prefix": "/v1"} + if api_key_secret is not None: + spec["credential"] = {"method": "APIKey", "apiKey": {"secretRef": {"name": api_key_secret, "key": "apiKey"}}} + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": {"name": name, "namespace": "ml-team"}, + "spec": spec, + "status": { + "conditions": [ + { + "type": "EndpointReady", + "status": "True" if ready else "False", + "reason": "EndpointUsable" if ready else "CredentialMissing", + "lastTransitionTime": "2026-06-08T00:00:00Z", + } + ] + }, + } + ) ) - d = ep.model_dump(exclude_none=True, mode="json", by_alias=True) - if composed: - d["metadata"]["labels"] = {"modelplane.ai/cluster": "gw-eu", "modelplane.ai/deployment": "d"} - d["status"] = { - "conditions": [ + + +def _api_key_secret(*, name: str, data: dict[str, str]) -> fnv1.Resource: + """A Secret in ml-team holding an endpoint's API key, as a credential requirement returns it.""" + return fnv1.Resource( + resource=resource.dict_to_struct( { - "type": "EndpointReady", - "status": "True" if ready else "False", - "reason": "EndpointUsable" if ready else "CredentialMissing", - "lastTransitionTime": "2026-06-08T00:00:00Z", + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": name, "namespace": "ml-team"}, + "data": data, } - ] - } - return d + ) + ) -def _gateway(*, client_ca: str | None = _CLIENT_CA, address: str | None = "203.0.113.1", tls: bool = False) -> dict: - spec = igv1alpha1.Spec(clusterName="gw-eu") - if tls: - spec.tls = igv1alpha1.Tls(certificateRefs=[igv1alpha1.CertificateRef(name="eu-tls")]) - gw = igv1alpha1.InferenceGateway( - apiVersion="modelplane.ai/v1alpha1", - kind="InferenceGateway", - metadata={"name": _GW}, - spec=spec, +def _composed_endpoint_backend() -> fnv1.Resource: + """The composed Backend for self, pinning this route's copy of gw-eu's CA and presenting its client certificate.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "Backend", + "metadata": {"name": "assistant-eu-self-a42e6", "namespace": "mp-ml-team-51733"}, + "spec": { + "endpoints": [{"fqdn": {"hostname": "gw-eu.example.com", "port": 443}}], + "tls": { + "caCertificateRefs": [ + {"kind": "ConfigMap", "group": "", "name": "assistant-eu-gw-eu-ca-3dd16"} + ], + "sni": "gw-eu.example.com", + "clientCertificateRef": { + "kind": "Secret", + "group": "", + "name": "assistant-eu-client-08324", + }, + }, + }, + } + }, + }, + } + ) ) - d = gw.model_dump(exclude_none=True, mode="json", by_alias=True) - status: dict = {} - if address: - status["address"] = address - if client_ca: - status["clientCACertificate"] = client_ca - if status: - d["status"] = status - return d - - -def _cluster(name: str, *, provider_config: str | None = "gw-eu-pc", ca: str | None = _CLUSTER_CA) -> dict: - c = icv1alpha1.InferenceCluster( - apiVersion="modelplane.ai/v1alpha1", - kind="InferenceCluster", - metadata={"name": name}, - spec=icv1alpha1.Spec( - cluster=icv1alpha1.Cluster( - source="Existing", - existing=icv1alpha1.Existing( - secretRef=icv1alpha1.SecretRef(name=f"{name}-kubeconfig", key="kubeconfig") - ), - ) - ), + + +def _composed_endpoint_ai_backend() -> fnv1.Resource: + """The composed AIServiceBackend for self, which keeps the caller header because Modelplane operates self.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "aigateway.envoyproxy.io/v1beta1", + "kind": "AIServiceBackend", + "metadata": {"name": "assistant-eu-self-a42e6", "namespace": "mp-ml-team-51733"}, + "spec": { + "schema": {"name": "OpenAI", "prefix": "/v1"}, + "backendRef": { + "group": "gateway.envoyproxy.io", + "kind": "Backend", + "name": "assistant-eu-self-a42e6", + }, + }, + } + }, + }, + } + ) ) - d = c.model_dump(exclude_none=True, mode="json", by_alias=True) - status: dict = {} - if provider_config: - status["providerConfigRef"] = {"name": provider_config} - if ca: - status["gateway"] = {"caCertificate": ca} - if status: - d["status"] = status - return d - - -def _secret(name: str, data: dict[str, str]) -> dict: - return { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": name, "namespace": _NS}, - "data": {k: base64.b64encode(v.encode()).decode() for k, v in data.items()}, - } - - -def _required(**resources) -> dict: # noqa: ANN003 - return { - name: fnv1.Resources(items=[fnv1.Resource(resource=resource.dict_to_struct(r)) for r in items]) - for name, items in resources.items() - } - - -def _requirements(entries: list[v1alpha1.Endpoint], *, credentials: dict[str, str] | None = None) -> fnv1.Requirements: - """The requirements the function emits: the named gateway, every cluster, a - ModelEndpoint selector per entry, and a Secret per endpoint that names a - credential (endpoint name -> Secret name).""" - reqs = { - "gateway": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name=_GW), - "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), - } - for entry in entries: - reqs[f"endpoints-{entry.name}"] = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="ModelEndpoint", - namespace=_NS, - match_labels=fnv1.MatchLabels(labels=dict(entry.selector.matchLabels)), + + +def _third_party_backend(*, name: str, hostname: str) -> fnv1.Resource: + """A composed Backend for a third-party endpoint, which trusts the system CAs.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "Backend", + "metadata": {"name": name, "namespace": "mp-ml-team-51733"}, + "spec": { + "endpoints": [{"fqdn": {"hostname": hostname, "port": 443}}], + "tls": {"wellKnownCACertificates": "System", "sni": hostname}, + }, + } + }, + }, + } + ) + ) + + +def _third_party_ai_backend(*, name: str, schema: str) -> fnv1.Resource: + """A composed AIServiceBackend for a third-party endpoint, which strips the caller header.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "aigateway.envoyproxy.io/v1beta1", + "kind": "AIServiceBackend", + "metadata": {"name": name, "namespace": "mp-ml-team-51733"}, + "spec": { + "schema": {"name": schema, "prefix": "/v1"}, + "backendRef": {"group": "gateway.envoyproxy.io", "kind": "Backend", "name": name}, + "headerMutation": {"remove": ["x-modelplane-caller"]}, + }, + } + }, + }, + } + ) + ) + + +def _credential(*, name: str, api_key: str) -> fnv1.Resource: + """The composed copy of an endpoint's API key Secret, under the apiKey key the AI Gateway reads.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": name, "namespace": "mp-ml-team-51733"}, + "type": "Opaque", + "data": {"apiKey": api_key}, + } + }, + }, + } + ) + ) + + +def _credential_policy(*, name: str, auth: dict) -> fnv1.Resource: + """The composed BackendSecurityPolicy sending an endpoint's API key the way auth says.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "aigateway.envoyproxy.io/v1beta1", + "kind": "BackendSecurityPolicy", + "metadata": {"name": name, "namespace": "mp-ml-team-51733"}, + "spec": { + **auth, + "targetRefs": [ + {"group": "aigateway.envoyproxy.io", "kind": "AIServiceBackend", "name": name} + ], + }, + } + }, + }, + } + ) + ) + + +def _cluster_ca() -> fnv1.Resource: + """The composed ConfigMap holding gw-eu's gateway CA, named for this route so no other route composes it.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "assistant-eu-gw-eu-ca-3dd16", "namespace": "mp-ml-team-51733"}, + "data": {"ca.crt": "-----BEGIN CERTIFICATE-----\ncluster\n-----END CERTIFICATE-----\n"}, + } + }, + }, + } ) - for endpoint, secret in (credentials or {}).items(): - reqs[f"credential-{endpoint}"] = fnv1.ResourceSelector( - api_version="v1", kind="Secret", namespace=_NS, match_name=secret + ) + + +def _client_certificate() -> fnv1.Resource: + """The composed client Certificate the composed endpoint's backend presents.""" + # Issued from the gateway's CA ClusterIssuer into this namespace. Named for + # this route, so no other route in the namespace composes it, and deleted + # with the route, so it sets no managementPolicies. + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.conditions) && " + "object.status.conditions.exists(c, c.type == 'Ready' && c.status == 'True')" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "Certificate", + "metadata": {"name": "assistant-eu-client-08324", "namespace": "mp-ml-team-51733"}, + "spec": { + "secretName": "assistant-eu-client-08324", + "commonName": "inference-gateway-eu", + "usages": ["client auth", "digital signature", "key encipherment"], + "duration": "2160h", + "renewBefore": "720h", + "privateKey": {"algorithm": "ECDSA", "size": 256, "rotationPolicy": "Always"}, + "issuerRef": { + "name": "inference-gateway-ca", + "kind": "ClusterIssuer", + "group": "cert-manager.io", + }, + }, + } + }, + }, + } ) - return fnv1.Requirements(resources=reqs) - - -def _not_ready( - status: dict, reason: str, message: str, requirements: fnv1.Requirements, *, warning: str | None = None -) -> fnv1.RunFunctionResponse: - """The whole response for a pass that composes nothing: the status counts so - far, a not-ready composite, one RoutingReady=False condition, and the reason - as a result. A warning about endpoints dropped before the tier emptied - precedes it.""" - results = [] - if warning is not None: - results.append(fnv1.Result(severity=fnv1.SEVERITY_WARNING, message=warning)) - results.append(fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message=message)) - return fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": status}), - ready=fnv1.READY_FALSE, - ) - ), - context=structpb.Struct(), - requirements=requirements, - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=reason, - message=message, - ) - ], - results=results, ) -def _manifest(rsp: fnv1.RunFunctionResponse, key: str) -> dict: - return resource.struct_to_dict(rsp.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] +def _ai_gateway_route(*, section_name: str, backend_refs: list[dict]) -> fnv1.Resource: + """The composed AIGatewayRoute matching ml-team/assistant, bound to the gateway's section_name listener.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.conditions) && " + "object.status.conditions.exists(c, c.type == 'Accepted' && c.status == 'True')" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "aigateway.envoyproxy.io/v1beta1", + "kind": "AIGatewayRoute", + "metadata": {"name": "assistant", "namespace": "mp-ml-team-51733"}, + "spec": { + # The route lives in the team's namespace but + # attaches across to the gateway. + "parentRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + "namespace": "modelplane-system", + "sectionName": section_name, + } + ], + "rules": [ + { + "matches": [ + { + "headers": [ + { + "type": "Exact", + "name": "x-ai-eg-model", + "value": "ml-team/assistant", + } + ] + } + ], + "backendRefs": backend_refs, + "timeouts": {"request": "600s"}, + "streamIdleTimeout": "0s", + "modelsOwnedBy": "ml-team", + } + ], + # Declaring the token costs is what makes the + # ext-proc ask a backend for usage on a streamed + # response, which otherwise reports none, and is + # where the metered counts in the access log + # come from. + "llmRequestCosts": [ + {"metadataKey": "llm_input_token", "type": "InputToken"}, + {"metadataKey": "llm_output_token", "type": "OutputToken"}, + {"metadataKey": "llm_total_token", "type": "TotalToken"}, + ], + }, + } + }, + }, + } + ) + ) def _to_dict(msg: message.Message) -> dict: @@ -255,475 +518,1517 @@ def _to_dict(msg: message.Message) -> dict: return json.loads(json_format.MessageToJson(msg, sort_keys=True)) -# Passes where a route can't be composed compose nothing and say why. Asserting -# the whole response proves nothing is composed against a cluster the route -# can't yet reach, rather than a subset being applied. -def _gates_cases() -> list[Case]: - composed = _endpoint("self", origin="https://gw-eu.example.com", composed=True) - return [ - Case( - name="the gateway's client PKI hasn't issued, so nothing can name its certificate", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) - ), - required_resources=_required( - gateway=[_gateway(client_ca=None)], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [composed]}, - ), - ), - want=_not_ready( - {"model": _MODEL, "endpoints": {"total": 0, "ready": 0}}, - fn.CONDITION_REASON_WAITING_FOR_GATEWAY, - "InferenceGateway eu has not published its client CA", - _requirements([_entry("d")]), - ), - ), - Case( - name="no selected endpoint is ready", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) - ), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", ready=False)]}, - ), - ), - want=_not_ready( - { - "model": _MODEL, - "address": "203.0.113.1", - "endpoints": {"total": 1, "ready": 0}, - }, - fn.CONDITION_REASON_NO_ENDPOINTS, - "None of the 1 selected ModelEndpoints is ready to carry traffic", - _requirements([_entry("d")]), +# Every ModelRoute here sets timeouts other than the ModelService's defaults, so +# the AIGatewayRoute's timeouts can only have come from the ModelRoute. Every +# object composed onto the gateway's cluster lands in mp-ml-team-51733, the +# namespace mirroring the route's own. compose-inference-cluster composes that +# namespace, so it isn't among the composed resources. +# +# The cases where the route can't be composed compose nothing and say why. Their +# whole responses show that no subset of the route is applied. +# +# Secret data is base64 encoded, as the API server stores it: c2stMQ== is +# "sk-1", c2stdG9n "sk-tog" and c2stcHJvdmlkZXI= "sk-provider". +COMPOSE_CASES = [ + Case( + name="ClientCAUnpublished", + reason="A ModelRoute whose gateway hasn't published its client CA composes nothing, because no backend could name its client certificate.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=False)] + ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources(items=[_composed_endpoint(model=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_model_route(address=None, total_endpoints=0, ready_endpoints=0)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="InferenceGateway eu has not published its client CA" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + } ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="InferenceGateway eu has not published its client CA", + ) + ], ), - Case( - name="a composed endpoint whose cluster withdrew its CA is dropped", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) + ), + # The unready endpoint has the composed one's name and origin but no cluster + # label. An unready endpoint is dropped before its cluster is read, so the + # label would change nothing. + Case( + name="EndpointNotReady", + reason="A ModelRoute whose only selected endpoint isn't ready composes nothing and reports NoReadyEndpoints.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] ), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu", ca=None)], - **{"endpoints-d": [composed]}, + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources( + items=[ + _third_party_endpoint( + name="self", + origin="https://gw-eu.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=False, + ) + ] ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=0) ), - want=_not_ready( - { - "model": _MODEL, - "address": "203.0.113.1", - "endpoints": {"total": 1, "ready": 0}, - }, - fn.CONDITION_REASON_NO_ENDPOINTS, - "None of the 1 selected ModelEndpoints is ready to carry traffic", - _requirements([_entry("d")]), - warning="Endpoints left out of the route, their cluster has published no gateway CA: self", - ), - ), - Case( - name="a credential Secret missing its key drops the endpoint", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("a")]))) - ), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-a": [_endpoint("wrongkey", origin="https://a.example.com", credential="k")], - "credential-wrongkey": [_secret("k", {"token": "sk-1"})], - }, + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="None of the 1 selected ModelEndpoints is ready to carry traffic", + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReadyEndpoints", + message="None of the 1 selected ModelEndpoints is ready to carry traffic", + ) + ], + ), + ), + Case( + name="ClusterCAUnpublished", + reason="A composed endpoint whose cluster hasn't published its gateway CA is left out of the route with a warning, leaving no ready endpoint.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=False)]), + "endpoints-d": fnv1.Resources(items=[_composed_endpoint(model=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=0) ), - want=_not_ready( - { - "model": _MODEL, - "address": "203.0.113.1", - "endpoints": {"total": 1, "ready": 0}, + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="Endpoints left out of the route, their cluster has published no gateway CA: self", + ), + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="None of the 1 selected ModelEndpoints is ready to carry traffic", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReadyEndpoints", + message="None of the 1 selected ModelEndpoints is ready to carry traffic", + ) + ], + ), + ), + Case( + name="CredentialKeyMissing", + reason="An endpoint whose credential Secret lacks the apiKey key is left out of the route with a warning, leaving no ready endpoint.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="a", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "a"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-a": fnv1.Resources( + items=[ + _third_party_endpoint( + name="wrongkey", + origin="https://a.example.com", + model=None, + schema="OpenAI", + api_key_secret="k", + ready=True, + ) + ] + ), + "credential-wrongkey": fnv1.Resources(items=[_api_key_secret(name="k", data={"token": "c2stMQ=="})]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=0) + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="Endpoints left out of the route, their credential Secret missing or missing its key: wrongkey", + ), + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="None of the 1 selected ModelEndpoints is ready to carry traffic", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-a": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "a"}), + ), + "credential-wrongkey": fnv1.ResourceSelector( + api_version="v1", kind="Secret", namespace="ml-team", match_name="k" + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReadyEndpoints", + message="None of the 1 selected ModelEndpoints is ready to carry traffic", + ) + ], + ), + ), + Case( + name="ComposedAndThirdParty", + reason="A composed endpoint at priority 0 and a third-party provider at priority 1 get backends, a credential, a cluster CA, a client certificate and a route.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + priority=0, + ), + v1alpha1.Endpoint( + name="together", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "together"}), + priority=1, + ), + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources(items=[_composed_endpoint(model="d")]), + "endpoints-together": fnv1.Resources( + items=[ + _third_party_endpoint( + name="together", + origin="https://api.together.xyz", + model="Qwen/Qwen2.5", + schema="OpenAI", + api_key_secret="together-key", + ready=True, + ) + ] + ), + "credential-together": fnv1.Resources( + items=[_api_key_secret(name="together-key", data={"apiKey": "c2stdG9n"})] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=2, ready_endpoints=2), + resources={ + "backend-self": _composed_endpoint_backend(), + "aibackend-self": _composed_endpoint_ai_backend(), + "backend-together": _third_party_backend( + name="assistant-eu-together-20044", hostname="api.together.xyz" + ), + "aibackend-together": _third_party_ai_backend(name="assistant-eu-together-20044", schema="OpenAI"), + "credential-together": _credential( + name="assistant-eu-together-credential-fe51d", api_key="c2stdG9n" + ), + "credpolicy-together": _credential_policy( + name="assistant-eu-together-20044", + auth={ + "type": "APIKey", + "apiKey": {"secretRef": {"name": "assistant-eu-together-credential-fe51d"}}, + }, + ), + "cluster-ca-gw-eu": _cluster_ca(), + "client-certificate": _client_certificate(), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[ + {"name": "assistant-eu-self-a42e6", "weight": 1, "priority": 0, "modelNameOverride": "d"}, + { + "name": "assistant-eu-together-20044", + "weight": 1, + "priority": 1, + "modelNameOverride": "Qwen/Qwen2.5", + }, + ], + ), }, - fn.CONDITION_REASON_NO_ENDPOINTS, - "None of the 1 selected ModelEndpoints is ready to carry traffic", - _requirements([_entry("a")], credentials={"wrongkey": "k"}), - warning=( - "Endpoints left out of the route, their credential Secret missing or missing its key: wrongkey" + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + "endpoints-together": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "together"}), + ), + "credential-together": fnv1.ResourceSelector( + api_version="v1", kind="Secret", namespace="ml-team", match_name="together-key" + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + Case( + name="GatewayWithoutTLS", + reason="On a gateway without TLS, a ModelRoute binds its AIGatewayRoute to the HTTP listener.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources(items=[_composed_endpoint(model=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=1), + resources={ + "backend-self": _composed_endpoint_backend(), + "aibackend-self": _composed_endpoint_ai_backend(), + "cluster-ca-gw-eu": _cluster_ca(), + "client-certificate": _client_certificate(), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[{"name": "assistant-eu-self-a42e6", "weight": 1, "priority": 0}], + ), + }, ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], ), - ] - - -@pytest.mark.parametrize("case", _gates_cases(), ids=lambda case: case.name) -def test_gates(case: Case) -> None: - """A pass where a route can't be composed composes nothing and says why.""" - got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) - - -def test_compose() -> None: - """A composed endpoint and a third-party provider get backends, a credential, a cluster CA and a route.""" - # A composed self-hosted endpoint at priority 0 and a third-party provider - # at priority 1. - entries = [_entry("d", priority=0), _entry("together", priority=1)] - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-d": [ - _endpoint("self", origin="https://gw-eu.example.com", model="d", composed=True), - ], - "endpoints-together": [ - _endpoint( - "together", - origin="https://api.together.xyz", - model="Qwen/Qwen2.5", - credential="together-key", - ), - ], - "credential-together": [_secret("together-key", {"apiKey": "sk-tog"})], + ), + # A TLS gateway serves inference on its HTTPS listener alone. Binding to :80 + # on a TLS gateway would carry credentials in the clear. + Case( + name="GatewayWithTLS", + reason="On a TLS gateway, a ModelRoute binds its AIGatewayRoute to the HTTPS listener.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=True, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources(items=[_composed_endpoint(model=None)]), }, ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - # The exact set, so an unexpected extra object fails the test. The - # mirrored namespace isn't here: compose-inference-cluster composes it. - assert set(got.desired.resources) == { - "client-certificate", - "backend-self", - "aibackend-self", - "backend-together", - "aibackend-together", - "credential-together", - "credpolicy-together", - "cluster-ca-gw-eu", - "route", - } - - route = _manifest(got, "route") - # The timeouts are the ModelRoute's, which _route_xr sets to values - # other than the ModelService's defaults. - assert route["spec"]["rules"] == [ - { - "matches": [{"headers": [{"type": "Exact", "name": "x-ai-eg-model", "value": _MODEL}]}], - "backendRefs": [ - {"name": _be("self"), "weight": 1, "priority": 0, "modelNameOverride": "d"}, - { - "name": _be("together"), - "weight": 1, - "priority": 1, - "modelNameOverride": "Qwen/Qwen2.5", + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=1), + resources={ + "backend-self": _composed_endpoint_backend(), + "aibackend-self": _composed_endpoint_ai_backend(), + "cluster-ca-gw-eu": _cluster_ca(), + "client-certificate": _client_certificate(), + "route": _ai_gateway_route( + section_name="https", + backend_refs=[{"name": "assistant-eu-self-a42e6", "weight": 1, "priority": 0}], + ), }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) ], - "timeouts": {"request": "600s"}, - "streamIdleTimeout": "0s", - "modelsOwnedBy": _NS, - } - ] - # Declaring the token costs is what makes the ext-proc ask a backend for - # usage on a streamed response, which otherwise reports none, and is - # where the metered counts in the access log come from. - assert route["spec"]["llmRequestCosts"] == [ - {"metadataKey": "llm_input_token", "type": "InputToken"}, - {"metadataKey": "llm_output_token", "type": "OutputToken"}, - {"metadataKey": "llm_total_token", "type": "TotalToken"}, - ] - - # The composed backend pins its cluster's CA and presents the client - # certificate, both this route's own; the third-party one uses the - # system trust store. - assert _manifest(got, "backend-self")["spec"]["tls"] == { - "caCertificateRefs": [{"kind": "ConfigMap", "group": "", "name": "assistant-eu-gw-eu-ca-3dd16"}], - "sni": "gw-eu.example.com", - "clientCertificateRef": {"kind": "Secret", "group": "", "name": "assistant-eu-client-08324"}, - } - assert _manifest(got, "backend-together")["spec"]["tls"] == { - "wellKnownCACertificates": "System", - "sni": "api.together.xyz", - } - - # The caller header is stripped only for the backend we don't operate. - assert "headerMutation" not in _manifest(got, "aibackend-self")["spec"] - assert _manifest(got, "aibackend-together")["spec"]["headerMutation"] == {"remove": ["x-modelplane-caller"]} - - # The credential is republished under the fixed apiKey key. - assert _manifest(got, "credential-together")["data"] == {"apiKey": base64.b64encode(b"sk-tog").decode()} - # Named for this route, so no other route in the namespace composes it. - assert _manifest(got, "cluster-ca-gw-eu") == { - "apiVersion": "v1", - "kind": "ConfigMap", - "metadata": {"name": "assistant-eu-gw-eu-ca-3dd16", "namespace": "mp-ml-team-51733"}, - "data": {"ca.crt": _CLUSTER_CA}, - } - - # Every composed object lands in the namespace mirroring the route's own, - # which compose-inference-cluster composes. - for key in ("backend-self", "backend-together", "credential-together", "cluster-ca-gw-eu", "route"): - assert _manifest(got, key)["metadata"]["namespace"] == "mp-ml-team-51733", key - - # The route lives in the team namespace but attaches across to the gateway. - assert route["spec"]["parentRefs"][0]["namespace"] == "modelplane-system" - - # The client certificate the backends present, issued from the gateway's - # CA ClusterIssuer into this namespace. Named for this route, so no other - # route in the namespace composes it, and deleted with the route. - assert ( - "managementPolicies" - not in resource.struct_to_dict(got.desired.resources["client-certificate"].resource)["spec"] - ) - assert _manifest(got, "client-certificate") == { - "apiVersion": "cert-manager.io/v1", - "kind": "Certificate", - "metadata": {"name": "assistant-eu-client-08324", "namespace": "mp-ml-team-51733"}, - "spec": { - "secretName": "assistant-eu-client-08324", - "commonName": "inference-gateway-eu", - "usages": ["client auth", "digital signature", "key encipherment"], - "duration": "2160h", - "renewBefore": "720h", - "privateKey": {"algorithm": "ECDSA", "size": 256, "rotationPolicy": "Always"}, - "issuerRef": {"name": "inference-gateway-ca", "kind": "ClusterIssuer", "group": "cert-manager.io"}, - }, - } - - -@pytest.mark.parametrize(("tls", "want"), [(False, "http"), (True, "https")]) -def test_route_binds_to_the_listener_matching_the_gateways_tls(tls: bool, want: str) -> None: - """The route binds to the HTTPS listener on a TLS gateway, and to the HTTP listener otherwise.""" - # A TLS gateway serves inference on its HTTPS listener alone, so the route - # binds there; without TLS there's only the HTTP listener. Binding to :80 on - # a TLS gateway would carry credentials in the clear. - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")])))), - required_resources=_required( - gateway=[_gateway(tls=tls)], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", composed=True)]}, ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert _manifest(got, "route")["spec"]["parentRefs"][0]["sectionName"] == want - - -def test_status_reports_address_and_counts() -> None: - """The status reports the model name, the gateway's address, and the endpoint counts.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")])))), - required_resources=_required( - gateway=[_gateway(address="203.0.113.9")], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", composed=True)]}, + ), + # This is GatewayWithoutTLS with the gateway at another address. Every other + # case's gateway is at 203.0.113.1, so only this one shows the status takes + # the gateway's address rather than a fixed one. + Case( + name="GatewayAddress", + reason="A ModelRoute reports its gateway's address and its endpoint counts in its status.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.9", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources(items=[_composed_endpoint(model=None)]), + }, ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert resource.struct_to_dict(got.desired.composite.resource)["status"] == { - "model": _MODEL, - "address": "203.0.113.9", - "endpoints": {"total": 1, "ready": 1}, - } - - -def test_an_endpoint_matched_twice_belongs_to_the_first_entry() -> None: - """An endpoint two entries match belongs to the first of them.""" - # A canary entry and a catch-all entry must not both weight one endpoint; - # the first that matches it wins. - entries = [_entry("kimi", name="canary", priority=0), _entry("kimi", name="catchall", priority=1)] - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-canary": [_endpoint("kimi-a", origin="https://a.example.com")], - "endpoints-catchall": [_endpoint("kimi-a", origin="https://a.example.com")], + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.9", total_endpoints=1, ready_endpoints=1), + resources={ + "backend-self": _composed_endpoint_backend(), + "aibackend-self": _composed_endpoint_ai_backend(), + "cluster-ca-gw-eu": _cluster_ca(), + "client-certificate": _client_certificate(), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[{"name": "assistant-eu-self-a42e6", "weight": 1, "priority": 0}], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + # A canary entry and a catch-all entry must not both weight one endpoint. The + # endpoint isn't Modelplane-composed, so no client certificate is issued. + Case( + name="EndpointMatchedTwice", + reason="An endpoint two entries both select gets one backendRef, not one per entry.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="canary", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "kimi"}), + priority=0, + ), + v1alpha1.Endpoint( + name="catchall", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "kimi"}), + priority=1, + ), + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-canary": fnv1.Resources( + items=[ + _third_party_endpoint( + name="kimi-a", + origin="https://a.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ) + ] + ), + "endpoints-catchall": fnv1.Resources( + items=[ + _third_party_endpoint( + name="kimi-a", + origin="https://a.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ) + ] + ), }, ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - # Neither endpoint is Modelplane-composed, so no client certificate is - # issued. - assert "client-certificate" not in got.desired.resources - refs = _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] - assert refs == [{"name": _be("kimi-a"), "weight": 1, "priority": 0}] - assert resource.struct_to_dict(got.desired.composite.resource)["status"]["endpoints"] == {"total": 1, "ready": 1} - - -def test_priorities_are_renumbered_without_gaps() -> None: - """The priorities of the tiers that have a ready endpoint are renumbered from 0, without gaps.""" + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=1), + resources={ + "backend-kimi-a": _third_party_backend(name="assistant-eu-kimi-a-bf6de", hostname="a.example.com"), + "aibackend-kimi-a": _third_party_ai_backend(name="assistant-eu-kimi-a-bf6de", schema="OpenAI"), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[{"name": "assistant-eu-kimi-a-bf6de", "weight": 1, "priority": 0}], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-canary": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "kimi"}), + ), + "endpoints-catchall": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "kimi"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), # A ModelService's priorities are an ordering; Envoy's are levels it walks - # from 0. A user writing 0 and 5, or a tier gone unready during a roll, - # would otherwise leave gaps in what Envoy gets. - entries = [_entry("a", priority=0), _entry("b", priority=5), _entry("c", priority=9)] - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - # The middle tier has no ready endpoint, so it drops out and - # must not leave a hole behind it. - "endpoints-a": [_endpoint("a-0", origin="https://a.example.com")], - "endpoints-b": [_endpoint("b-0", origin="https://b.example.com", ready=False)], - "endpoints-c": [_endpoint("c-0", origin="https://c.example.com")], + # from 0. A user writing 0 and 5, or a tier gone unready during a roll, would + # otherwise leave gaps in what Envoy gets. + Case( + name="PriorityGaps", + reason="Entries at priorities 0, 5 and 9, the middle one with no ready endpoint, compose backendRefs at priorities 0 and 1.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="a", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "a"}), + priority=0, + ), + v1alpha1.Endpoint( + name="b", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "b"}), + priority=5, + ), + v1alpha1.Endpoint( + name="c", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "c"}), + priority=9, + ), + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-a": fnv1.Resources( + items=[ + _third_party_endpoint( + name="a-0", + origin="https://a.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ) + ] + ), + "endpoints-b": fnv1.Resources( + items=[ + _third_party_endpoint( + name="b-0", + origin="https://b.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=False, + ) + ] + ), + "endpoints-c": fnv1.Resources( + items=[ + _third_party_endpoint( + name="c-0", + origin="https://c.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ) + ] + ), }, ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - refs = _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] - assert [r["priority"] for r in refs] == [0, 1], "two tiers survive, renumbered 0 and 1" - - -@dataclasses.dataclass -class WeightCase: - """A weight-distribution case: the entries and the endpoints each matched, - and the whole backendRefs list the route should carry.""" - - name: str - entries: list[v1alpha1.Endpoint] - endpoints: dict[str, list[dict]] - want_refs: list[dict] - - -def _weight_distribution_cases() -> list[WeightCase]: - def _origins(*names_: str) -> list[dict]: - return [_endpoint(n, origin=f"https://{n}.example.com") for n in names_] - - def _ref(ep: str, weight: int, priority: int = 0) -> dict: - return {"name": _be(ep), "weight": weight, "priority": priority} - - return [ - WeightCase( - # An entry's weight is written once but applied per backend, so it - # spreads over the endpoints it matched while the ratio between - # entries survives: 90 over three is 30 each, 10 over one is 10, - # reduced by the gcd to the smallest equivalent integers. - name="a weight spreads across a tier's endpoints, ratio preserved", - entries=[_entry("big", weight=90), _entry("small", weight=10)], - endpoints={ - "endpoints-big": _origins("big-0", "big-1", "big-2"), - "endpoints-small": _origins("small-0"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=3, ready_endpoints=2), + resources={ + "backend-a-0": _third_party_backend(name="assistant-eu-a-0-b66b8", hostname="a.example.com"), + "aibackend-a-0": _third_party_ai_backend(name="assistant-eu-a-0-b66b8", schema="OpenAI"), + "backend-c-0": _third_party_backend(name="assistant-eu-c-0-782db", hostname="c.example.com"), + "aibackend-c-0": _third_party_ai_backend(name="assistant-eu-c-0-782db", schema="OpenAI"), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[ + {"name": "assistant-eu-a-0-b66b8", "weight": 1, "priority": 0}, + {"name": "assistant-eu-c-0-782db", "weight": 1, "priority": 1}, + ], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-a": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "a"}), + ), + "endpoints-b": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "b"}), + ), + "endpoints-c": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "c"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + # An entry's weight is written once but applied per backend, so it spreads + # over the endpoints it matched while the ratio between entries survives: 90 + # over three is 30 each, 10 over one is 10, reduced by the gcd to the + # smallest equivalent integers. + Case( + name="WeightSpread", + reason="Entries weighted 90 over three endpoints and 10 over one compose backendRefs weighted 3, 3, 3 and 1.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="big", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "big"}), + weight=90, + ), + v1alpha1.Endpoint( + name="small", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "small"}), + weight=10, + ), + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-big": fnv1.Resources( + items=[ + _third_party_endpoint( + name="big-0", + origin="https://big-0.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ), + _third_party_endpoint( + name="big-1", + origin="https://big-1.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ), + _third_party_endpoint( + name="big-2", + origin="https://big-2.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ), + ] + ), + "endpoints-small": fnv1.Resources( + items=[ + _third_party_endpoint( + name="small-0", + origin="https://small-0.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ) + ] + ), }, - want_refs=[ - _ref("big-0", 3), - _ref("big-1", 3), - _ref("big-2", 3), - _ref("small-0", 1), - ], - ), - WeightCase( - # Weight 1 over five endpoints must floor none of them to 0, which - # would drop them from the load assignment rather than share. - name="a weight below its endpoint count floors no endpoint", - entries=[_entry("many", weight=1)], - endpoints={"endpoints-many": _origins("many-0", "many-1", "many-2", "many-3", "many-4")}, - want_refs=[_ref(f"many-{i}", 1) for i in range(5)], - ), - WeightCase( - # A max-weight entry beside a tiny one spread over two endpoints - # scales past the per-backendRef limit even though every weight is - # in bounds, so it rescales to the limit rather than composing a - # route the API server rejects. - name="an extreme but valid ratio is clamped to the limit", - entries=[_entry("big", weight=1000000, priority=0), _entry("small", weight=1, priority=0)], - endpoints={"endpoints-big": _origins("big-0"), "endpoints-small": _origins("small-0", "small-1")}, - want_refs=[_ref("big-0", 1000000), _ref("small-0", 1), _ref("small-1", 1)], - ), - WeightCase( - # The remainder is handed to the first endpoints of a tier, so the - # order must be the endpoints' names rather than the API server's - # unspecified list order, or the composed weights churn. - name="endpoints are ordered by name for a stable split", - entries=[_entry("d")], - endpoints={"endpoints-d": _origins("z", "a", "m")}, - want_refs=[_ref("a", 1), _ref("m", 1), _ref("z", 1)], - ), - ] - - -@pytest.mark.parametrize("case", _weight_distribution_cases(), ids=lambda case: case.name) -def test_weight_distribution(case: WeightCase) -> None: - """Each entry's weight is distributed across the endpoints it matched.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(case.entries)))), - required_resources=_required(gateway=[_gateway()], clusters=[_cluster("gw-eu")], **case.endpoints), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] == case.want_refs - - -@dataclasses.dataclass -class CredentialCase: - """A credential case: the API a keyed backend speaks, and the whole - BackendSecurityPolicy composed for it.""" - - name: str - api: mev1alpha1.Api | None - want: dict - - -def _credential_policy_cases() -> list[CredentialCase]: - secret = resource.child_name(f"{_SVC}-{_GW}", "provider", "credential") - target = {"group": "aigateway.envoyproxy.io", "kind": "AIServiceBackend", "name": _be("provider")} - return [ - CredentialCase( - name="a backend speaking OpenAI's API gets the key as a bearer token", - api=None, - want={ - "apiVersion": "aigateway.envoyproxy.io/v1beta1", - "kind": "BackendSecurityPolicy", - "metadata": {"name": _be("provider"), "namespace": "mp-ml-team-51733"}, - "spec": { - "type": "APIKey", - "apiKey": {"secretRef": {"name": secret}}, - "targetRefs": [target], + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=4, ready_endpoints=4), + resources={ + "backend-big-0": _third_party_backend( + name="assistant-eu-big-0-1e944", hostname="big-0.example.com" + ), + "aibackend-big-0": _third_party_ai_backend(name="assistant-eu-big-0-1e944", schema="OpenAI"), + "backend-big-1": _third_party_backend( + name="assistant-eu-big-1-91055", hostname="big-1.example.com" + ), + "aibackend-big-1": _third_party_ai_backend(name="assistant-eu-big-1-91055", schema="OpenAI"), + "backend-big-2": _third_party_backend( + name="assistant-eu-big-2-5c954", hostname="big-2.example.com" + ), + "aibackend-big-2": _third_party_ai_backend(name="assistant-eu-big-2-5c954", schema="OpenAI"), + "backend-small-0": _third_party_backend( + name="assistant-eu-small-0-60d20", hostname="small-0.example.com" + ), + "aibackend-small-0": _third_party_ai_backend(name="assistant-eu-small-0-60d20", schema="OpenAI"), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[ + {"name": "assistant-eu-big-0-1e944", "weight": 3, "priority": 0}, + {"name": "assistant-eu-big-1-91055", "weight": 3, "priority": 0}, + {"name": "assistant-eu-big-2-5c954", "weight": 3, "priority": 0}, + {"name": "assistant-eu-small-0-60d20", "weight": 1, "priority": 0}, + ], + ), }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-big": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "big"}), + ), + "endpoints-small": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "small"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + # Weight 1 over five endpoints must floor none of them to 0, which would drop + # them from the load assignment rather than share. + Case( + name="WeightBelowCount", + reason="An entry weighted 1 over five endpoints composes a backendRef weighted 1 for each of them.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="many", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "many"}), + weight=1, + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-many": fnv1.Resources( + items=[ + _third_party_endpoint( + name="many-0", + origin="https://many-0.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ), + _third_party_endpoint( + name="many-1", + origin="https://many-1.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ), + _third_party_endpoint( + name="many-2", + origin="https://many-2.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ), + _third_party_endpoint( + name="many-3", + origin="https://many-3.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ), + _third_party_endpoint( + name="many-4", + origin="https://many-4.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ), + ] + ), }, ), - CredentialCase( - name="a backend speaking Anthropic's API gets the key in x-api-key", - api=mev1alpha1.Api(schema="Anthropic"), - want={ - "apiVersion": "aigateway.envoyproxy.io/v1beta1", - "kind": "BackendSecurityPolicy", - "metadata": {"name": _be("provider"), "namespace": "mp-ml-team-51733"}, - "spec": { - "type": "AnthropicAPIKey", - "anthropicAPIKey": {"secretRef": {"name": secret}}, - "targetRefs": [target], + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=5, ready_endpoints=5), + resources={ + "backend-many-0": _third_party_backend( + name="assistant-eu-many-0-e4e60", hostname="many-0.example.com" + ), + "aibackend-many-0": _third_party_ai_backend(name="assistant-eu-many-0-e4e60", schema="OpenAI"), + "backend-many-1": _third_party_backend( + name="assistant-eu-many-1-b8a15", hostname="many-1.example.com" + ), + "aibackend-many-1": _third_party_ai_backend(name="assistant-eu-many-1-b8a15", schema="OpenAI"), + "backend-many-2": _third_party_backend( + name="assistant-eu-many-2-a7a3b", hostname="many-2.example.com" + ), + "aibackend-many-2": _third_party_ai_backend(name="assistant-eu-many-2-a7a3b", schema="OpenAI"), + "backend-many-3": _third_party_backend( + name="assistant-eu-many-3-db5a5", hostname="many-3.example.com" + ), + "aibackend-many-3": _third_party_ai_backend(name="assistant-eu-many-3-db5a5", schema="OpenAI"), + "backend-many-4": _third_party_backend( + name="assistant-eu-many-4-bd6f9", hostname="many-4.example.com" + ), + "aibackend-many-4": _third_party_ai_backend(name="assistant-eu-many-4-bd6f9", schema="OpenAI"), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[ + {"name": "assistant-eu-many-0-e4e60", "weight": 1, "priority": 0}, + {"name": "assistant-eu-many-1-b8a15", "weight": 1, "priority": 0}, + {"name": "assistant-eu-many-2-a7a3b", "weight": 1, "priority": 0}, + {"name": "assistant-eu-many-3-db5a5", "weight": 1, "priority": 0}, + {"name": "assistant-eu-many-4-bd6f9", "weight": 1, "priority": 0}, + ], + ), }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-many": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "many"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + # A max-weight entry beside a tiny one spread over two endpoints scales past + # the per-backendRef limit even though every weight is in bounds, so it + # rescales to the limit rather than composing a route the API server rejects. + Case( + name="ExtremeRatio", + reason="Entries weighted 1000000 over one endpoint and 1 over two compose backendRefs clamped to the weight limit: 1000000, 1 and 1.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="big", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "big"}), + priority=0, + weight=1000000, + ), + v1alpha1.Endpoint( + name="small", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "small"}), + priority=0, + weight=1, + ), + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-big": fnv1.Resources( + items=[ + _third_party_endpoint( + name="big-0", + origin="https://big-0.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ) + ] + ), + "endpoints-small": fnv1.Resources( + items=[ + _third_party_endpoint( + name="small-0", + origin="https://small-0.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ), + _third_party_endpoint( + name="small-1", + origin="https://small-1.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ), + ] + ), }, ), - ] - - -@pytest.mark.parametrize("case", _credential_policy_cases(), ids=lambda case: case.name) -def test_credential_policy(case: CredentialCase) -> None: - """A keyed backend's BackendSecurityPolicy sends the key the way its API expects.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")])))), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-d": [ - _endpoint( - "provider", - origin="https://api.example.com", - api=case.api, - credential="provider-key", - ) - ], - "credential-provider": [_secret("provider-key", {"apiKey": "sk-provider"})], + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=3, ready_endpoints=3), + resources={ + "backend-big-0": _third_party_backend( + name="assistant-eu-big-0-1e944", hostname="big-0.example.com" + ), + "aibackend-big-0": _third_party_ai_backend(name="assistant-eu-big-0-1e944", schema="OpenAI"), + "backend-small-0": _third_party_backend( + name="assistant-eu-small-0-60d20", hostname="small-0.example.com" + ), + "aibackend-small-0": _third_party_ai_backend(name="assistant-eu-small-0-60d20", schema="OpenAI"), + "backend-small-1": _third_party_backend( + name="assistant-eu-small-1-cad94", hostname="small-1.example.com" + ), + "aibackend-small-1": _third_party_ai_backend(name="assistant-eu-small-1-cad94", schema="OpenAI"), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[ + {"name": "assistant-eu-big-0-1e944", "weight": 1000000, "priority": 0}, + {"name": "assistant-eu-small-0-60d20", "weight": 1, "priority": 0}, + {"name": "assistant-eu-small-1-cad94", "weight": 1, "priority": 0}, + ], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-big": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "big"}), + ), + "endpoints-small": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "small"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + # The remainder is handed to the first endpoints of a tier, so the order must + # be the endpoints' names rather than the API server's unspecified list order, + # or the composed weights churn. + Case( + name="UnsortedEndpoints", + reason="Endpoints listed as z, a and m compose backendRefs in name order: a, m, z.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources( + items=[ + _third_party_endpoint( + name="z", + origin="https://z.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ), + _third_party_endpoint( + name="a", + origin="https://a.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ), + _third_party_endpoint( + name="m", + origin="https://m.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + ), + ] + ), }, ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert _manifest(got, "credpolicy-provider") == case.want + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=3, ready_endpoints=3), + resources={ + "backend-a": _third_party_backend(name="assistant-eu-a-18a56", hostname="a.example.com"), + "aibackend-a": _third_party_ai_backend(name="assistant-eu-a-18a56", schema="OpenAI"), + "backend-m": _third_party_backend(name="assistant-eu-m-4d1e4", hostname="m.example.com"), + "aibackend-m": _third_party_ai_backend(name="assistant-eu-m-4d1e4", schema="OpenAI"), + "backend-z": _third_party_backend(name="assistant-eu-z-e5d6f", hostname="z.example.com"), + "aibackend-z": _third_party_ai_backend(name="assistant-eu-z-e5d6f", schema="OpenAI"), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[ + {"name": "assistant-eu-a-18a56", "weight": 1, "priority": 0}, + {"name": "assistant-eu-m-4d1e4", "weight": 1, "priority": 0}, + {"name": "assistant-eu-z-e5d6f", "weight": 1, "priority": 0}, + ], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + Case( + name="OpenAIAPIKey", + reason="A third-party endpoint speaking OpenAI's API gets a credential policy that sends its key as a bearer token.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources( + items=[ + _third_party_endpoint( + name="provider", + origin="https://api.example.com", + model=None, + schema="OpenAI", + api_key_secret="provider-key", + ready=True, + ) + ] + ), + "credential-provider": fnv1.Resources( + items=[_api_key_secret(name="provider-key", data={"apiKey": "c2stcHJvdmlkZXI="})] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=1), + resources={ + "backend-provider": _third_party_backend( + name="assistant-eu-provider-348cb", hostname="api.example.com" + ), + "aibackend-provider": _third_party_ai_backend(name="assistant-eu-provider-348cb", schema="OpenAI"), + "credential-provider": _credential( + name="assistant-eu-provider-credential-4d66b", api_key="c2stcHJvdmlkZXI=" + ), + "credpolicy-provider": _credential_policy( + name="assistant-eu-provider-348cb", + auth={ + "type": "APIKey", + "apiKey": {"secretRef": {"name": "assistant-eu-provider-credential-4d66b"}}, + }, + ), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[{"name": "assistant-eu-provider-348cb", "weight": 1, "priority": 0}], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + "credential-provider": fnv1.ResourceSelector( + api_version="v1", kind="Secret", namespace="ml-team", match_name="provider-key" + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + Case( + name="AnthropicAPIKey", + reason="A third-party endpoint speaking Anthropic's API gets a credential policy that sends its key in x-api-key.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources( + items=[ + _third_party_endpoint( + name="provider", + origin="https://api.example.com", + model=None, + schema="Anthropic", + api_key_secret="provider-key", + ready=True, + ) + ] + ), + "credential-provider": fnv1.Resources( + items=[_api_key_secret(name="provider-key", data={"apiKey": "c2stcHJvdmlkZXI="})] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=1), + resources={ + "backend-provider": _third_party_backend( + name="assistant-eu-provider-348cb", hostname="api.example.com" + ), + "aibackend-provider": _third_party_ai_backend( + name="assistant-eu-provider-348cb", schema="Anthropic" + ), + "credential-provider": _credential( + name="assistant-eu-provider-credential-4d66b", api_key="c2stcHJvdmlkZXI=" + ), + "credpolicy-provider": _credential_policy( + name="assistant-eu-provider-348cb", + auth={ + "type": "AnthropicAPIKey", + "anthropicAPIKey": {"secretRef": {"name": "assistant-eu-provider-credential-4d66b"}}, + }, + ), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[{"name": "assistant-eu-provider-348cb", "weight": 1, "priority": 0}], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + "credential-provider": fnv1.ResourceSelector( + api_version="v1", kind="Secret", namespace="ml-team", match_name="provider-key" + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), +] + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes a ModelRoute's backends and AIGatewayRoute, or composes nothing and says why.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-model-service/tests/test_fn.py b/functions/compose-model-service/tests/test_fn.py index 4fb2975b9..55c5a27e6 100644 --- a/functions/compose-model-service/tests/test_fn.py +++ b/functions/compose-model-service/tests/test_fn.py @@ -25,13 +25,8 @@ from google.protobuf import duration_pb2 as durationpb from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb -from models.ai.modelplane.inferencegateway import v1alpha1 as igv1alpha1 -from models.ai.modelplane.modelroute import v1alpha1 as mrtv1alpha1 from models.ai.modelplane.modelservice import v1alpha1 - -_NS = "ml-team" -_SVC = "assistant" -_MODEL = f"{_NS}/{_SVC}" +from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @dataclasses.dataclass @@ -39,379 +34,434 @@ class Case: """A test case for compose-model-service.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse -def _entry(deployment: str, *, priority: int | None = None, weight: int | None = None) -> v1alpha1.Endpoint: - kwargs = {} - if priority is not None: - kwargs["priority"] = priority - if weight is not None: - kwargs["weight"] = weight - return v1alpha1.Endpoint( - name=deployment, - selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": deployment}), - **kwargs, - ) - - -def _service(entries: list[v1alpha1.Endpoint], labels: dict[str, str] | None = None) -> dict: - xr = v1alpha1.ModelService( - apiVersion="modelplane.ai/v1alpha1", - kind="ModelService", - metadata={"name": _SVC, "namespace": _NS, **({"labels": labels} if labels else {})}, - # Not the defaults, so a route carrying the defaults fails to match. - spec=v1alpha1.Spec(endpoints=entries, timeouts=v1alpha1.Timeouts(request="600s", idle="0s")), +def _model_service(*, labels: dict[str, str] | None) -> fnv1.Resource: + """The ModelService XR assistant in ml-team, serving kimi-k2.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ModelService( + apiVersion="modelplane.ai/v1alpha1", + kind="ModelService", + metadata=metav1.ObjectMeta(name="assistant", namespace="ml-team", labels=labels), + spec=v1alpha1.Spec( + endpoints=[ + v1alpha1.Endpoint( + name="kimi-k2", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "kimi-k2"}), + ) + ], + timeouts=v1alpha1.Timeouts(request="600s", idle="0s"), + ), + ).model_dump(exclude_none=True, mode="json", by_alias=True) + ) ) - return xr.model_dump(exclude_none=True, mode="json", by_alias=True) -def _gateway(name: str, cluster: str, *, selector: dict[str, str] | None = None, address: str | None = None) -> dict: - gw = igv1alpha1.InferenceGateway( - apiVersion="modelplane.ai/v1alpha1", - kind="InferenceGateway", - metadata={"name": name}, - spec=igv1alpha1.Spec( - clusterName=cluster, - **({"serviceSelector": igv1alpha1.ServiceSelector(matchLabels=selector)} if selector else {}), +def _desired_model_service(*, total_routes: int, ready_routes: int, ready: fnv1.Ready) -> fnv1.Resource: + """The desired ModelService XR, reporting its model and how many of its routes are ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"model": "ml-team/assistant", "routes": {"total": total_routes, "ready": ready_routes}}} ), + ready=ready, ) - d = gw.model_dump(exclude_none=True, mode="json", by_alias=True) - if address: - d["status"] = {"address": address} - return d -def _route(gateway: str, cluster: str) -> dict: - """The ModelRoute the function composes for one gateway, as a plain dict. +def _inference_gateway( + *, name: str, cluster_name: str, service_selector: dict | None, address: str | None +) -> fnv1.Resource: + """An InferenceGateway, as the gateways requirement returns it.""" + spec: dict = {"clusterName": cluster_name} + if service_selector is not None: + spec["serviceSelector"] = service_selector + gateway: dict = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceGateway", + "metadata": {"name": name}, + "spec": spec, + } + if address is not None: + gateway["status"] = {"address": address} + return fnv1.Resource(resource=resource.dict_to_struct(gateway)) - The endpoints are spelled out rather than derived from the service's, so a - bug in the copy the function does can't hide in an expectation computed the - same way. Every case here drives the single kimi-k2 entry, which the copy - fills to its priority/weight defaults. The cluster label carries the gateway's - clusterName, which compose-inference-cluster selects routes by. - """ - route = mrtv1alpha1.ModelRoute( - apiVersion="modelplane.ai/v1alpha1", - kind="ModelRoute", - metadata={ - "name": resource.child_name(_SVC, gateway), - "namespace": _NS, - "labels": { - "modelplane.ai/service": _SVC, - "modelplane.ai/gateway": gateway, - "modelplane.ai/cluster": cluster, - }, - }, - spec=mrtv1alpha1.Spec( - gatewayName=gateway, - serviceName=_SVC, - endpoints=[ - mrtv1alpha1.Endpoint( - name="kimi-k2", - selector=mrtv1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "kimi-k2"}), - priority=0, - weight=1, - ) - ], - timeouts=mrtv1alpha1.Timeouts(request="600s", idle="0s"), + +def _model_route(*, name: str, gateway: str, cluster: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed ModelRoute pinning assistant to gateway on cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelRoute", + "metadata": { + "name": name, + "namespace": "ml-team", + "labels": { + "modelplane.ai/service": "assistant", + "modelplane.ai/gateway": gateway, + "modelplane.ai/cluster": cluster, + }, + }, + "spec": { + "gatewayName": gateway, + "serviceName": "assistant", + "endpoints": [ + { + "name": "kimi-k2", + "selector": {"matchLabels": {"modelplane.ai/deployment": "kimi-k2"}}, + "priority": 0, + "weight": 1, + } + ], + "timeouts": {"request": "600s", "idle": "0s"}, + }, + } ), + ready=ready, ) - return route.model_dump(exclude_none=True, mode="json", by_alias=True) - - -def _observed_route(gateway: str, ready: bool) -> fnv1.Resource: - """A composed ModelRoute as observed back, Ready or not.""" - d = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelRoute", - "metadata": {"name": resource.child_name(_SVC, gateway), "namespace": _NS}, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True" if ready else "False", - "reason": "Available" if ready else "Creating", - "lastTransitionTime": "2026-06-08T00:00:00Z", - } - ] - }, - } - return fnv1.Resource(resource=resource.dict_to_struct(d)) -def _required(**resources) -> dict: # noqa: ANN003 - return { - name: fnv1.Resources(items=[fnv1.Resource(resource=resource.dict_to_struct(r)) for r in items]) - for name, items in resources.items() - } +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) -def _compose_cases() -> list[Case]: - entries = [_entry("kimi-k2")] - return [ - Case( - name="gateways not resolved yet: require them and wait", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries)))), +# Every ModelService here sets timeouts that aren't the defaults, so a route +# carrying the defaults fails to match. Each composed ModelRoute's cluster label +# carries its gateway's clusterName, which compose-inference-cluster selects +# routes by. +COMPOSE_CASES = [ + Case( + name="GatewaysUnresolved", + reason="Until Crossplane resolves the InferenceGateways, a ModelService requires them and reports WaitingForGateways.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_model_service(labels=None)), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_service(total_routes=0, ready_routes=0, ready=fnv1.READY_FALSE) ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 0, "ready": 0}}} - ), - ready=fnv1.READY_FALSE, - ) - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" - ), - } - ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_WAITING_FOR_GATEWAYS, - message="Waiting for the gateways to resolve", - ) - ], - results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the gateways to resolve")], + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the gateways to resolve")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + } ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateways", + message="Waiting for the gateways to resolve", + ) + ], ), - Case( - name="no gateway selects the service: unreachable, and say so", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_service(entries, labels={"region": "us"})) - ) - ), - required_resources=_required(gateways=[_gateway("eu", "gw-eu", selector={"region": "eu"})]), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 0, "ready": 0}}} - ), - ready=fnv1.READY_FALSE, - ) - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" + ), + Case( + name="NoGatewaySelects", + reason="A ModelService whose labels match no gateway's serviceSelector composes no route and reports NoGatewayServesThisService.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_model_service(labels={"region": "us"})), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="eu", + cluster_name="gw-eu", + service_selector={"matchLabels": {"region": "eu"}}, + address=None, ), - } + ] ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_NO_GATEWAY, - message=( - "No InferenceGateway's serviceSelector matches this service's labels, " - "so no caller can reach it" - ), - ) - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message=( - "No InferenceGateway's serviceSelector matches this service's labels, " - "so no caller can reach it" - ), - ) - ], - ), + }, ), - Case( - # A gateway with no address is left out of readiness, but with no - # other gateway there is nowhere a caller could reach the service. - name="its only gateway has no address yet: not RoutingReady", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - ), - required_resources=_required(gateways=[_gateway("eu", "gw-eu")]), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_service(total_routes=0, ready_routes=0, ready=fnv1.READY_FALSE) ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 1, "ready": 0}}} - ), - ready=fnv1.READY_FALSE, + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message=( + "No InferenceGateway's serviceSelector matches this service's labels, so no caller can reach it" ), - resources={ - "route-eu": fnv1.Resource(resource=resource.dict_to_struct(_route("eu", "gw-eu"))), - }, - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" - ), - } - ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_WAITING_FOR_GATEWAYS, - message="Waiting for gateways to come up: eu", - ) - ], - results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for gateways to come up: eu")], + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + } ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoGatewayServesThisService", + message=( + "No InferenceGateway's serviceSelector matches this service's labels, so no caller can reach it" + ), + ) + ], ), - Case( - name="two gateways serve it: a ModelRoute each, waiting for both routes", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - ), - required_resources=_required( - gateways=[ - _gateway("eu", "gw-eu", address="203.0.113.1"), - _gateway("us", "gw-us", address="203.0.113.2"), - ], + ), + # A gateway with no address is left out of readiness, but with no other + # gateway there is nowhere a caller could reach the service. + Case( + name="OnlyGatewayNoAddress", + reason="A ModelService whose only gateway has no address yet composes a ModelRoute for it but reports WaitingForGateways.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_model_service(labels=None)), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway(name="eu", cluster_name="gw-eu", service_selector=None, address=None), + ] ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_service(total_routes=1, ready_routes=0, ready=fnv1.READY_FALSE), + resources={ + "route-eu": _model_route( + name="assistant-eu-b0e0c", gateway="eu", cluster="gw-eu", ready=fnv1.READY_UNSPECIFIED + ), + }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 2, "ready": 0}}} + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for gateways to come up: eu")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateways", + message="Waiting for gateways to come up: eu", + ) + ], + ), + ), + Case( + name="RoutesNotReady", + reason="A ModelService that two gateways serve composes a ModelRoute for each and reports WaitingForRoutes while neither route is ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_model_service(labels=None)), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="eu", cluster_name="gw-eu", service_selector=None, address="203.0.113.1" ), - ready=fnv1.READY_FALSE, - ), - resources={ - "route-eu": fnv1.Resource(resource=resource.dict_to_struct(_route("eu", "gw-eu"))), - "route-us": fnv1.Resource(resource=resource.dict_to_struct(_route("us", "gw-us"))), - }, - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" + _inference_gateway( + name="us", cluster_name="gw-us", service_selector=None, address="203.0.113.2" ), - } + ] ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_WAITING_FOR_ROUTES, - message="Waiting for routes on gateways: eu, us", - ) - ], - results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for routes on gateways: eu, us")], - ), + }, ), - Case( - name="both routes accepted: ModelRoutes ready, service RoutingReady", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - resources={ - "route-eu": _observed_route("eu", ready=True), - "route-us": _observed_route("us", ready=True), - }, - ), - required_resources=_required( - gateways=[ - _gateway("eu", "gw-eu", address="203.0.113.1"), - _gateway("us", "gw-us", address="203.0.113.2"), - ], - ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_service(total_routes=2, ready_routes=0, ready=fnv1.READY_FALSE), + resources={ + "route-eu": _model_route( + name="assistant-eu-b0e0c", gateway="eu", cluster="gw-eu", ready=fnv1.READY_UNSPECIFIED + ), + "route-us": _model_route( + name="assistant-us-04e03", gateway="us", cluster="gw-us", ready=fnv1.READY_UNSPECIFIED + ), + }, + ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for routes on gateways: eu, us")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + } ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoutes", + message="Waiting for routes on gateways: eu, us", + ) + ], + ), + ), + Case( + name="BothRoutesReady", + reason="With both its ModelRoutes observed ready, a ModelService marks them ready and reports RoutingReady.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_service(labels=None), + resources={ + "route-eu": fnv1.Resource( resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 2, "ready": 2}}} - ), - ready=fnv1.READY_TRUE, + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelRoute", + "metadata": {"name": "assistant-eu-b0e0c", "namespace": "ml-team"}, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + } + ] + }, + } + ) ), - resources={ - "route-eu": fnv1.Resource( - resource=resource.dict_to_struct(_route("eu", "gw-eu")), ready=fnv1.READY_TRUE + "route-us": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelRoute", + "metadata": {"name": "assistant-us-04e03", "namespace": "ml-team"}, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + } + ] + }, + } + ) + ), + }, + ), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="eu", cluster_name="gw-eu", service_selector=None, address="203.0.113.1" ), - "route-us": fnv1.Resource( - resource=resource.dict_to_struct(_route("us", "gw-us")), ready=fnv1.READY_TRUE + _inference_gateway( + name="us", cluster_name="gw-us", service_selector=None, address="203.0.113.2" ), - }, + ] ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_service(total_routes=2, ready_routes=2, ready=fnv1.READY_TRUE), + resources={ + "route-eu": _model_route( + name="assistant-eu-b0e0c", gateway="eu", cluster="gw-eu", ready=fnv1.READY_TRUE + ), + "route-us": _model_route( + name="assistant-us-04e03", gateway="us", cluster="gw-us", ready=fnv1.READY_TRUE + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="RoutesAccepted", + ) + ], + ), + ), + # A gateway still coming up is excluded from readiness rather than failing it. + Case( + name="OneGatewayNoAddress", + reason="With one gateway's ModelRoute ready and the other gateway without an address, a ModelService reports RoutingReady and is ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_service(labels=None), + resources={ + "route-eu": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelRoute", + "metadata": {"name": "assistant-eu-b0e0c", "namespace": "ml-team"}, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + } + ] + }, + } + ) + ), + }, + ), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="eu", cluster_name="gw-eu", service_selector=None, address="203.0.113.1" ), - } + _inference_gateway(name="us", cluster_name="gw-us", service_selector=None, address=None), + ] ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_TRUE, - reason=fn.CONDITION_REASON_ROUTES_ACCEPTED, - ) - ], + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_service(total_routes=2, ready_routes=1, ready=fnv1.READY_TRUE), + resources={ + "route-eu": _model_route( + name="assistant-eu-b0e0c", gateway="eu", cluster="gw-eu", ready=fnv1.READY_TRUE + ), + "route-us": _model_route( + name="assistant-us-04e03", gateway="us", cluster="gw-us", ready=fnv1.READY_UNSPECIFIED + ), + }, ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="RoutesAccepted", + ) + ], ), - ] - - -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + ), +] -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction composes a ModelRoute per serving gateway and reports routing readiness.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) - - -def test_absent_selector_serves_every_service() -> None: - """A gateway with no serviceSelector serves the service, and one with no address doesn't block readiness.""" - # A gateway with no serviceSelector serves the service, and one still - # coming up (no address) is excluded from readiness rather than failing - # it: both RoutingReady and the service's own Ready ignore its unready - # ModelRoute. - entries = [_entry("kimi-k2")] - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - resources={"route-eu": _observed_route("eu", ready=True)}, - ), - required_resources=_required( - gateways=[ - _gateway("eu", "gw-eu", address="203.0.113.1"), - _gateway("us", "gw-us"), # no address: still coming up - ], - ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert "route-eu" in got.desired.resources - assert "route-us" in got.desired.resources - cond = next(c for c in got.conditions if c.type == fn.CONDITION_TYPE_ROUTING_READY) - assert cond.status == fnv1.STATUS_CONDITION_TRUE, "us has no address, so it doesn't block" - assert got.desired.composite.ready == fnv1.READY_TRUE, "nor does its unready ModelRoute" + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-nebius-cluster/tests/test_fn.py b/functions/compose-nebius-cluster/tests/test_fn.py index 74b131657..5f61cc045 100644 --- a/functions/compose-nebius-cluster/tests/test_fn.py +++ b/functions/compose-nebius-cluster/tests/test_fn.py @@ -17,6 +17,7 @@ import asyncio import dataclasses import json +from typing import Any import pytest from crossplane.function import resource @@ -34,275 +35,243 @@ class Case: """A test case for compose-nebius-cluster.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse -# The Nebius ClusterProviderConfig the function reads the credentials -# Secret off. -_NEBIUS_PROVIDER_CONFIG = { - "apiVersion": "nebius.m.upbound.io/v1beta1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "default"}, - "spec": { - "identity": {"type": "ServiceAccount"}, - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "crossplane-system", - "name": "nebius-credentials", - "key": "credentials.json", - }, - }, - "projectID": "project-e00test", - }, -} - -_PROVIDER_CONFIG_SELECTOR = fnv1.ResourceSelector( - api_version="nebius.m.upbound.io/v1beta1", - kind="ClusterProviderConfig", - match_name="default", -) - -# Name of the composed cloud-init Secret. Derived like the function derives -# it - the hash suffix depends only on the parent and child names. -_CLOUD_INIT_SECRET_NAME = resource.child_name("test-cluster", "cloud-init") - -# The cloud-init user data mounting the cache filesystem on every node. -_CLOUD_INIT = ( - "#cloud-config\n" - "runcmd:\n" - " - mkdir -p /mnt/data\n" - " - mount -t virtiofs modelplane-cache /mnt/data\n" - ' - printf "modelplane-cache /mnt/data virtiofs defaults,nofail 0 2\\n" >> /etc/fstab\n' -) - -# The cache filesystem attachment and cloud-init reference every node group -# template carries. -_TEMPLATE_CACHE_MOUNT = { - "filesystems": [ - { - "attachMode": "READ_WRITE", - "mountTag": "modelplane-cache", - "existingFilesystem": {"idSelector": {"matchControllerRef": True}}, - }, - ], - "cloudInitUserDataSecretRef": {"name": _CLOUD_INIT_SECRET_NAME, "key": "userData"}, -} - - -def _xr( - pools: list[v1alpha1.NodePool], - credentials: v1alpha1.Credentials | None = None, -) -> dict: - """A NebiusCluster XR with the given node pools, as a request dict.""" - return v1alpha1.NebiusCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), +def _xr(*, credentials: v1alpha1.Credentials | None, node_pools: list[v1alpha1.NodePool]) -> fnv1.Resource: + """The observed NebiusCluster XR, with the given credentials and node pools.""" + xr = v1alpha1.NebiusCluster( + metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), spec=v1alpha1.Spec( - nodePools=pools, credentials=credentials, + nodePools=node_pools, ), - ).model_dump(exclude_none=True, mode="json") + ) + return fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json", by_alias=True))) -def _req( - pools: list[v1alpha1.NodePool], - observed_resources: dict[str, fnv1.Resource] | None = None, - credentials: v1alpha1.Credentials | None = None, - *, - with_provider_config: bool = True, - provider_config_resource: dict | None = None, -) -> fnv1.RunFunctionRequest: - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(pools, credentials))), - resources=observed_resources or {}, +def _desired_xr(*, credentials_secret: bool) -> fnv1.Resource: + """The desired XR's status, which names the credentials Secret only if credentials_secret.""" + secrets = [{"type": "Kubeconfig", "name": "test-cluster-kubeconfig-55b57", "key": "kubeconfig"}] + if credentials_secret: + secrets.append( + { + "type": "NebiusServiceAccountCredentials", + "name": "nebius-credentials", + "key": "credentials.json", + "namespace": "crossplane-system", + }, + ) + return fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"secrets": secrets, "cache": {"storageClassName": "modelplane-rwx-fs"}}}, ), ) - if with_provider_config: - pc = provider_config_resource if provider_config_resource is not None else _NEBIUS_PROVIDER_CONFIG - req.required_resources["nebius-provider-config"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(pc)), - ) - return req - -def _network(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "vpc.nebius.m.upbound.io/v1beta1", - "kind": "Network", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": {"name": "test-cluster"}, - }, - } - -def _subnet(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "vpc.nebius.m.upbound.io/v1beta1", - "kind": "Subnet", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "name": "test-cluster", - "networkIdSelector": {"matchControllerRef": True}, - "ipv4PrivatePools": {"useNetworkPools": True}, - }, - }, - } +def _nebius_provider_config(*, kind: str, name: str, namespace: str | None) -> fnv1.Resource: + """The Nebius provider config the XR's credentials name, sourcing them from a Secret.""" + metadata = {"name": name} + if namespace is not None: + metadata["namespace"] = namespace + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "nebius.m.upbound.io/v1beta1", + "kind": kind, + "metadata": metadata, + "spec": { + "identity": {"type": "ServiceAccount"}, + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "crossplane-system", + "name": "nebius-credentials", + "key": "credentials.json", + }, + }, + "projectID": "project-e00test", + }, + } + ), + ) -def _cluster(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", - "kind": "Cluster", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "name": "test-cluster", - "controlPlane": { - "version": "1.34", - "subnetIdSelector": {"matchControllerRef": True}, - "endpoints": {"publicEndpoint": {}}, +def _network(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed VPC network.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vpc.nebius.m.upbound.io/v1beta1", + "kind": "Network", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": {"name": "test-cluster"}, }, - }, - "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, - }, - } + } + ), + ready=ready, + ) -def _filesystem(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "compute.nebius.m.upbound.io/v1beta1", - "kind": "Filesystem", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "name": "test-cluster-cache", - "type": "NETWORK_SSD", - "sizeGibibytes": 1024, - }, - }, - } +def _subnet(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed subnet.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vpc.nebius.m.upbound.io/v1beta1", + "kind": "Subnet", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "name": "test-cluster", + "networkIdSelector": {"matchControllerRef": True}, + "ipv4PrivatePools": {"useNetworkPools": True}, + }, + }, + } + ), + ready=ready, + ) -def _cloud_init_secret() -> dict: - return { - "apiVersion": "v1", - "kind": "Secret", - "metadata": { - "name": _CLOUD_INIT_SECRET_NAME, - "namespace": "modelplane-system", - }, - "type": "Opaque", - "stringData": {"userData": _CLOUD_INIT}, - } +def _cluster(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed mk8s cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", + "kind": "Cluster", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "name": "test-cluster", + "controlPlane": { + "version": "1.34", + "subnetIdSelector": {"matchControllerRef": True}, + "endpoints": {"publicEndpoint": {}}, + }, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, + }, + } + ), + ready=ready, + ) -def _csi_release() -> dict: - return { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "Release", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { - "kind": "ProviderConfig", - "name": "test-cluster-kubeconfig-55b57", - }, - "forProvider": { - "chart": { - "name": "csi-mounted-fs-path", - "repository": "oci://cr.eu-north1.nebius.cloud/mk8s/helm", - "version": "0.1.6", - }, - "namespace": "kube-system", - "values": { - "dataDir": "/mnt/data/csi-mounted-fs-path-data/", - # The node plugin must tolerate the GPU taint so engine - # pods on GPU nodes can mount cache PVCs. - "tolerations": [ - { - "key": "nvidia.com/gpu", - "operator": "Exists", - "effect": "NoSchedule", - }, - ], +def _filesystem(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed cache filesystem.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.nebius.m.upbound.io/v1beta1", + "kind": "Filesystem", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "name": "test-cluster-cache", + "type": "NETWORK_SSD", + "sizeGibibytes": 1024, + }, }, - }, - }, - } + } + ), + ready=ready, + ) -def _storage_class() -> dict: - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { - "kind": "ProviderConfig", - "name": "test-cluster-kubeconfig-55b57", - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "storage.k8s.io/v1", - "kind": "StorageClass", - "metadata": {"name": "modelplane-rwx-fs"}, - "provisioner": "mounted-fs-path.csi.nebius.ai", - "volumeBindingMode": "WaitForFirstConsumer", +def _cloud_init_secret() -> fnv1.Resource: + """The Ready Secret holding the cloud-init user data that mounts the cache filesystem on every node.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": { + "name": "test-cluster-cloud-init-fd2f2", + "namespace": "modelplane-system", }, - }, - }, - } + "type": "Opaque", + "stringData": { + "userData": ( + "#cloud-config\n" + "runcmd:\n" + " - mkdir -p /mnt/data\n" + " - mount -t virtiofs modelplane-cache /mnt/data\n" + ' - printf "modelplane-cache /mnt/data virtiofs defaults,nofail 0 2\\n" >> /etc/fstab\n' + ), + }, + } + ), + ready=fnv1.READY_TRUE, + ) -def _nodegroup_system(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", - "kind": "NodeGroup", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "name": "test-cluster-system", - "parentIdSelector": {"matchControllerRef": True}, - "version": "1.34", - "autoscaling": {"minNodeCount": 1, "maxNodeCount": 2}, - "template": { - "resources": {"platform": "cpu-d3", "preset": "4vcpu-16gb"}, - "bootDisk": {"sizeGibibytes": 100, "type": "NETWORK_SSD"}, - "networkInterfaces": [ - {"subnetIdSelector": {"matchControllerRef": True}}, - ], - **_TEMPLATE_CACHE_MOUNT, - "metadata": {"labels": {"modelplane.ai/pool": "system"}}, +def _nodegroup_system(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed system node group.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "name": "test-cluster-system", + "parentIdSelector": {"matchControllerRef": True}, + "version": "1.34", + "autoscaling": {"minNodeCount": 1, "maxNodeCount": 2}, + "template": { + "resources": {"platform": "cpu-d3", "preset": "4vcpu-16gb"}, + "bootDisk": {"sizeGibibytes": 100, "type": "NETWORK_SSD"}, + "networkInterfaces": [ + {"subnetIdSelector": {"matchControllerRef": True}}, + ], + "filesystems": [ + { + "attachMode": "READ_WRITE", + "mountTag": "modelplane-cache", + "existingFilesystem": {"idSelector": {"matchControllerRef": True}}, + }, + ], + "cloudInitUserDataSecretRef": {"name": "test-cluster-cloud-init-fd2f2", "key": "userData"}, + "metadata": {"labels": {"modelplane.ai/pool": "system"}}, + }, + }, }, - }, - }, - } + } + ), + ready=ready, + ) def _nodegroup_gpu( - template_extra: dict, - cred_kind: str = "ClusterProviderConfig", - cred_name: str = "default", - **for_provider_extra: object, -) -> dict: - """A GPU node group golden with the standard template, merged with the - given scaling config and extra template fields.""" - template = { + *, + cred_kind: str, + cred_name: str, + autoscaling: dict | None, + fixed_node_count: int | None, + fabric: str | None, + ready: fnv1.Ready, +) -> fnv1.Resource: + """The composed node group for the gpu-h100 pool, on fabric's GPU cluster if any.""" + template: dict[str, Any] = { "resources": {"platform": "gpu-h100-sxm", "preset": "8gpu-128vcpu-1600gb"}, "bootDisk": {"sizeGibibytes": 200, "type": "NETWORK_SSD"}, "networkInterfaces": [ {"subnetIdSelector": {"matchControllerRef": True}}, ], - **_TEMPLATE_CACHE_MOUNT, + "filesystems": [ + { + "attachMode": "READ_WRITE", + "mountTag": "modelplane-cache", + "existingFilesystem": {"idSelector": {"matchControllerRef": True}}, + }, + ], + "cloudInitUserDataSecretRef": {"name": "test-cluster-cloud-init-fd2f2", "key": "userData"}, "metadata": { "labels": { "modelplane.ai/pool": "gpu-h100", @@ -314,500 +283,892 @@ def _nodegroup_gpu( {"key": "nvidia.com/gpu", "value": "true", "effect": "NO_SCHEDULE"}, ], } - template.update(template_extra) - return { - "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", - "kind": "NodeGroup", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "name": "test-cluster-gpu-h100", - "parentIdSelector": {"matchControllerRef": True}, - "version": "1.34", - "template": template, - **for_provider_extra, - }, - }, + if fabric is not None: + template["gpuCluster"] = { + "idSelector": {"matchControllerRef": True, "matchLabels": {"modelplane.ai/fabric": fabric}}, + } + for_provider: dict[str, Any] = { + "name": "test-cluster-gpu-h100", + "parentIdSelector": {"matchControllerRef": True}, + "version": "1.34", + "template": template, } + if autoscaling is not None: + for_provider["autoscaling"] = autoscaling + if fixed_node_count is not None: + for_provider["fixedNodeCount"] = fixed_node_count + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": for_provider, + }, + } + ), + ready=ready, + ) -def _provider_config(api_version: str, kind: str) -> dict: - return { - "apiVersion": api_version, - "kind": kind, - "metadata": {"name": "test-cluster-kubeconfig-55b57"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "name": "test-cluster-kubeconfig-55b57", - "namespace": "modelplane-system", - "key": "kubeconfig", - }, - }, - "identity": { - "type": "NebiusServiceAccountCredentials", - "source": "Secret", - "secretRef": { - "name": "nebius-credentials", - "namespace": "crossplane-system", - "key": "credentials.json", +def _provider_config(*, api_version: str) -> fnv1.Resource: + """A Ready ProviderConfig targeting the cluster as the Nebius service account.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": api_version, + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster-kubeconfig-55b57"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", + "key": "kubeconfig", + }, + }, + "identity": { + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": { + "name": "nebius-credentials", + "namespace": "crossplane-system", + "key": "credentials.json", + }, + }, }, - }, - }, - } + } + ), + ready=fnv1.READY_TRUE, + ) -def _status(*, with_credentials: bool = True) -> dict: - secrets: list[dict] = [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-55b57", - "key": "kubeconfig", - }, - ] - if with_credentials: - secrets.append( - { - "type": "NebiusServiceAccountCredentials", - "name": "nebius-credentials", - "key": "credentials.json", - "namespace": "crossplane-system", - }, - ) - return { - "status": { - "secrets": secrets, - "cache": {"storageClassName": "modelplane-rwx-fs"}, - }, - } +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) -def _observed_ready(desired: dict) -> fnv1.Resource: - """An observed variant of a desired resource with a Ready=True condition.""" - observed = { - **desired, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - }, - } - return fnv1.Resource(resource=resource.dict_to_struct(observed)) - - -_GPU_POOL = v1alpha1.NodePool( - name="gpu-h100", - role="GPU", - platform="gpu-h100-sxm", - preset="8gpu-128vcpu-1600gb", - diskSizeGb=200, - maxNodeCount=4, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), -) - - -def _compose_cases() -> list[Case]: - """The cases for test_compose, with requirements patched onto their wants by position.""" - cases = [ - Case( - name="first pass composes infra resources; autoscaling from maxNodeCount", - req=_req([_GPU_POOL]), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - # The CSI driver release and StorageClass aren't - # composed yet: the cluster isn't observed, so the - # ProviderConfigs can't reach it. - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - # nodeCount defaults to 1 and minNodeCount is - # unset, so autoscaling starts at the node count. - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, +# Every case is a NebiusCluster named test-cluster in modelplane-system. Every +# response requires the Nebius provider config the XR's credentials name, which +# is the ClusterProviderConfig named default unless the XR says otherwise. The +# function reads the credentials Secret off it. +COMPOSE_CASES = [ + # The pool sets maxNodeCount but not minNodeCount, and its nodeCount is 1, + # so this case can't tell an autoscaling floor taken from nodeCount from one + # fixed at 1. The CSI driver release and StorageClass aren't composed yet: + # the cluster isn't observed, so the ProviderConfigs can't reach it. + Case( + name="FirstPass", + reason="With nothing observed, a NebiusCluster composes its Nebius infrastructure and ProviderConfigs, autoscaling the GPU pool from one node to maxNodeCount.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-h100", + role="GPU", + platform="gpu-h100-sxm", + preset="8gpu-128vcpu-1600gb", + diskSizeGb=200, + nodeCount=1, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), ), - }, + ], ), - context=structpb.Struct(), ), - ), - Case( - name="provider config not yet fetched gates provider configs, not infra", - req=_req([_GPU_POOL], with_provider_config=False), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_status(with_credentials=False)), - ), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), - ), - }, + required_resources={ + "nebius-provider-config": fnv1.Resources( + items=[_nebius_provider_config(kind="ClusterProviderConfig", name="default", namespace=None)], ), - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for Nebius ClusterProviderConfig default", + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(credentials_secret=True), + resources={ + "network": _network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "filesystem": _filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - ], - context=structpb.Struct(), + "cloud-init": _cloud_init_secret(), + "nodegroup-system": _nodegroup_system( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-gpu-h100": _nodegroup_gpu( + cred_kind="ClusterProviderConfig", + cred_name="default", + autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, + fixed_node_count=None, + fabric=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, ), - ), - Case( - name="deleted provider config keeps credentials from the observed ProviderConfig", - req=_req( - [_GPU_POOL], - observed_resources={ - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "nebius-provider-config": fnv1.ResourceSelector( + api_version="nebius.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", ), }, - with_provider_config=False, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ), + ), + Case( + name="ProviderConfigNotFetched", + reason="Before the Nebius provider config is fetched, and with no ProviderConfig observed, a NebiusCluster composes its infrastructure but not the ProviderConfigs, and reports it's waiting.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-h100", + role="GPU", + platform="gpu-h100-sxm", + preset="8gpu-128vcpu-1600gb", + diskSizeGb=200, + nodeCount=1, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), ), - }, + ], ), - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Nebius ClusterProviderConfig default not found; keeping the " - "credentials the composed ProviderConfig already carries", - ), - ], - context=structpb.Struct(), ), ), - Case( - name="fixed-size fabric pool composes a GPU cluster and fixedNodeCount", - req=_req( - [ - v1alpha1.NodePool( - name="gpu-h100", - role="GPU", - platform="gpu-h100-sxm", - preset="8gpu-128vcpu-1600gb", - diskSizeGb=200, - nodeCount=2, - fabric=v1alpha1.Fabric( - type="InfiniBand", - infiniband=v1alpha1.Infiniband(fabric="fabric-2"), - ), - gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(credentials_secret=False), + resources={ + "network": _network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "filesystem": _filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cloud-init": _cloud_init_secret(), + "nodegroup-system": _nodegroup_system( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - ] + "nodegroup-gpu-h100": _nodegroup_gpu( + cred_kind="ClusterProviderConfig", + cred_name="default", + autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, + fixed_node_count=None, + fabric=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for Nebius ClusterProviderConfig default", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "nebius-provider-config": fnv1.ResourceSelector( + api_version="nebius.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", + ), + }, + ), + ), + ), + # Keeping the observed credentials stops the function tearing the + # ProviderConfigs out of desired state. Here the config hasn't been fetched + # yet, and the function falls back the same way when it's gone. + Case( + name="ObservedCredentials", + reason="With the Nebius provider config not yet fetched but a composed ProviderConfig observed, a NebiusCluster keeps that ProviderConfig's credentials and still composes the ProviderConfigs.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-h100", + role="GPU", + platform="gpu-h100-sxm", + preset="8gpu-128vcpu-1600gb", + diskSizeGb=200, + nodeCount=1, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), ), - "gpu-cluster-fabric-2": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "compute.nebius.m.upbound.io/v1beta1", - "kind": "GpuCluster", - "metadata": {"labels": {"modelplane.ai/fabric": "fabric-2"}}, - "spec": { - "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, - "forProvider": { - "name": "test-cluster-fabric-2", - "infinibandFabric": "fabric-2", + ], + ), + resources={ + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster-kubeconfig-55b57"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", + "key": "kubeconfig", }, }, - } - ), - ), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu( - { - "gpuCluster": { - "idSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/fabric": "fabric-2"}, - }, + "identity": { + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": { + "name": "nebius-credentials", + "namespace": "crossplane-system", + "key": "credentials.json", }, }, - fixedNodeCount=2, - ), - ), + }, + } ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + ), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(credentials_secret=True), + resources={ + "network": _network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "filesystem": _filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cloud-init": _cloud_init_secret(), + "nodegroup-system": _nodegroup_system( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-gpu-h100": _nodegroup_gpu( + cred_kind="ClusterProviderConfig", + cred_name="default", + autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, + fixed_node_count=None, + fabric=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Nebius ClusterProviderConfig default not found; keeping the " + "credentials the composed ProviderConfig already carries", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "nebius-provider-config": fnv1.ResourceSelector( + api_version="nebius.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", + ), + }, + ), + ), + ), + Case( + name="FixedInfiniBandPool", + reason="A NebiusCluster with a fixed-size pool on InfiniBand fabric-2 composes a GpuCluster for the fabric and a node group of fixedNodeCount 2 on it.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-h100", + role="GPU", + platform="gpu-h100-sxm", + preset="8gpu-128vcpu-1600gb", + diskSizeGb=200, + nodeCount=2, + fabric=v1alpha1.Fabric( + type="InfiniBand", + infiniband=v1alpha1.Infiniband(fabric="fabric-2"), ), - ready=fnv1.READY_TRUE, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), ), - }, + ], ), - context=structpb.Struct(), ), + required_resources={ + "nebius-provider-config": fnv1.Resources( + items=[_nebius_provider_config(kind="ClusterProviderConfig", name="default", namespace=None)], + ), + }, ), - Case( - name="marks managed resources ready from observed conditions", - req=_req( - [_GPU_POOL], - observed_resources={ - "network": _observed_ready(_network()), - "subnet": _observed_ready(_subnet()), - "cluster": _observed_ready(_cluster()), - "filesystem": _observed_ready(_filesystem()), - "release-csi-mounted-fs-path": _observed_ready(_csi_release()), - "nodegroup-system": _observed_ready(_nodegroup_system()), - "nodegroup-gpu-h100": _observed_ready( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(credentials_secret=True), + resources={ + "network": _network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "filesystem": _filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cloud-init": _cloud_init_secret(), + "gpu-cluster-fabric-2": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.nebius.m.upbound.io/v1beta1", + "kind": "GpuCluster", + "metadata": {"labels": {"modelplane.ai/fabric": "fabric-2"}}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "name": "test-cluster-fabric-2", + "infinibandFabric": "fabric-2", + }, + }, + } + ), + ), + "nodegroup-system": _nodegroup_system( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-gpu-h100": _nodegroup_gpu( + cred_kind="ClusterProviderConfig", + cred_name="default", + autoscaling=None, + fixed_node_count=2, + fabric="fabric-2", + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ready=fnv1.READY_TRUE, - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ready=fnv1.READY_TRUE, + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "nebius-provider-config": fnv1.ResourceSelector( + api_version="nebius.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", + ), + }, + ), + ), + ), + # The release tolerates the GPU taint so its node plugin runs on the GPU + # nodes, where engine pods mount cache PVCs. + Case( + name="ResourcesReady", + reason="With the cluster, its other Nebius resources and the CSI driver release observed Ready, a NebiusCluster marks them ready and composes the release and StorageClass.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pools=[ + v1alpha1.NodePool( + name="gpu-h100", + role="GPU", + platform="gpu-h100-sxm", + preset="8gpu-128vcpu-1600gb", + diskSizeGb=200, + nodeCount=1, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, + ], + ), + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vpc.nebius.m.upbound.io/v1beta1", + "kind": "Network", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": {"name": "test-cluster"}, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } ), - "filesystem": fnv1.Resource( - resource=resource.dict_to_struct(_filesystem()), - ready=fnv1.READY_TRUE, + ), + "subnet": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vpc.nebius.m.upbound.io/v1beta1", + "kind": "Subnet", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "name": "test-cluster", + "networkIdSelector": {"matchControllerRef": True}, + "ipv4PrivatePools": {"useNetworkPools": True}, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } ), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", + "kind": "Cluster", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "name": "test-cluster", + "controlPlane": { + "version": "1.34", + "subnetIdSelector": {"matchControllerRef": True}, + "endpoints": {"publicEndpoint": {}}, + }, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } ), - # The cluster is observed, so the CSI driver - # release and StorageClass are composed too. - "release-csi-mounted-fs-path": fnv1.Resource( - resource=resource.dict_to_struct(_csi_release()), - ready=fnv1.READY_TRUE, + ), + "filesystem": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.nebius.m.upbound.io/v1beta1", + "kind": "Filesystem", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "name": "test-cluster-cache", + "type": "NETWORK_SSD", + "sizeGibibytes": 1024, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } ), - "storage-class-rwx-fs": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class()), - ready=fnv1.READY_TRUE, + ), + "release-csi-mounted-fs-path": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "forProvider": { + "chart": { + "name": "csi-mounted-fs-path", + "repository": "oci://cr.eu-north1.nebius.cloud/mk8s/helm", + "version": "0.1.6", + }, + "namespace": "kube-system", + "values": { + "dataDir": "/mnt/data/csi-mounted-fs-path-data/", + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + }, + ], + }, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } ), - "nodegroup-system": fnv1.Resource( - resource=resource.dict_to_struct(_nodegroup_system()), - ready=fnv1.READY_TRUE, + ), + "nodegroup-system": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "name": "test-cluster-system", + "parentIdSelector": {"matchControllerRef": True}, + "version": "1.34", + "autoscaling": {"minNodeCount": 1, "maxNodeCount": 2}, + "template": { + "resources": {"platform": "cpu-d3", "preset": "4vcpu-16gb"}, + "bootDisk": {"sizeGibibytes": 100, "type": "NETWORK_SSD"}, + "networkInterfaces": [ + {"subnetIdSelector": {"matchControllerRef": True}}, + ], + "filesystems": [ + { + "attachMode": "READ_WRITE", + "mountTag": "modelplane-cache", + "existingFilesystem": {"idSelector": {"matchControllerRef": True}}, + }, + ], + "cloudInitUserDataSecretRef": { + "name": "test-cluster-cloud-init-fd2f2", + "key": "userData", + }, + "metadata": {"labels": {"modelplane.ai/pool": "system"}}, + }, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } ), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), - ready=fnv1.READY_TRUE, + ), + "nodegroup-gpu-h100": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "name": "test-cluster-gpu-h100", + "parentIdSelector": {"matchControllerRef": True}, + "version": "1.34", + "template": { + "resources": {"platform": "gpu-h100-sxm", "preset": "8gpu-128vcpu-1600gb"}, + "bootDisk": {"sizeGibibytes": 200, "type": "NETWORK_SSD"}, + "networkInterfaces": [ + {"subnetIdSelector": {"matchControllerRef": True}}, + ], + "filesystems": [ + { + "attachMode": "READ_WRITE", + "mountTag": "modelplane-cache", + "existingFilesystem": {"idSelector": {"matchControllerRef": True}}, + }, + ], + "cloudInitUserDataSecretRef": { + "name": "test-cluster-cloud-init-fd2f2", + "key": "userData", + }, + "metadata": { + "labels": { + "modelplane.ai/pool": "gpu-h100", + "modelplane.ai/gpu": "nvidia-h100", + }, + }, + "gpuSettings": {"driversPreset": "cuda13.0"}, + "taints": [ + {"key": "nvidia.com/gpu", "value": "true", "effect": "NO_SCHEDULE"}, + ], + }, + "autoscaling": {"minNodeCount": 1, "maxNodeCount": 4}, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ), + }, + ), + required_resources={ + "nebius-provider-config": fnv1.Resources( + items=[_nebius_provider_config(kind="ClusterProviderConfig", name="default", namespace=None)], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(credentials_secret=True), + resources={ + "network": _network(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "subnet": _subnet(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "cluster": _cluster(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "filesystem": _filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "cloud-init": _cloud_init_secret(), + "release-csi-mounted-fs-path": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "forProvider": { + "chart": { + "name": "csi-mounted-fs-path", + "repository": "oci://cr.eu-north1.nebius.cloud/mk8s/helm", + "version": "0.1.6", + }, + "namespace": "kube-system", + "values": { + "dataDir": "/mnt/data/csi-mounted-fs-path-data/", + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + }, + ], + }, + }, + }, + } ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_TRUE, + ), + "storage-class-rwx-fs": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "storage.k8s.io/v1", + "kind": "StorageClass", + "metadata": {"name": "modelplane-rwx-fs"}, + "provisioner": "mounted-fs-path.csi.nebius.ai", + "volumeBindingMode": "WaitForFirstConsumer", + }, + }, + }, + } ), - }, - ), - context=structpb.Struct(), + ready=fnv1.READY_TRUE, + ), + "nodegroup-system": _nodegroup_system( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "nodegroup-gpu-h100": _nodegroup_gpu( + cred_kind="ClusterProviderConfig", + cred_name="default", + autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, + fixed_node_count=None, + fabric=None, + ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, ), - ), - Case( - name="custom credentials flow through to all cloud MRs", - req=_req( - [_GPU_POOL], - credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-nebius-account"), - provider_config_resource={ - "apiVersion": "nebius.m.upbound.io/v1beta1", - "kind": "ProviderConfig", - "metadata": {"name": "my-nebius-account", "namespace": "crossplane-system"}, - "spec": { - "identity": {"type": "ServiceAccount"}, - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "crossplane-system", - "name": "nebius-credentials", - "key": "credentials.json", - }, - }, - "projectID": "project-e00test", - }, + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "nebius-provider-config": fnv1.ResourceSelector( + api_version="nebius.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", + ), }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network("ProviderConfig", "my-nebius-account")), - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet("ProviderConfig", "my-nebius-account")), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster("ProviderConfig", "my-nebius-account")), - ), - "filesystem": fnv1.Resource( - resource=resource.dict_to_struct(_filesystem("ProviderConfig", "my-nebius-account")), - ), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_system("ProviderConfig", "my-nebius-account"), - ), - ), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu( - {}, - "ProviderConfig", - "my-nebius-account", - autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, - ), - ), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ), + ), + # The XR names a namespaced ProviderConfig, so the function requires it from + # the XR's own namespace. The ProviderConfig returned here sits in + # crossplane-system, which Crossplane wouldn't return for that selector. + # Status names crossplane-system as the credentials Secret's namespace. That + # is both the ProviderConfig's namespace and the one its secretRef sets, so + # this case can't show which of them the function reads. + Case( + name="CustomCredentials", + reason="The ProviderConfig a NebiusCluster's spec.credentials names becomes every Nebius managed resource's providerConfigRef.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-nebius-account"), + node_pools=[ + v1alpha1.NodePool( + name="gpu-h100", + role="GPU", + platform="gpu-h100-sxm", + preset="8gpu-128vcpu-1600gb", + diskSizeGb=200, + nodeCount=1, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ], + ), + ), + required_resources={ + "nebius-provider-config": fnv1.Resources( + items=[ + _nebius_provider_config( + kind="ProviderConfig", name="my-nebius-account", namespace="crossplane-system" ), - }, + ], ), - context=structpb.Struct(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(credentials_secret=True), + resources={ + "network": _network( + cred_kind="ProviderConfig", cred_name="my-nebius-account", ready=fnv1.READY_UNSPECIFIED + ), + "subnet": _subnet( + cred_kind="ProviderConfig", cred_name="my-nebius-account", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster( + cred_kind="ProviderConfig", cred_name="my-nebius-account", ready=fnv1.READY_UNSPECIFIED + ), + "filesystem": _filesystem( + cred_kind="ProviderConfig", cred_name="my-nebius-account", ready=fnv1.READY_UNSPECIFIED + ), + "cloud-init": _cloud_init_secret(), + "nodegroup-system": _nodegroup_system( + cred_kind="ProviderConfig", cred_name="my-nebius-account", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-gpu-h100": _nodegroup_gpu( + cred_kind="ProviderConfig", + cred_name="my-nebius-account", + autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, + fixed_node_count=None, + fabric=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "nebius-provider-config": fnv1.ResourceSelector( + api_version="nebius.m.upbound.io/v1beta1", + kind="ProviderConfig", + match_name="my-nebius-account", + namespace="modelplane-system", + ), + }, ), ), - ] - - # Every compose path declares the provider config requirement; the - # selector kind and name vary by credentials. - custom_creds_selector = fnv1.ResourceSelector( - api_version="nebius.m.upbound.io/v1beta1", - kind="ProviderConfig", - match_name="my-nebius-account", - namespace="modelplane-system", - ) - for case in cases[:-1]: - case.want.requirements.resources["nebius-provider-config"].CopyFrom(_PROVIDER_CONFIG_SELECTOR) - cases[-1].want.requirements.resources["nebius-provider-config"].CopyFrom(custom_creds_selector) - return cases - - -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + ), +] -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction composes a NebiusCluster's network, mk8s cluster, node groups and ProviderConfigs.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-serving-stack/tests/test_collector.py b/functions/compose-serving-stack/tests/test_collector.py index d6692f3ba..406a3bceb 100644 --- a/functions/compose-serving-stack/tests/test_collector.py +++ b/functions/compose-serving-stack/tests/test_collector.py @@ -12,8 +12,16 @@ # See the License for the specific language governing permissions and # limitations under the License. -"""Tests for the collector this stack composes.""" +"""Tests for the collector this stack composes. +A table per collector function, each comparing what it renders whole: the +config, the OTTL statements and the blocks they run in, the manifests, the +exporters and the authenticators. Then the properties no single rendering +shows: what the substrate job's port rewrite does to an address, and what holds +of the unit conversions and the built-in mappings. +""" + +import dataclasses import re import typing @@ -24,390 +32,1177 @@ from models.ai.modelplane.telemetrydestination import v1alpha1 as tdv1alpha1 from pydantic import ValidationError -# A client authenticator: an exporter needs one of those, not the oidc -# extension, which authenticates callers of a receiver. -_EXTENSIONS = {"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}} +@dataclasses.dataclass +class ConfigCase: + """A test case for collector.config.""" -def _sink(name: str = "primary", type_: str = "otlphttp", secret: str | None = None) -> tdv1alpha1.Sink: - return tdv1alpha1.Sink.model_validate( - { - "name": name, - "type": type_, - "endpoint": "https://otel.acme.example", - **({"secretRef": {"name": secret}, "auth": {"bearerTokenKey": "token"}} if secret else {}), - } - ) + name: str + reason: str + cluster: str + mappings: list[mmv1alpha1.MetricMapping] + sinks: list[tdv1alpha1.Sink] + extensions: dict + want: dict -_SINKS = [_sink()] +@dataclasses.dataclass +class PortRewriteCase: + """A test case for the substrate job's rewrite of a pod's address onto its annotated port.""" + name: str + reason: str + address: str + want: str -def _metric_statements() -> list[str]: - """The rename statements, which are the last of the four blocks.""" - return collector.statements(list(stacks.BUILTIN_MAPPINGS))[-1] +@dataclasses.dataclass +class StatementsCase: + """A test case for collector.statements.""" -def _config(*, extensions: dict | None = None, sinks: list | None = None) -> dict: - return yaml.safe_load( - collector.config( - "prod-us-east", - list(stacks.BUILTIN_MAPPINGS), - _SINKS if sinks is None else sinks, - _EXTENSIONS if extensions is None else extensions, - ) - ) + name: str + reason: str + mappings: list[mmv1alpha1.MetricMapping] + want: tuple[list[str], list[str], list[str], list[str]] # (extract, scale, datapoint, metric) -def _objects(secret: str | None = None) -> dict: - sinks = [_sink(secret=secret)] if secret else _SINKS - return {k: m for k, m, _ in collector.objects("prod-us-east", list(stacks.BUILTIN_MAPPINGS), sinks, _EXTENSIONS)} +@dataclasses.dataclass +class TransformCase: + """A test case for collector._transform.""" + name: str + reason: str + mappings: list[mmv1alpha1.MetricMapping] + want: dict -def test_pipeline_order() -> None: - """The rename runs before the identity is lifted onto the resource. - Discovery writes the identity onto each datapoint and groupbyattrs - lifts it; a statement matching on a metric's name has to run while the - datapoints are still where the rename can reach them. - """ - procs = _config()["service"]["pipelines"]["metrics"]["processors"] - assert procs.index("transform/modelplane") < procs.index("groupbyattrs/identity") - assert procs[-1] == "batch" +@dataclasses.dataclass +class ObjectsCase: + """A test case for collector.objects.""" + name: str + reason: str + cluster: str + mappings: list[mmv1alpha1.MetricMapping] + sinks: list[tdv1alpha1.Sink] + extensions: dict + want: list[tuple[str, dict, str | None]] -def test_only_modelplane_leaves_the_cluster() -> None: - """A series the statements didn't rename is dropped.""" - assert "filter/modelplane" in _config()["processors"] - assert "filter/modelplane" in _config()["service"]["pipelines"]["metrics"]["processors"] +@dataclasses.dataclass +class ExportersCase: + """A test case for collector.exporters.""" -def test_cluster_is_stamped_here() -> None: - """One receiver downstream sees a merged stream and can't tell senders apart.""" - attrs = _config()["processors"]["resource/cluster"]["attributes"] - assert attrs == [{"key": "cluster", "value": "prod-us-east", "action": "upsert"}] + name: str + reason: str + sinks: list[tdv1alpha1.Sink] + want: dict -def test_the_jobs_cover_disjoint_pods() -> None: - """A pod two jobs both collect arrives twice, under two job names.""" - jobs = { - j["job_name"]: j["relabel_configs"] for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"] - } - substrate = jobs["modelplane-substrate"] - - def predicate(rules: list[dict], action: str) -> set[tuple]: - return {(tuple(r["source_labels"]), r["regex"]) for r in rules if r.get("action") == action} - - # Everything another job keeps, the substrate job drops on the same terms. - for job in ("modelplane-engines", "modelplane-gateway", "modelplane-gpu"): - for kept in predicate(jobs[job], "keep"): - if kept[0] == ("__meta_kubernetes_pod_container_port_name",): - continue # a port filter, not a pod filter - assert kept in predicate(substrate, "drop"), f"{job} keeps {kept}, substrate does not drop it" - - -def test_only_the_identity_survives_to_the_exporter() -> None: - """Discovery attaches the pod's name and uid; neither is the deployment's.""" - blocks = _config()["processors"]["transform/identity"]["metric_statements"] - statement = blocks[0]["statements"][0] - # OTTL quotes with double quotes. A Python list renders single ones and - # the collector refuses to start, which a unit test on shape won't catch. - assert "'" not in statement - assert 'keep_keys(resource.attributes, ["cluster"' in statement - pipeline = _config()["service"]["pipelines"]["metrics"]["processors"] - assert pipeline.index("transform/identity") < pipeline.index("groupbyattrs/identity") - - -def test_the_identity_is_lifted_onto_the_resource() -> None: - """Without this a series arrives carrying only the cluster. - - Discovery writes the identity onto each datapoint. An exporter that - flattens a series into labels reads the resource, so something has to - move it, and this is the only processor that does. Removing it as a - no-op strips every series of what says who it belongs to - verified on - a cluster, where the resource came back carrying `cluster` alone. - """ - cfg = _config() - assert cfg["processors"]["groupbyattrs/identity"]["keys"] == list(collector._IDENTITY) - assert "groupbyattrs/identity" in cfg["service"]["pipelines"]["metrics"]["processors"] - - -def test_a_part_is_extracted_before_anything_selects_on_it() -> None: - """A label or a unit for an extracted part names a metric that must exist. - - The extraction mints `_count`, and a datapoint statement for it - selects on that name. Run the datapoint block first and it matches - nothing, silently. - """ - mapping = mmv1alpha1.MetricMapping.model_validate( - { - "spec": { - "metrics": [ - { - "from": "my_engine_duration_ms", - "to": "modelplane_requests_total", - "part": "Count", - "fromUnit": "Milliseconds", - "labels": [{"name": "status", "value": "ok"}], - } - ] - } - } - ) - blocks = collector._transform([mapping])["metric_statements"] - contexts = [b["context"] for b in blocks] - assert contexts == ["metric", "metric", "datapoint", "metric"] - assert "extract_count_metric" in blocks[0]["statements"][0] - # Everything selecting on the extracted name comes after the extraction. - for block in blocks[1:]: - for statement in block["statements"]: - assert "my_engine_duration_ms_count" in statement - - -def test_every_job_carries_something_unique_to_its_target() -> None: - """Two producers whose series are identical are one series, and one is lost. - - The modelplane identity names an engine and nothing else: a gateway pod - carries none of it, two replicas of a substrate controller share a - namespace, and a ModelReplica with copies > 1 runs several pods under - one replica index. - """ - assert "service.instance.id" in collector._IDENTITY - statement = _config()["processors"]["transform/identity"]["metric_statements"][0]["statements"][0] - assert '"service.instance.id"' in statement - - -def test_a_scrape_spike_cannot_take_the_collector_down() -> None: - """Nothing bounds what one interval brings off a fleet of engines.""" - cfg = _config() - assert "memory_limiter" in cfg["processors"] - assert cfg["service"]["pipelines"]["metrics"]["processors"][0] == "memory_limiter" - - -def test_the_port_rewrite_matches_an_ipv6_pod() -> None: - """__address__ is [2001:db8::1]:9090 there, which [^:]+ never matches.""" - rule = next( - r - for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"] - if j["job_name"] == "modelplane-substrate" - for r in j["relabel_configs"] - if r.get("target_label") == "__address__" - ) - for address in ("10.1.0.5:8000", "[2001:db8::1]:9090"): - matched = re.fullmatch(rule["regex"], f"{address};9402") - assert matched is not None, address - assert matched.expand(r"\1:\2").endswith(":9402") - - -def test_engine_scrape_selects_the_port_by_name() -> None: - """Matching by number would find the pd-sidecar on a disaggregated pod.""" - jobs = {j["job_name"]: j for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"]} - keeps = [r for r in jobs["modelplane-engines"]["relabel_configs"] if r.get("action") == "keep"] - assert "__meta_kubernetes_pod_container_port_name" in [k["source_labels"][0] for k in keeps] - - -def test_gateway_has_a_target_of_its_own() -> None: - """Its GenAI metrics are on the ext-proc sidecar, not the proxy's port.""" - jobs = [j["job_name"] for j in _config()["receivers"]["prometheus"]["config"]["scrape_configs"]] - assert "modelplane-gateway" in jobs - - -@pytest.mark.parametrize("type_", ["prometheus_remote_write", "prometheusremotewrite", "prometheus"]) -def test_both_spellings_of_remote_write_keep_their_identity(type_: str) -> None: - """The exporter registers as prometheus_remote_write in 0.161.0. - - prometheusremotewrite is the older name it still answers to. A sink - writing the one the collector's own documentation gives would - otherwise match no default here and export every series stripped of - the cluster, deployment, engine and role it belongs to - silently, - because the sink itself works. - """ - exporters = _config(sinks=[_sink(type_=type_)])["exporters"] - exporter = next(v for k, v in exporters.items() if k.startswith(f"{type_}/")) - assert exporter["resource_to_telemetry_conversion"]["enabled"] - - -def test_extensions_are_declared_to_the_service() -> None: - """An authenticator the service doesn't list is one the collector won't load.""" - assert _config()["service"]["extensions"] == ["oauth2client/acme"] - assert "extensions" not in _config(extensions={})["service"] - - -def test_energy_is_scaled_before_it_is_renamed() -> None: - """DCGM counts millijoules, and the name says joules. - - The scale is a block ahead of the renames, not a line ahead. The - processor finishes a block over every metric before the next one - starts, so a rename sharing the block would strand every metric - after the first at millijoules. - """ - blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] - assert [b["context"] for b in blocks] == ["metric", "metric"] - assert any("scale_metric(0.001)" in st for st in blocks[0]["statements"]) - assert any("modelplane_energy_joules_total" in st for st in blocks[-1]["statements"]) - - -def test_a_conversion_reaches_a_histogram_bucket() -> None: - """Setting value_double converts a gauge and leaves a histogram lying. - - A histogram holds its measurements in its sum, its minimum and maximum - and every bucket boundary, none of which is value_double. Renaming one - to seconds with its buckets still at milliseconds puts every quantile - a thousand times out, and nothing says so. - """ - blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] - for block in blocks: - for statement in block["statements"]: - assert "value_double" not in statement - scales = next(b for b in blocks if any("scale_metric" in st for st in b["statements"])) - assert scales["context"] == "metric" - - -def test_every_conversion_factor_is_a_float_literal() -> None: - """scale_metric takes a float, and 1048576 is an integer to OTTL. - - The collector refuses to start on it - "must be a float" - which - takes the whole cluster's telemetry down, and nothing short of - running the collector catches it. - """ - for unit, factor in collector._UNIT_FACTOR.items(): - assert "." in factor, f"{unit}: an OTTL float literal needs a decimal point" - float(factor) +@dataclasses.dataclass +class AuthenticatorsCase: + """A test case for collector.authenticators.""" + name: str + reason: str + sinks: list[tdv1alpha1.Sink] + want: dict -def test_a_conversion_cannot_drop_the_batch_it_rides_in() -> None: - """scale_metric refuses an exponential histogram. - Under the default error mode that one refusal fails the whole batch: - every metric from every pod in the scrape is lost, not the one it - could not convert. - """ - blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] - scales = next(b for b in blocks if any("scale_metric" in st for st in b["statements"])) - assert scales["error_mode"] == "ignore" +def _sink(*, name: str, type_: str, secret: str | None) -> tdv1alpha1.Sink: + """A sink exporting to otel.acme.example, with a bearer token read from secret if there is one.""" + return tdv1alpha1.Sink.model_validate( + { + "name": name, + "type": type_, + "endpoint": "https://otel.acme.example", + **({"secretRef": {"name": secret}, "auth": {"bearerTokenKey": "token"}} if secret else {}), + } + ) -def test_dcgm_units_are_converted_to_the_unit_the_name_claims() -> None: - """DCGM reports mJ and MiB; the names say joules and bytes.""" - blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] - scales = " ".join(blocks[0]["statements"]) - assert "DCGM_FI_DEV_TOTAL_ENERGY_CONSUMPTION" in scales - assert "DCGM_FI_DEV_FB_USED" in scales +# This builds CONFIG_CASES' whole want, which the helper rule forbids. The config +# runs to nearly 300 lines, and the cases differ only in their exporters and +# extensions, each a keyword argument here. Writing it out once per case would +# hide the parts that vary. OBJECTS_CASES embeds the same config in the +# collector's ConfigMap, which is the use the helper rule allows. +def _config( + *, exporters: dict, pipeline_exporters: list[str], extensions: dict | None, service_extensions: list[str] | None +) -> dict: + """The collector's config for prod-us-east with the built-in mappings, as the dict config dumps to YAML.""" + service: dict = { + "pipelines": { + "metrics": { + "receivers": ["prometheus"], + # The renames run before groupbyattrs lifts the identity, while + # the datapoints are still where a statement matching on a + # metric's name can reach them. + "processors": [ + "memory_limiter", + "resource/cluster", + "transform/identity", + "transform/modelplane", + "groupbyattrs/identity", + "filter/modelplane", + "batch", + ], + "exporters": pipeline_exporters, + } + }, + } + if service_extensions is not None: + # An authenticator the service doesn't list is one the collector won't + # load. + service["extensions"] = service_extensions + config: dict = { + "receivers": { + "prometheus": { + "config": { + "scrape_configs": [ + { + "job_name": "modelplane-engines", + "scrape_interval": "15s", + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + { + "source_labels": ["__meta_kubernetes_pod_label_modelplane_ai_deployment"], + "action": "keep", + "regex": ".+", + }, + # By name: matching by number would find the + # pd-sidecar on a disaggregated pod. + { + "source_labels": ["__meta_kubernetes_pod_container_port_name"], + "action": "keep", + "regex": "http", + }, + { + "source_labels": ["__meta_kubernetes_pod_label_modelplane_ai_deployment"], + "target_label": "deployment", + }, + { + "source_labels": ["__meta_kubernetes_pod_label_modelplane_ai_replica"], + "target_label": "replica", + }, + { + "source_labels": ["__meta_kubernetes_pod_label_modelplane_ai_engine"], + "target_label": "engine", + }, + { + "source_labels": ["__meta_kubernetes_pod_label_modelplane_ai_role"], + "target_label": "role", + }, + {"source_labels": ["__meta_kubernetes_namespace"], "target_label": "namespace"}, + ], + }, + { + "job_name": "modelplane-gateway", + "scrape_interval": "15s", + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + { + "source_labels": [ + "__meta_kubernetes_pod_label_gateway_envoyproxy_io_owning_gateway_name" + ], + "action": "keep", + "regex": ".+", + }, + { + "source_labels": ["__meta_kubernetes_pod_container_port_name"], + "action": "keep", + "regex": "metrics", + }, + { + "source_labels": ["__meta_kubernetes_pod_annotation_prometheus_io_path"], + "action": "replace", + "target_label": "__metrics_path__", + "regex": "(.+)", + }, + ], + }, + # A target of the gateway's own: its GenAI metrics are on + # the ext-proc sidecar, not the proxy's port. + { + "job_name": "modelplane-gateway-genai", + "scrape_interval": "15s", + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + { + "source_labels": [ + "__meta_kubernetes_pod_label_gateway_envoyproxy_io_owning_gateway_name" + ], + "action": "keep", + "regex": ".+", + }, + { + "source_labels": ["__meta_kubernetes_pod_container_port_name"], + "action": "keep", + "regex": "aigw-admin", + }, + ], + }, + { + "job_name": "modelplane-gpu", + "scrape_interval": "15s", + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + { + "source_labels": [ + "__meta_kubernetes_pod_label_app_kubernetes_io_name", + "__meta_kubernetes_pod_label_app", + ], + "action": "keep", + "regex": ".*dcgm.*", + }, + { + "source_labels": ["__meta_kubernetes_pod_container_port_name"], + "action": "keep", + "regex": "metrics", + }, + {"source_labels": ["__meta_kubernetes_pod_node_name"], "target_label": "node"}, + ], + }, + { + "job_name": "modelplane-substrate", + "scrape_interval": "30s", + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + { + "source_labels": ["__meta_kubernetes_pod_annotation_prometheus_io_scrape"], + "action": "keep", + "regex": "true", + }, + # Every pod a job above keeps, dropped on the same + # terms, so the jobs cover disjoint pods: a pod two + # jobs both collect arrives twice, under two job + # names. + { + "source_labels": [ + "__meta_kubernetes_pod_label_gateway_envoyproxy_io_owning_gateway_name" + ], + "action": "drop", + "regex": ".+", + }, + { + "source_labels": ["__meta_kubernetes_pod_label_modelplane_ai_deployment"], + "action": "drop", + "regex": ".+", + }, + { + "source_labels": [ + "__meta_kubernetes_pod_label_app_kubernetes_io_name", + "__meta_kubernetes_pod_label_app", + ], + "action": "drop", + "regex": ".*dcgm.*", + }, + { + "source_labels": ["__meta_kubernetes_pod_annotation_prometheus_io_path"], + "action": "replace", + "target_label": "__metrics_path__", + "regex": "(.+)", + }, + { + "source_labels": [ + "__address__", + "__meta_kubernetes_pod_annotation_prometheus_io_port", + ], + "action": "replace", + "target_label": "__address__", + "regex": r"(\[.+\]|[^:]+)(?::\d+)?;(\d+)", + "replacement": "$1:$2", + }, + {"source_labels": ["__meta_kubernetes_namespace"], "target_label": "namespace"}, + ], + }, + ] + } + } + }, + "processors": { + # First in the pipeline: nothing bounds what one interval brings off + # a fleet of engines. + "memory_limiter": {"check_interval": "1s", "limit_percentage": 80, "spike_limit_percentage": 25}, + # Stamped here, because one receiver downstream sees a merged stream + # and can't tell senders apart. + "resource/cluster": {"attributes": [{"key": "cluster", "value": "prod-us-east", "action": "upsert"}]}, + # Only the identity survives to the exporter: discovery attaches the + # pod's name and uid, and neither is the deployment's. It includes + # service.instance.id because two producers whose series are + # identical are one series, and one is lost. OTTL quotes with double + # quotes, and the single ones a Python list renders stop the + # collector starting. + "transform/identity": { + "metric_statements": [ + { + "context": "resource", + "statements": [ + 'keep_keys(resource.attributes, ["cluster", "deployment", "engine", "namespace", "node", "replica", "role", "service.instance.id", "service.name"])' + ], + } + ] + }, + "transform/modelplane": { + "metric_statements": [ + # DCGM reports mJ and MiB, and the names say joules and bytes. + # scale_metric converts a histogram's sum, bounds and buckets + # where setting value_double would convert a gauge and leave a + # histogram lying. It refuses an exponential histogram, which + # under the default error mode fails the whole batch. A block + # ahead of the renames rather than a line, because the + # processor finishes a block over every metric before the next + # one starts. + { + "context": "metric", + "error_mode": "ignore", + "statements": [ + 'scale_metric(1048576.0) where metric.name == "DCGM_FI_DEV_FB_USED"', + 'scale_metric(0.001) where metric.name == "DCGM_FI_DEV_TOTAL_ENERGY_CONSUMPTION"', + ], + }, + { + "context": "metric", + "statements": [ + 'set(metric.name, "modelplane_frontend_request_duration_seconds") where metric.name == "gen_ai_server_request_duration_seconds"', + 'set(metric.name, "modelplane_frontend_ttft_seconds") where metric.name == "gen_ai_server_time_to_first_token_seconds"', + 'set(metric.name, "modelplane_frontend_tpot_seconds") where metric.name == "gen_ai_server_time_per_output_token_seconds"', + 'set(metric.name, "modelplane_request_ttft_seconds") where metric.name == "vllm:time_to_first_token_seconds"', + 'set(metric.name, "modelplane_request_duration_seconds") where metric.name == "vllm:e2e_request_latency_seconds"', + 'set(metric.name, "modelplane_request_queue_seconds") where metric.name == "vllm:request_queue_time_seconds"', + 'set(metric.name, "modelplane_request_prefill_seconds") where metric.name == "vllm:request_prefill_time_seconds"', + 'set(metric.name, "modelplane_request_decode_seconds") where metric.name == "vllm:request_decode_time_seconds"', + 'set(metric.name, "modelplane_request_input_tokens") where metric.name == "vllm:request_prompt_tokens"', + 'set(metric.name, "modelplane_request_output_tokens") where metric.name == "vllm:request_generation_tokens"', + 'set(metric.name, "modelplane_requests_running") where metric.name == "vllm:num_requests_running"', + 'set(metric.name, "modelplane_requests_waiting") where metric.name == "vllm:num_requests_waiting"', + 'set(metric.name, "modelplane_kv_cache_utilization_ratio") where metric.name == "vllm:kv_cache_usage_perc"', + 'set(metric.name, "modelplane_requests_preempted_total") where metric.name == "vllm:num_preemptions_total"', + 'set(metric.name, "modelplane_prefix_cache_hits_total") where metric.name == "vllm:prefix_cache_hits_total"', + 'set(metric.name, "modelplane_prefix_cache_lookups_total") where metric.name == "vllm:prefix_cache_queries_total"', + 'set(metric.name, "modelplane_requests_running") where metric.name == "sglang:num_running_reqs"', + 'set(metric.name, "modelplane_requests_waiting") where metric.name == "sglang:num_queue_reqs"', + 'set(metric.name, "modelplane_kv_cache_utilization_ratio") where metric.name == "sglang:token_usage"', + 'set(metric.name, "modelplane_request_input_tokens") where metric.name == "sglang:prompt_tokens_histogram"', + 'set(metric.name, "modelplane_request_output_tokens") where metric.name == "sglang:generation_tokens_histogram"', + 'set(metric.name, "modelplane_route_decision_seconds") where metric.name == "llm_d_epp_scheduler_e2e_duration_seconds"', + 'set(metric.name, "modelplane_gpu_memory_used_bytes") where metric.name == "DCGM_FI_DEV_FB_USED"', + 'set(metric.name, "modelplane_gpu_compute_active_ratio") where metric.name == "DCGM_FI_PROF_GR_ENGINE_ACTIVE"', + 'set(metric.name, "modelplane_gpu_tensor_active_ratio") where metric.name == "DCGM_FI_PROF_PIPE_TENSOR_ACTIVE"', + 'set(metric.name, "modelplane_gpu_memory_bandwidth_ratio") where metric.name == "DCGM_FI_PROF_DRAM_ACTIVE"', + 'set(metric.name, "modelplane_gpu_temperature_celsius") where metric.name == "DCGM_FI_DEV_GPU_TEMP"', + 'set(metric.name, "modelplane_gpu_power_watts") where metric.name == "DCGM_FI_DEV_POWER_USAGE"', + 'set(metric.name, "modelplane_energy_joules_total") where metric.name == "DCGM_FI_DEV_TOTAL_ENERGY_CONSUMPTION"', + ], + }, + ] + }, + # Lifts the identity discovery wrote onto each datapoint onto the + # resource, where an exporter that flattens a series into labels + # reads it. Without it a series arrives carrying only the cluster. + "groupbyattrs/identity": { + "keys": [ + "cluster", + "namespace", + "deployment", + "replica", + "engine", + "role", + "node", + "service.name", + "service.instance.id", + ] + }, + # Only modelplane_* leaves the cluster: a series the statements + # didn't rename is dropped. + "filter/modelplane": {"metrics": {"metric": ['not IsMatch(name, "^modelplane_.*")']}}, + "batch": {"timeout": "10s"}, + }, + "exporters": exporters, + "service": service, + } + if extensions is not None: + config["extensions"] = extensions + return config + + +def _service_account() -> dict: + """The collector's ServiceAccount.""" + return { + "apiVersion": "v1", + "kind": "ServiceAccount", + "metadata": { + "name": "modelplane-collector", + "namespace": "modelplane-system", + "labels": {"app.kubernetes.io/name": "modelplane-collector", "app.kubernetes.io/managed-by": "modelplane"}, + }, + } -def test_every_unit_the_api_offers_has_a_conversion() -> None: - """A unit the API accepts with no conversion here is a KeyError at render time.""" - annotation = mmv1alpha1.Metric.model_fields["fromUnit"].annotation - literal = next(a for a in typing.get_args(annotation) if typing.get_origin(a) is typing.Literal) - assert set(typing.get_args(literal)) == set(collector._UNIT_FACTOR) +def _cluster_role() -> dict: + """The collector's ClusterRole.""" + return { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "ClusterRole", + "metadata": { + "name": "modelplane-collector", + "labels": {"app.kubernetes.io/name": "modelplane-collector", "app.kubernetes.io/managed-by": "modelplane"}, + }, + # Read only: service discovery needs to list pods, and nothing needs to + # write. + "rules": [ + { + "apiGroups": [""], + "resources": ["pods", "services", "endpoints", "nodes", "nodes/metrics"], + "verbs": ["get", "list", "watch"], + }, + {"nonResourceURLs": ["/metrics"], "verbs": ["get"]}, + ], + } -def test_a_percentage_is_divided_into_a_ratio() -> None: - """A component counting 0 to 100 under a name that says a ratio is 100x out. +def _cluster_role_binding() -> dict: + """The collector's ClusterRoleBinding.""" + return { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "ClusterRoleBinding", + "metadata": { + "name": "modelplane-collector", + "labels": {"app.kubernetes.io/name": "modelplane-collector", "app.kubernetes.io/managed-by": "modelplane"}, + }, + "roleRef": {"apiGroup": "rbac.authorization.k8s.io", "kind": "ClusterRole", "name": "modelplane-collector"}, + "subjects": [{"kind": "ServiceAccount", "name": "modelplane-collector", "namespace": "modelplane-system"}], + } - vLLM and SGLang both publish a fraction, so no built-in needs this, but - vLLM's is called kv_cache_usage_perc - the name is no guide, and an - engine that means it has to be able to say so. - """ - mapping = mmv1alpha1.MetricMapping.model_validate( - { - "spec": { - "metrics": [ - { - "from": "my_engine_cache_percent", - "to": "modelplane_kv_cache_utilization_ratio", - "fromUnit": "Percent", - } - ] - } - } - ) - _, scale, _, _ = collector.statements([mapping]) - assert scale == ['scale_metric(0.01) where metric.name == "my_engine_cache_percent"'] +def _config_map(*, config: dict) -> dict: + """The ConfigMap holding the collector's config, as YAML.""" + return { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": { + "name": "modelplane-collector", + "namespace": "modelplane-system", + "labels": {"app.kubernetes.io/name": "modelplane-collector", "app.kubernetes.io/managed-by": "modelplane"}, + }, + "data": {"collector.yaml": yaml.safe_dump(config, sort_keys=False)}, + } -def test_a_metric_name_cannot_end_the_comparison_early() -> None: - """A quote in `from` would rename whatever the rest of the line matched.""" - with pytest.raises(ValidationError, match="String should match pattern"): - mmv1alpha1.Metric.model_validate({"from": 'x" or true or name == "y', "to": "modelplane_x"}) - for mapping in stacks.BUILTIN_MAPPINGS: - for m in mapping.spec.metrics: - round_tripped = mmv1alpha1.Metric.model_validate({"from": m.from_, "to": m.to}) - assert round_tripped.from_ == m.from_ +def _deployment( + *, config_hash: str, volumes: list[dict], volume_mounts: list[dict], env_from: list[dict] | None +) -> dict: + """The collector's Deployment, restarted by its config's hash and mounting volumes.""" + container: dict = { + "name": "collector", + "image": "otel/opentelemetry-collector-contrib:0.161.0", + "args": ["--config=/conf/collector.yaml"], + "volumeMounts": volume_mounts, + "resources": {"requests": {"cpu": "100m", "memory": "256Mi"}, "limits": {"memory": "512Mi"}}, + } + if env_from is not None: + container["envFrom"] = env_from + return { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": { + "name": "modelplane-collector", + "namespace": "modelplane-system", + "labels": {"app.kubernetes.io/name": "modelplane-collector", "app.kubernetes.io/managed-by": "modelplane"}, + }, + "spec": { + "replicas": 1, + "selector": {"matchLabels": {"app.kubernetes.io/name": "modelplane-collector"}}, + "template": { + "metadata": { + "labels": {"app.kubernetes.io/name": "modelplane-collector"}, + "annotations": {"modelplane.ai/config-hash": config_hash}, + }, + "spec": { + "serviceAccountName": "modelplane-collector", + "containers": [container], + "volumes": volumes, + }, + }, + }, + } -def test_a_label_value_cannot_end_the_string_it_sits_in() -> None: - """`from` is pattern-constrained; a label's value cannot be. - A value and a `values` remap carry whatever vocabulary the component - already writes, so the schema has to take free text. A quote in one - would close the OTTL literal early and leave the remainder of the - value as OTTL - at best the collector refuses to start. - """ - mapping = mmv1alpha1.MetricMapping.model_validate( - { - "spec": { - "metrics": [ - { - "from": "my_engine_finish", - "to": "modelplane_requests_total", - "labels": [{"name": "reason", "from": "finish", "values": {'ab"c': 'x"y'}}], +# Every case renders the built-in mappings, read from the stacks package rather +# than written as literals, because fn.py always renders them and so every real +# config carries them. None of these cases is about them: BuiltIn in +# STATEMENTS_CASES pins what they compile to. The cost is that changing a +# built-in changes _config's transform/modelplane block too. +CONFIG_CASES = [ + ConfigCase( + name="OneSink", + reason=( + "With one sink and one extension, the collector exports the built-in mappings' series to the sink, " + "stamped with the cluster, and loads the extension." + ), + cluster="prod-us-east", + mappings=list(stacks.BUILTIN_MAPPINGS), + sinks=[_sink(name="primary", type_="otlphttp", secret=None)], + # A client authenticator: an exporter needs one of those, not the oidc + # extension, which authenticates callers of a receiver. + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + want=_config( + exporters={"otlphttp/primary": {"endpoint": "https://otel.acme.example"}}, + pipeline_exporters=["otlphttp/primary"], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + service_extensions=["oauth2client/acme"], + ), + ), + ConfigCase( + name="NoExtensions", + reason="With no extensions, the config declares none, to the service or at the top level.", + cluster="prod-us-east", + mappings=list(stacks.BUILTIN_MAPPINGS), + sinks=[_sink(name="primary", type_="otlphttp", secret=None)], + extensions={}, + want=_config( + exporters={"otlphttp/primary": {"endpoint": "https://otel.acme.example"}}, + pipeline_exporters=["otlphttp/primary"], + extensions=None, + service_extensions=None, + ), + ), + # The remote-write exporter registers as prometheus_remote_write in 0.161.0 + # and still answers to the older prometheusremotewrite. A sink writing the + # one the collector's own documentation gives would otherwise match no + # default and export every series stripped of the cluster, deployment, + # engine and role it belongs to - silently, because the sink itself works. + ConfigCase( + name="RemoteWrite", + reason="A prometheus_remote_write sink carries each series' resource attributes as labels.", + cluster="prod-us-east", + mappings=list(stacks.BUILTIN_MAPPINGS), + sinks=[_sink(name="primary", type_="prometheus_remote_write", secret=None)], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + want=_config( + exporters={ + "prometheus_remote_write/primary": { + "resource_to_telemetry_conversion": {"enabled": True}, + "endpoint": "https://otel.acme.example", + } + }, + pipeline_exporters=["prometheus_remote_write/primary"], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + service_extensions=["oauth2client/acme"], + ), + ), + ConfigCase( + name="RemoteWriteOldName", + reason="A prometheusremotewrite sink carries each series' resource attributes as labels.", + cluster="prod-us-east", + mappings=list(stacks.BUILTIN_MAPPINGS), + sinks=[_sink(name="primary", type_="prometheusremotewrite", secret=None)], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + want=_config( + exporters={ + "prometheusremotewrite/primary": { + "resource_to_telemetry_conversion": {"enabled": True}, + "endpoint": "https://otel.acme.example", + } + }, + pipeline_exporters=["prometheusremotewrite/primary"], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + service_extensions=["oauth2client/acme"], + ), + ), + ConfigCase( + name="Prometheus", + reason="A prometheus sink carries each series' resource attributes as labels.", + cluster="prod-us-east", + mappings=list(stacks.BUILTIN_MAPPINGS), + sinks=[_sink(name="primary", type_="prometheus", secret=None)], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + want=_config( + exporters={ + "prometheus/primary": { + "resource_to_telemetry_conversion": {"enabled": True}, + "endpoint": "https://otel.acme.example", + } + }, + pipeline_exporters=["prometheus/primary"], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + service_extensions=["oauth2client/acme"], + ), + ), +] + + +@pytest.mark.parametrize("case", CONFIG_CASES, ids=lambda case: case.name) +def test_config(case: ConfigCase) -> None: + """config renders the collector's whole configuration.""" + got = collector.config(case.cluster, case.mappings, case.sinks, case.extensions) + # The cases write the config as the dict the function dumps, in the order + # it builds it, so they read as config rather than as wrapped YAML. PyYAML + # dumps it on both sides. + assert got == yaml.safe_dump(case.want, sort_keys=False), case.reason + + +PORT_REWRITE_CASES = [ + PortRewriteCase( + name="IPv4", + reason="An IPv4 pod's address moves onto its annotated port.", + address="10.1.0.5:8000", + want="10.1.0.5:9402", + ), + PortRewriteCase( + name="IPv6", + reason="An IPv6 pod's bracketed address, which [^:]+ never matches, moves onto its annotated port.", + address="[2001:db8::1]:9090", + want="[2001:db8::1]:9402", + ), +] + + +@pytest.mark.parametrize("case", PORT_REWRITE_CASES, ids=lambda case: case.name) +def test_port_rewrite(case: PortRewriteCase) -> None: + """The substrate job rewrites a pod's address onto the port it annotates.""" + # This checks what the rewrite does to an address, which comparing the + # config can't, so it picks the one rule out of a rendered config. The + # config renders the built-in mappings, read from the stacks package, as + # fn.py always does, though the rule doesn't depend on them. + config = yaml.safe_load( + collector.config( + "prod-us-east", + list(stacks.BUILTIN_MAPPINGS), + [_sink(name="primary", type_="otlphttp", secret=None)], + {"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + ) + ) + rule = next( + r + for j in config["receivers"]["prometheus"]["config"]["scrape_configs"] + if j["job_name"] == "modelplane-substrate" + for r in j["relabel_configs"] + if r.get("target_label") == "__address__" + ) + matched = re.fullmatch(rule["regex"], f"{case.address};9402") + got = matched.expand(r"\1:\2") if matched else None + assert got == case.want, case.reason + + +STATEMENTS_CASES = [ + # The built-in mappings, read from the stacks package rather than restated, + # because what the collector compiles them to is what this case pins. + StatementsCase( + name="BuiltIn", + reason="The built-in mappings convert DCGM's units, then rename every series they map.", + mappings=list(stacks.BUILTIN_MAPPINGS), + want=( + [], + [ + 'scale_metric(1048576.0) where metric.name == "DCGM_FI_DEV_FB_USED"', + 'scale_metric(0.001) where metric.name == "DCGM_FI_DEV_TOTAL_ENERGY_CONSUMPTION"', + ], + [], + [ + 'set(metric.name, "modelplane_frontend_request_duration_seconds") where metric.name == "gen_ai_server_request_duration_seconds"', + 'set(metric.name, "modelplane_frontend_ttft_seconds") where metric.name == "gen_ai_server_time_to_first_token_seconds"', + 'set(metric.name, "modelplane_frontend_tpot_seconds") where metric.name == "gen_ai_server_time_per_output_token_seconds"', + 'set(metric.name, "modelplane_request_ttft_seconds") where metric.name == "vllm:time_to_first_token_seconds"', + 'set(metric.name, "modelplane_request_duration_seconds") where metric.name == "vllm:e2e_request_latency_seconds"', + 'set(metric.name, "modelplane_request_queue_seconds") where metric.name == "vllm:request_queue_time_seconds"', + 'set(metric.name, "modelplane_request_prefill_seconds") where metric.name == "vllm:request_prefill_time_seconds"', + 'set(metric.name, "modelplane_request_decode_seconds") where metric.name == "vllm:request_decode_time_seconds"', + 'set(metric.name, "modelplane_request_input_tokens") where metric.name == "vllm:request_prompt_tokens"', + 'set(metric.name, "modelplane_request_output_tokens") where metric.name == "vllm:request_generation_tokens"', + 'set(metric.name, "modelplane_requests_running") where metric.name == "vllm:num_requests_running"', + 'set(metric.name, "modelplane_requests_waiting") where metric.name == "vllm:num_requests_waiting"', + 'set(metric.name, "modelplane_kv_cache_utilization_ratio") where metric.name == "vllm:kv_cache_usage_perc"', + 'set(metric.name, "modelplane_requests_preempted_total") where metric.name == "vllm:num_preemptions_total"', + 'set(metric.name, "modelplane_prefix_cache_hits_total") where metric.name == "vllm:prefix_cache_hits_total"', + 'set(metric.name, "modelplane_prefix_cache_lookups_total") where metric.name == "vllm:prefix_cache_queries_total"', + # Not SGLang's latency histograms, sglang:time_to_first_token_seconds + # and sglang:inter_token_latency: their buckets resolve to 100ms + # where vLLM's resolve to 1ms. + 'set(metric.name, "modelplane_requests_running") where metric.name == "sglang:num_running_reqs"', + 'set(metric.name, "modelplane_requests_waiting") where metric.name == "sglang:num_queue_reqs"', + 'set(metric.name, "modelplane_kv_cache_utilization_ratio") where metric.name == "sglang:token_usage"', + 'set(metric.name, "modelplane_request_input_tokens") where metric.name == "sglang:prompt_tokens_histogram"', + 'set(metric.name, "modelplane_request_output_tokens") where metric.name == "sglang:generation_tokens_histogram"', + 'set(metric.name, "modelplane_route_decision_seconds") where metric.name == "llm_d_epp_scheduler_e2e_duration_seconds"', + 'set(metric.name, "modelplane_gpu_memory_used_bytes") where metric.name == "DCGM_FI_DEV_FB_USED"', + 'set(metric.name, "modelplane_gpu_compute_active_ratio") where metric.name == "DCGM_FI_PROF_GR_ENGINE_ACTIVE"', + 'set(metric.name, "modelplane_gpu_tensor_active_ratio") where metric.name == "DCGM_FI_PROF_PIPE_TENSOR_ACTIVE"', + 'set(metric.name, "modelplane_gpu_memory_bandwidth_ratio") where metric.name == "DCGM_FI_PROF_DRAM_ACTIVE"', + 'set(metric.name, "modelplane_gpu_temperature_celsius") where metric.name == "DCGM_FI_DEV_GPU_TEMP"', + 'set(metric.name, "modelplane_gpu_power_watts") where metric.name == "DCGM_FI_DEV_POWER_USAGE"', + 'set(metric.name, "modelplane_energy_joules_total") where metric.name == "DCGM_FI_DEV_TOTAL_ENERGY_CONSUMPTION"', + ], + ), + ), + # vLLM and SGLang both publish a fraction, so no built-in needs this, but + # vLLM's is called kv_cache_usage_perc: the name is no guide, and an engine + # that means it has to be able to say so. + StatementsCase( + name="Percent", + reason="A metric counted in percent is scaled by 0.01 into the ratio its new name claims.", + mappings=[ + mmv1alpha1.MetricMapping.model_validate( + { + "spec": { + "metrics": [ + { + "from": "my_engine_cache_percent", + "to": "modelplane_kv_cache_utilization_ratio", + "fromUnit": "Percent", + } + ] } - ] + } + ) + ], + want=( + [], + ['scale_metric(0.01) where metric.name == "my_engine_cache_percent"'], + [], + [ + 'set(metric.name, "modelplane_kv_cache_utilization_ratio") where metric.name == "my_engine_cache_percent"' + ], + ), + ), + # `from` is pattern-constrained, but a label's value can't be: a value and + # a `values` remap carry whatever vocabulary the component already writes, + # so the schema has to take free text. An unescaped quote would close the + # OTTL literal early and leave the rest of the value as OTTL. + StatementsCase( + name="QuotedLabelValue", + reason="A quote in a remapped label value is escaped, so it can't end the OTTL string it sits in.", + mappings=[ + mmv1alpha1.MetricMapping.model_validate( + { + "spec": { + "metrics": [ + { + "from": "my_engine_finish", + "to": "modelplane_requests_total", + "labels": [{"name": "reason", "from": "finish", "values": {'ab"c': 'x"y'}}], + } + ] + } + } + ) + ], + want=( + [], + [], + [ + 'set(datapoint.attributes["reason"], datapoint.attributes["finish"]) where metric.name == "my_engine_finish"', + r'set(datapoint.attributes["reason"], "x\"y") where metric.name == "my_engine_finish" and datapoint.attributes["finish"] == "ab\"c"', + 'delete_key(datapoint.attributes, "finish") where metric.name == "my_engine_finish"', + ], + ['set(metric.name, "modelplane_requests_total") where metric.name == "my_engine_finish"'], + ), + ), + # A label carried onto its own name is how a mapping remaps values in place. + # The delete that stops a carried label costing twice the cardinality would + # take the label the statements before it just set. + StatementsCase( + name="LabelOntoItself", + reason="A label carried onto its own name has its values remapped and isn't deleted afterwards.", + mappings=[ + mmv1alpha1.MetricMapping.model_validate( + { + "spec": { + "metrics": [ + { + "from": "my_engine_finish", + "to": "modelplane_requests_total", + "labels": [{"name": "reason", "from": "reason", "values": {"eos": "stop"}}], + } + ] + } + } + ) + ], + want=( + [], + [], + [ + 'set(datapoint.attributes["reason"], datapoint.attributes["reason"]) where metric.name == "my_engine_finish"', + 'set(datapoint.attributes["reason"], "stop") where metric.name == "my_engine_finish" and datapoint.attributes["reason"] == "eos"', + ], + ['set(metric.name, "modelplane_requests_total") where metric.name == "my_engine_finish"'], + ), + ), +] + + +@pytest.mark.parametrize("case", STATEMENTS_CASES, ids=lambda case: case.name) +def test_statements(case: StatementsCase) -> None: + """statements compiles the mappings to OTTL extract, scale, datapoint and metric statements.""" + got = collector.statements(case.mappings) + assert got == case.want, case.reason + + +TRANSFORM_CASES = [ + # The extraction mints my_engine_duration_ms_count, and the conversion, the + # label and the rename all select on that name. Run any of them first and it + # matches nothing, silently. + TransformCase( + name="ExtractedPart", + reason="A histogram part is extracted in a block ahead of everything that selects on its name.", + mappings=[ + mmv1alpha1.MetricMapping.model_validate( + { + "spec": { + "metrics": [ + { + "from": "my_engine_duration_ms", + "to": "modelplane_requests_total", + "part": "Count", + "fromUnit": "Milliseconds", + "labels": [{"name": "status", "value": "ok"}], + } + ] + } + } + ) + ], + want={ + "metric_statements": [ + { + "context": "metric", + "statements": ['extract_count_metric(true) where metric.name == "my_engine_duration_ms"'], + }, + { + "context": "metric", + "error_mode": "ignore", + "statements": ['scale_metric(0.001) where metric.name == "my_engine_duration_ms_count"'], + }, + { + "context": "datapoint", + "statements": [ + 'set(datapoint.attributes["status"], "ok") where metric.name == "my_engine_duration_ms_count"' + ], + }, + { + "context": "metric", + "statements": [ + 'set(metric.name, "modelplane_requests_total") where metric.name == "my_engine_duration_ms_count"' + ], + }, + ] + }, + ), +] + + +@pytest.mark.parametrize("case", TRANSFORM_CASES, ids=lambda case: case.name) +def test_transform(case: TransformCase) -> None: + """_transform orders the mappings' statements into the blocks the transform processor runs.""" + got = collector._transform(case.mappings) + assert got == case.want, case.reason + + +# Every case renders the built-in mappings, read from the stacks package, for +# the reason CONFIG_CASES gives. The config hashes are of a config that carries +# them, so changing a built-in changes every hash here too. +OBJECTS_CASES = [ + # The config hash is a literal, so it pins one computed in another process: + # hash() is seeded per process, and would redeploy the collector on every + # reconcile. + ObjectsCase( + name="NoSecret", + reason="A sink with no Secret gets a collector that mounts only its config.", + cluster="prod-us-east", + mappings=list(stacks.BUILTIN_MAPPINGS), + sinks=[_sink(name="primary", type_="otlphttp", secret=None)], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + want=[ + ("collector-serviceaccount", _service_account(), None), + ("collector-clusterrole", _cluster_role(), None), + ("collector-clusterrolebinding", _cluster_role_binding(), None), + ( + "collector-config", + _config_map( + config=_config( + exporters={"otlphttp/primary": {"endpoint": "https://otel.acme.example"}}, + pipeline_exporters=["otlphttp/primary"], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + service_extensions=["oauth2client/acme"], + ) + ), + None, + ), + ( + "collector", + _deployment( + config_hash="ad2c3f990baaeb4c", + volumes=[{"name": "config", "configMap": {"name": "modelplane-collector"}}], + volume_mounts=[{"name": "config", "mountPath": "/conf"}], + env_from=None, + ), + "object.status.readyReplicas > 0", + ), + ], + ), + # Mounted as a file as well as the environment, because a rotated token in + # an environment variable needs a restart to be read. + ObjectsCase( + name="Secret", + reason="A sink's Secret mounts as a file under the sink's own directory and as environment variables.", + cluster="prod-us-east", + mappings=list(stacks.BUILTIN_MAPPINGS), + sinks=[_sink(name="primary", type_="otlphttp", secret="telemetry-credentials")], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + want=[ + ("collector-serviceaccount", _service_account(), None), + ("collector-clusterrole", _cluster_role(), None), + ("collector-clusterrolebinding", _cluster_role_binding(), None), + ( + "collector-config", + _config_map( + config=_config( + exporters={ + "otlphttp/primary": { + "endpoint": "https://otel.acme.example", + "auth": {"authenticator": "bearertokenauth/primary"}, + } + }, + pipeline_exporters=["otlphttp/primary"], + extensions={ + "oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}, + "bearertokenauth/primary": {"filename": "/etc/modelplane/telemetry/primary/token"}, + }, + service_extensions=["bearertokenauth/primary", "oauth2client/acme"], + ) + ), + None, + ), + ( + "collector", + _deployment( + config_hash="9f072b028dc68ea5", + volumes=[ + {"name": "config", "configMap": {"name": "modelplane-collector"}}, + {"name": "credentials-primary", "secret": {"secretName": "telemetry-credentials"}}, + ], + volume_mounts=[ + {"name": "config", "mountPath": "/conf"}, + { + "name": "credentials-primary", + "mountPath": "/etc/modelplane/telemetry/primary", + "readOnly": True, + }, + ], + env_from=[{"secretRef": {"name": "telemetry-credentials"}}], + ), + "object.status.readyReplicas > 0", + ), + ], + ), + # Two sinks can both hold a key called token, and neither reads the other's. + ObjectsCase( + name="TwoSecrets", + reason="Two sinks' Secrets mount under a directory each.", + cluster="prod-us-east", + mappings=list(stacks.BUILTIN_MAPPINGS), + sinks=[ + _sink(name="vendor", type_="otlphttp", secret="vendor-token"), + _sink(name="prometheus", type_="prometheusremotewrite", secret="prom-token"), + ], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + want=[ + ("collector-serviceaccount", _service_account(), None), + ("collector-clusterrole", _cluster_role(), None), + ("collector-clusterrolebinding", _cluster_role_binding(), None), + ( + "collector-config", + _config_map( + config=_config( + exporters={ + "otlphttp/vendor": { + "endpoint": "https://otel.acme.example", + "auth": {"authenticator": "bearertokenauth/vendor"}, + }, + "prometheusremotewrite/prometheus": { + "resource_to_telemetry_conversion": {"enabled": True}, + "endpoint": "https://otel.acme.example", + "auth": {"authenticator": "bearertokenauth/prometheus"}, + }, + }, + pipeline_exporters=["otlphttp/vendor", "prometheusremotewrite/prometheus"], + extensions={ + "oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}, + "bearertokenauth/vendor": {"filename": "/etc/modelplane/telemetry/vendor/token"}, + "bearertokenauth/prometheus": {"filename": "/etc/modelplane/telemetry/prometheus/token"}, + }, + service_extensions=[ + "bearertokenauth/prometheus", + "bearertokenauth/vendor", + "oauth2client/acme", + ], + ) + ), + None, + ), + ( + "collector", + _deployment( + config_hash="176c01ff260e4716", + volumes=[ + {"name": "config", "configMap": {"name": "modelplane-collector"}}, + {"name": "credentials-vendor", "secret": {"secretName": "vendor-token"}}, + {"name": "credentials-prometheus", "secret": {"secretName": "prom-token"}}, + ], + volume_mounts=[ + {"name": "config", "mountPath": "/conf"}, + { + "name": "credentials-vendor", + "mountPath": "/etc/modelplane/telemetry/vendor", + "readOnly": True, + }, + { + "name": "credentials-prometheus", + "mountPath": "/etc/modelplane/telemetry/prometheus", + "readOnly": True, + }, + ], + env_from=[{"secretRef": {"name": "vendor-token"}}, {"secretRef": {"name": "prom-token"}}], + ), + "object.status.readyReplicas > 0", + ), + ], + ), +] + + +@pytest.mark.parametrize("case", OBJECTS_CASES, ids=lambda case: case.name) +def test_objects(case: ObjectsCase) -> None: + """objects composes the collector's manifests and their readiness queries.""" + got = collector.objects(case.cluster, case.mappings, case.sinks, case.extensions) + assert got == case.want, case.reason + + +EXPORTERS_CASES = [ + # The collector names a second instance of a component /. + ExportersCase( + name="TwoOfOneType", + reason="Two sinks of one type render as two exporters, each named for its sink.", + sinks=[ + _sink(name="a", type_="otlphttp", secret=None), + _sink(name="b", type_="otlphttp", secret=None), + ], + want={ + "otlphttp/a": {"endpoint": "https://otel.acme.example"}, + "otlphttp/b": {"endpoint": "https://otel.acme.example"}, + }, + ), + ExportersCase( + name="NoEndpoint", + reason="A sink that addresses its destination another way, by brokers or not at all, renders no endpoint.", + sinks=[ + tdv1alpha1.Sink.model_validate( + {"name": "bus", "type": "kafka", "config": {"brokers": ["kafka.acme.example:9092"]}} + ), + tdv1alpha1.Sink.model_validate({"name": "seen", "type": "debug"}), + ], + want={"kafka/bus": {"brokers": ["kafka.acme.example:9092"]}, "debug/seen": {}}, + ), + # The collector carries no credential on an exporter, only a reference to + # the authenticator AUTHENTICATORS_CASES composes. + ExportersCase( + name="Auth", + reason="A sink with auth references the authenticator composed for it.", + sinks=[_sink(name="primary", type_="otlphttp", secret="telemetry-credentials")], + want={ + "otlphttp/primary": { + "endpoint": "https://otel.acme.example", + "auth": {"authenticator": "bearertokenauth/primary"}, } - } - ) - _, _, datapoint, _ = collector.statements([mapping]) - joined = " ".join(datapoint) - assert r'"ab\"c"' in joined - assert r'"x\"y"' in joined + }, + ), + # The endpoint is Modelplane's, and goes on after the operator's config. + ExportersCase( + name="ConfigEndpoint", + reason="An endpoint in a sink's own config can't redirect it, though the rest of that config applies.", + sinks=[ + tdv1alpha1.Sink.model_validate( + { + "name": "primary", + "type": "otlphttp", + "endpoint": "https://otel.acme.example", + "config": {"endpoint": "https://elsewhere.example", "compression": "gzip"}, + } + ) + ], + want={"otlphttp/primary": {"endpoint": "https://otel.acme.example", "compression": "gzip"}}, + ), +] + + +@pytest.mark.parametrize("case", EXPORTERS_CASES, ids=lambda case: case.name) +def test_exporters(case: ExportersCase) -> None: + """exporters renders each sink as a collector exporter.""" + got = collector.exporters(case.sinks) + assert got == case.want, case.reason + + +AUTHENTICATORS_CASES = [ + AuthenticatorsCase( + name="BearerToken", + reason="A sink with a bearer token key gets an authenticator reading the token from its mounted Secret.", + sinks=[_sink(name="primary", type_="otlphttp", secret="telemetry-credentials")], + want={"bearertokenauth/primary": {"filename": "/etc/modelplane/telemetry/primary/token"}}, + ), +] + + +@pytest.mark.parametrize("case", AUTHENTICATORS_CASES, ids=lambda case: case.name) +def test_authenticators(case: AuthenticatorsCase) -> None: + """authenticators composes an extension for each sink that asks for auth.""" + got = collector.authenticators(case.sinks) + assert got == case.want, case.reason + + +def test_unit_factors() -> None: + """Every unit conversion factor is an OTTL float literal.""" + # scale_metric takes a float, and 1048576 is an integer to OTTL. The + # collector refuses to start on it - "must be a float" - which takes the + # whole cluster's telemetry down, and nothing short of running the + # collector catches it. + not_floats = [unit for unit, factor in collector._UNIT_FACTOR.items() if "." not in factor] + assert not_floats == [], "an OTTL float literal needs a decimal point" + for factor in collector._UNIT_FACTOR.values(): + float(factor) -def test_carrying_a_label_onto_itself_keeps_it() -> None: - """`from` equal to `name` is how a mapping remaps values in place. +def test_units_converted() -> None: + """Every unit the MetricMapping API offers has a conversion factor.""" + # A unit the API accepts with no conversion is a KeyError at render time. + annotation = mmv1alpha1.Metric.model_fields["fromUnit"].annotation + literal = next(a for a in typing.get_args(annotation) if typing.get_origin(a) is typing.Literal) + assert set(collector._UNIT_FACTOR) == set(typing.get_args(literal)) - The delete that stops a carried label costing twice the cardinality - would otherwise take the label the statements before it just set, and - the series would lose the label entirely. - """ - mapping = mmv1alpha1.MetricMapping.model_validate( - { - "spec": { - "metrics": [ - { - "from": "my_engine_finish", - "to": "modelplane_requests_total", - "labels": [{"name": "reason", "from": "reason", "values": {"eos": "stop"}}], - } - ] - } - } - ) - _, _, datapoint, _ = collector.statements([mapping]) - assert not [st for st in datapoint if st.startswith("delete_key")] - - -def test_a_value_rewrite_never_lands_in_the_metric_context() -> None: - """value_double is a datapoint path; the collector refuses to start on it here.""" - blocks = _config()["processors"]["transform/modelplane"]["metric_statements"] - metric_block = next(b for b in blocks if b["context"] == "metric") - assert not [st for st in metric_block["statements"] if "value_double" in st] - for block in blocks: - for st in block["statements"]: - assert "set(name," not in st - assert "set(value_double," not in st - - -def test_sglang_carries_no_queue_time_or_preemption() -> None: - """SGLang publishes neither, so there is nothing to rename onto them. - - Checked against a running SGLang v0.4.9.post2: it has no per-request - queue-time metric and no retraction counters at all. The nearest - thing, sglang:avg_request_queue_latency, is a gauge of the mean over - the last batch - a different measurement from vLLM's per-request - histogram, and one name holding both makes a fleet quantile - meaningless. - """ + +def test_metric_name_pattern() -> None: + """The MetricMapping schema rejects a quote in a metric name.""" + # A quote in `from` would end the OTTL comparison early, and rename + # whatever the rest of the line matched. + with pytest.raises(ValidationError, match="String should match pattern"): + mmv1alpha1.Metric.model_validate({"from": 'x" or true or name == "y', "to": "modelplane_x"}) + # This re-validates every built-in metric, read from the stacks package, and + # can't fail: stacks.metrics builds each one with Metric.model_validate at + # import, so a name the pattern rejected would fail this module's import + # before it got here. + builtin = [m for mapping in stacks.BUILTIN_MAPPINGS for m in mapping.spec.metrics] + got = [mmv1alpha1.Metric.model_validate({"from": m.from_, "to": m.to}).from_ for m in builtin] + assert got == [m.from_ for m in builtin] + + +def test_sglang_mappings() -> None: + """The SGLang built-in renames nothing onto queue time or preemption.""" + # SGLang publishes neither. Checked against a running SGLang v0.4.9.post2: + # it has no per-request queue-time metric and no retraction counters at + # all. The nearest thing, sglang:avg_request_queue_latency, is a gauge of + # the mean over the last batch - a different measurement from vLLM's + # per-request histogram, and one name holding both makes a fleet quantile + # meaningless. The built-in mappings are read from the stacks package, + # because they're what this checks. sglang = { m.from_: m.to for mapping in stacks.BUILTIN_MAPPINGS @@ -415,99 +1210,11 @@ def test_sglang_carries_no_queue_time_or_preemption() -> None: if m.from_.startswith("sglang:") } assert sglang, "the SGLang built-in went missing" - assert "modelplane_request_queue_seconds" not in sglang.values() - assert "modelplane_requests_preempted_total" not in sglang.values() - assert not [k for k in sglang if "retracted" in k or "queue_time" in k] - - -def test_sglang_latency_histograms_are_not_renamed() -> None: - """Their buckets resolve to 100ms where vLLM's resolve to 1ms.""" - joined = " ".join(_metric_statements()) - assert "sglang:time_to_first_token_seconds" not in joined - assert "sglang:inter_token_latency" not in joined - - -def test_config_hash_is_stable_across_processes() -> None: - """hash() is seeded per process, so it would redeploy on every reconcile.""" - first = _objects()["collector"]["spec"]["template"]["metadata"]["annotations"] - second = _objects()["collector"]["spec"]["template"]["metadata"]["annotations"] - assert first == second - assert re.search(r"^[0-9a-f]{16}$", first["modelplane.ai/config-hash"]) - - -def test_credentials_mount_as_a_file_and_an_environment_variable() -> None: - """A rotated token in an environment variable needs a restart to be read.""" - pod = _objects(secret="telemetry-credentials")["collector"]["spec"]["template"]["spec"] - assert "credentials-primary" in [v["name"] for v in pod["volumes"]] - assert pod["containers"][0]["envFrom"] == [{"secretRef": {"name": "telemetry-credentials"}}] - - -def test_each_sink_gets_its_own_credential_directory() -> None: - """Two sinks can both hold a key called token, and neither reads the other's.""" - sinks = [ - _sink(name="vendor", secret="vendor-token"), - _sink(name="prometheus", type_="prometheusremotewrite", secret="prom-token"), - ] - pod = {k: m for k, m, _ in collector.objects("prod-us-east", list(stacks.BUILTIN_MAPPINGS), sinks, _EXTENSIONS)}[ - "collector" - ]["spec"]["template"]["spec"] - mounts = {m["name"]: m["mountPath"] for m in pod["containers"][0]["volumeMounts"]} - assert mounts["credentials-vendor"] == "/etc/modelplane/telemetry/vendor" - assert mounts["credentials-prometheus"] == "/etc/modelplane/telemetry/prometheus" - - -def test_two_sinks_of_one_type_do_not_collide() -> None: - """The collector names a second instance of a component /.""" - rendered = collector.exporters([_sink(name="a"), _sink(name="b")]) - assert sorted(rendered) == ["otlphttp/a", "otlphttp/b"] - - -def test_a_sink_that_addresses_its_destination_another_way() -> None: - """Kafka takes brokers, the debug exporter nothing; neither has an endpoint.""" - sinks = [ - tdv1alpha1.Sink.model_validate( - {"name": "bus", "type": "kafka", "config": {"brokers": ["kafka.acme.example:9092"]}} - ), - tdv1alpha1.Sink.model_validate({"name": "seen", "type": "debug"}), - ] - rendered = collector.exporters(sinks) - assert "endpoint" not in rendered["kafka/bus"] - assert rendered["kafka/bus"]["brokers"] == ["kafka.acme.example:9092"] - assert rendered["debug/seen"] == {} - - -def test_auth_composes_its_own_authenticator() -> None: - """The collector carries no credential on an exporter, only a reference.""" - sink = _sink(secret="telemetry-credentials") - assert collector.authenticators([sink]) == { - "bearertokenauth/primary": {"filename": "/etc/modelplane/telemetry/primary/token"} + unpublished = { + source: target + for source, target in sglang.items() + if target in {"modelplane_request_queue_seconds", "modelplane_requests_preempted_total"} + or "retracted" in source + or "queue_time" in source } - assert collector.exporters([sink])["otlphttp/primary"]["auth"] == {"authenticator": "bearertokenauth/primary"} - - -def test_a_sinks_own_config_cannot_redirect_it() -> None: - """The endpoint is Modelplane's, and goes on after the operator's config.""" - sink = tdv1alpha1.Sink.model_validate( - { - "name": "primary", - "type": "otlphttp", - "endpoint": "https://otel.acme.example", - "config": {"endpoint": "https://elsewhere.example", "compression": "gzip"}, - } - ) - rendered = collector.exporters([sink])["otlphttp/primary"] - assert rendered["endpoint"] == "https://otel.acme.example" - assert rendered["compression"] == "gzip" - - -def test_no_secret_mounts_nothing() -> None: - pod = _objects()["collector"]["spec"]["template"]["spec"] - assert [v["name"] for v in pod["volumes"]] == ["config"] - assert "envFrom" not in pod["containers"][0] - - -def test_rbac_is_read_only() -> None: - """Service discovery needs to list pods, and nothing needs to write.""" - rules = _objects()["collector-clusterrole"]["rules"] - verbs = {v for r in rules for v in r["verbs"]} - assert verbs == {"get", "list", "watch"} + assert unpublished == {}, "SGLang publishes no queue time or preemption to rename" diff --git a/functions/compose-serving-stack/tests/test_fn.py b/functions/compose-serving-stack/tests/test_fn.py index 1529e3f2b..fa9a76625 100644 --- a/functions/compose-serving-stack/tests/test_fn.py +++ b/functions/compose-serving-stack/tests/test_fn.py @@ -14,17 +14,20 @@ """Tests for the compose-serving-stack function. -Two layers. The Case table compares whole RunFunctionResponses for the -Existing/Dynamo stack across the reconcile passes; its expectations are -built from the provider models with literal arguments typed here, never -from the stacks package, so a stack-data change shows up as a test diff. -The golden inventory then pins the composed-resource key set - the -identity contract; renaming a key deletes and recreates the remote -resource - for every cloud and stack, as frozen literals. +Three tables. COMPOSE_CASES compares whole RunFunctionResponses: the +Existing/Dynamo stack across the reconcile passes, a non-GCP identity secret, +the Existing/Standard stack's gateway with and without a client CA, Civo's +per-pool NVLink disable, and the telemetry collector a GKE stack composes for +its TelemetryDestinations. Its expectations are literals typed here, never read +from the stacks package, so a stack-data change shows up as a test diff. Only +the vendored CRD bundles are read from their files. +COMPOSED_RESOURCE_KEYS_CASES then pins the composed-resource key set - the +identity contract; renaming a key deletes and recreates the remote resource - +for every cloud and stack. CLUSTER_NAME_CASES pins the cluster name every +exported series is stamped with. """ import asyncio -import copy import dataclasses import json import pathlib @@ -33,1637 +36,7832 @@ import yaml from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 -from function import fn +from function import fn, stacks from google.protobuf import duration_pb2 as durationpb from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.servingstack import v1alpha1 -from models.io.crossplane.m.helm.providerconfig import v1beta1 as helmpcv1beta1 -from models.io.crossplane.m.helm.release import v1beta1 as helmv1beta1 -from models.io.crossplane.m.kubernetes.object import v1alpha1 as k8sobjv1alpha1 -from models.io.crossplane.m.kubernetes.providerconfig import ( - v1alpha1 as k8spcv1alpha1, -) -from models.io.crossplane.protection.usage import v1beta1 as usagev1beta1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -# Precomputed child_name value for test-backend. -_PC_NAME = "test-backend-cluster-63fde" - -_RELEASE_REF = ("helm.m.crossplane.io/v1beta1", "Release") -_OBJECT_REF = ("kubernetes.m.crossplane.io/v1alpha1", "Object") - -_GATEWAY_READY_CEL = "has(object.status.addresses) && object.status.addresses.size() > 0" -_CERTIFICATE_READY_CEL = ( - "has(object.status) && has(object.status.conditions) && " - "object.status.conditions.exists(c, c.type == 'Ready' && c.status == 'True')" -) -_BUNDLE_SYNCED_CEL = ( - "has(object.status) && has(object.status.conditions) && " - "object.status.conditions.exists(c, c.type == 'Synced' && c.status == 'True')" -) -_POLICY_ACCEPTED_CEL = ( - "has(object.status) && has(object.status.ancestors) && " - "object.status.ancestors.exists(a, has(a.conditions) && " - "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" -) -_MODELEXPRESS_READY_CEL = ( - 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")' -) - -# The name InferenceGateways reach the test stack's gateway by, and one -# InferenceGateway's client CA for it to trust. With both, the gateway serves. -_GATEWAY_HOSTNAME = "test-backend.gateways.example.com" -_CLIENT_CA = "-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n" - -# Resolve the vendored CRD bundles via the installed function package: -# the sandboxed test check runs against the venv's copy, not the tree. -_CRDS_DIR = pathlib.Path(fn.__file__).parent / "stacks" / "crds" - - -def _crds(filename: str) -> list[dict]: - """The CRDs a vendored bundle carries, content straight from the file.""" - return [ - doc - for doc in yaml.safe_load_all((_CRDS_DIR / filename).read_text()) - if doc and doc.get("kind") == "CustomResourceDefinition" - ] +@dataclasses.dataclass +class ComposeCase: + """A test case for RunFunction's whole response.""" -def _stack(labels: dict[str, str] | None) -> v1alpha1.ServingStack: - return v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="local-serving-stack-d4206", labels=labels), - spec=v1alpha1.Spec( - cloud="Existing", - secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], - gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), - ), - ) + name: str + reason: str + req: fnv1.RunFunctionRequest + want: fnv1.RunFunctionResponse -def test_it_is_the_composite_an_operator_named() -> None: - """A ServingStack's own name is generated and carries a suffix.""" - assert fn._cluster_name(_stack({"crossplane.io/composite": "local"})) == "local" - - -def test_it_falls_back_to_the_stack() -> None: - """Better a generated name on the series than none at all.""" - assert fn._cluster_name(_stack(None)) == "local-serving-stack-d4206" - - -def _request(cloud: str, stack: str, observed: dict | None = None) -> fnv1.RunFunctionRequest: - """Build a RunFunctionRequest for a test-backend ServingStack.""" - return fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud=cloud, # ty: ignore[invalid-argument-type] # cases pass values of the literal - stack=stack, # ty: ignore[invalid-argument-type] - secrets=[ - v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), - v1alpha1.Secret( - type="GoogleApplicationCredentials", name="sa-secret", key="private_key" - ), - ], - gateway=v1alpha1.Gateway( - hostname=_GATEWAY_HOSTNAME, - clientCAs=[v1alpha1.ClientCA(name="eu", certificate=_CLIENT_CA)], - ), - ), - ).model_dump(exclude_none=True, mode="json") - ), - ), - resources=observed or {}, - ), +@dataclasses.dataclass +class ComposedResourceKeysCase: + """A test case for the composed-resource keys RunFunction renders.""" + + name: str + reason: str + req: fnv1.RunFunctionRequest + want: set[str] + + +@dataclasses.dataclass +class ClusterNameCase: + """A test case for fn._cluster_name.""" + + name: str + reason: str + xr: v1alpha1.ServingStack + want: str + + +def _crd(*, filename: str, name: str) -> dict: + """The CRD named name in the vendored bundle filename, as the file has it.""" + # The vendored CRD bundles are upstream release artifacts, a thousand lines + # of schema, so the expectations read them rather than restating them. They + # resolve via the installed function package, because the sandboxed test + # check runs against the venv's copy, not the tree. + bundle = pathlib.Path(fn.__file__).parent / "stacks" / "crds" / filename + return next( + doc + for doc in yaml.safe_load_all(bundle.read_text()) + if doc and doc["kind"] == "CustomResourceDefinition" and doc["metadata"]["name"] == name ) -def _release( - key: str, - release: str, - namespace: str, - chart: str, - repository: str, - version: str, - values: dict | None = None, +def _serving_stack( *, - wait: bool = False, + cloud: stacks.Cloud, + stack: stacks.Stack, + secrets: list[v1alpha1.Secret], + gateway: v1alpha1.Gateway, + gpu: v1alpha1.Gpu | None, ) -> fnv1.Resource: - """The expected Release for a Chart entry, built from literal arguments.""" - model = helmv1beta1.Release( - metadata=metav1.ObjectMeta( - annotations={"crossplane.io/external-name": release}, - labels={"modelplane.ai/resource": key}, - ), - spec=helmv1beta1.Spec( - providerConfigRef=helmv1beta1.ProviderConfigRef(kind="ProviderConfig", name=_PC_NAME), - forProvider=helmv1beta1.ForProvider( - chart=helmv1beta1.Chart(name=chart, repository=repository, version=version), - namespace=namespace, - ), - ), + """The observed ServingStack, test-backend in namespace test-ns.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ServingStack( + metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), + spec=v1alpha1.Spec(cloud=cloud, stack=stack, secrets=secrets, gateway=gateway, gpu=gpu), + ).model_dump(exclude_none=True, mode="json", by_alias=True) + ) ) - if wait: - model.spec.forProvider.wait = True - model.spec.forProvider.waitTimeout = "10m" - if values: - model.spec.forProvider.values = values - res = fnv1.Resource() - resource.update(res, model) - return res - - -def _object( - key: str, - manifest: dict, - cel: str | None = None, - *, - labeled: bool = True, - management_policies: list | None = None, -) -> fnv1.Resource: - """The expected Object for one manifest, built from literal arguments. - The gateway PKI objects carry no resource label (nothing selects them in - a Usage), so labeled=False builds them without one. - """ - model = k8sobjv1alpha1.Object( - # Omit metadata entirely when unlabeled: a null metadata would serialize - # into the composed resource rather than being absent, as it is when the - # function passes none. - **({"metadata": metav1.ObjectMeta(labels={"modelplane.ai/resource": key})} if labeled else {}), - spec=k8sobjv1alpha1.Spec( - providerConfigRef=k8sobjv1alpha1.ProviderConfigRef(kind="ProviderConfig", name=_PC_NAME), - forProvider=k8sobjv1alpha1.ForProvider(manifest=manifest), - ), - ) - if management_policies: - model.spec.managementPolicies = management_policies - if cel is not None: - model.spec.readiness = k8sobjv1alpha1.Readiness(policy="DeriveFromCelQuery", celQuery=cel) - res = fnv1.Resource() - resource.update(res, model) - return res - - -def _usage(of_ref: tuple[str, str], of_key: str, by_ref: tuple[str, str], by_key: str) -> fnv1.Resource: - """The expected teardown Usage for one dependency edge, ready on arrival.""" - res = fnv1.Resource() - resource.update( - res, - usagev1beta1.Usage( - spec=usagev1beta1.Spec( - of=usagev1beta1.Of( - apiVersion=of_ref[0], - kind=of_ref[1], - resourceSelector=usagev1beta1.ResourceSelectorModel( - matchControllerRef=True, - matchLabels={"modelplane.ai/resource": of_key}, - ), - ), - by=usagev1beta1.By( - apiVersion=by_ref[0], - kind=by_ref[1], - resourceSelector=usagev1beta1.ResourceSelector( - matchControllerRef=True, - matchLabels={"modelplane.ai/resource": by_key}, - ), - ), - replayDeletion=True, - ), - ), - ) - res.ready = fnv1.READY_TRUE - return res +def _desired_serving_stack(*, gateway: dict | None) -> fnv1.Resource: + """The desired ServingStack, publishing gateway in its status if there is one.""" + status = {} if gateway is None else {"gateway": gateway} + return fnv1.Resource(resource=resource.dict_to_struct({"status": status})) -def _provider_configs(*, ready: bool = True) -> dict[str, fnv1.Resource]: - """The two expected ProviderConfigs. - Ready only once observed: on the first pass they and the Usages are - the whole desired state, and ready-on-arrival would let the - composite report Ready before any stack component exists. - """ - k8s = fnv1.Resource() - resource.update( - k8s, - k8spcv1alpha1.ProviderConfig( - metadata=metav1.ObjectMeta(name=_PC_NAME), - spec=k8spcv1alpha1.Spec( - credentials=k8spcv1alpha1.Credentials( - source="Secret", - secretRef=k8spcv1alpha1.SecretRef(name="kube-secret", namespace="test-ns", key="kubeconfig"), - ), - identity=k8spcv1alpha1.Identity( - type="GoogleApplicationCredentials", - source="Secret", - secretRef=k8spcv1alpha1.SecretRef(name="sa-secret", namespace="test-ns", key="private_key"), - ), - ), - ), - ) - if ready: - k8s.ready = fnv1.READY_TRUE - helm = fnv1.Resource() - resource.update( - helm, - helmpcv1beta1.ProviderConfig( - metadata=metav1.ObjectMeta(name=_PC_NAME), - spec=helmpcv1beta1.Spec( - credentials=helmpcv1beta1.Credentials( - source="Secret", - secretRef=helmpcv1beta1.SecretRef(name="kube-secret", namespace="test-ns", key="kubeconfig"), - ), - identity=helmpcv1beta1.Identity( - type="GoogleApplicationCredentials", - source="Secret", - secretRef=helmpcv1beta1.SecretRef(name="sa-secret", namespace="test-ns", key="private_key"), - ), - ), - ), +def _observed_ready() -> fnv1.Resource: + """An observed composed resource whose Ready condition is True.""" + return fnv1.Resource( + resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) ) - if ready: - helm.ready = fnv1.READY_TRUE - return {"provider-config-kubernetes": k8s, "provider-config-helm": helm} -def _observed_pcs() -> dict[str, fnv1.Resource]: - """Observed ProviderConfigs, which gate the rest of the stack open.""" - return { - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - {"apiVersion": "kubernetes.m.crossplane.io/v1alpha1", "kind": "ProviderConfig"} - ) - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct({"apiVersion": "helm.m.crossplane.io/v1beta1", "kind": "ProviderConfig"}) - ), - } +def _observed_kubernetes_provider_config() -> fnv1.Resource: + """The observed provider-kubernetes ProviderConfig, which has no conditions.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + {"apiVersion": "kubernetes.m.crossplane.io/v1alpha1", "kind": "ProviderConfig"} + ) + ) -# The Usages every Existing/Dynamo pass composes: the two hand-written -# gateway-chain edges, and one derived edge per depends_on in the joined -# stack data. -_EXISTING_DYNAMO_USAGES = { - "usage-gateway-class-by-gateway": _usage(_OBJECT_REF, "gateway-class", _OBJECT_REF, "gateway"), - "usage-envoy-gateway-by-gateway-class": _usage(_RELEASE_REF, "envoy-gateway", _OBJECT_REF, "gateway-class"), - "usage-cert-manager-by-envoy-gateway": _usage(_RELEASE_REF, "cert-manager", _RELEASE_REF, "envoy-gateway"), - "usage-ai-gateway-crds-by-ai-gateway": _usage(_RELEASE_REF, "ai-gateway-crds", _RELEASE_REF, "ai-gateway"), - "usage-gateway-namespace-by-gateway-proxy": _usage(_OBJECT_REF, "gateway-namespace", _OBJECT_REF, "gateway-proxy"), - "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( - _RELEASE_REF, "cert-manager", _OBJECT_REF, "gateway-selfsigned-issuer" - ), - "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( - _OBJECT_REF, "gateway-namespace", _OBJECT_REF, "gateway-selfsigned-issuer" - ), - "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( - _OBJECT_REF, "gateway-selfsigned-issuer", _RELEASE_REF, "trust-manager" - ), - "usage-kai-scheduler-by-kai-queue-root": _usage(_RELEASE_REF, "kai-scheduler", _OBJECT_REF, "kai-queue-root"), - "usage-kai-scheduler-by-kai-queue": _usage(_RELEASE_REF, "kai-scheduler", _OBJECT_REF, "kai-queue"), - "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _usage( - _OBJECT_REF, "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", _OBJECT_REF, "modelexpress-server" - ), - "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _usage( - _OBJECT_REF, "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", _OBJECT_REF, "modelexpress-server" - ), -} +def _observed_helm_provider_config() -> fnv1.Resource: + """The observed provider-helm ProviderConfig, which has no conditions.""" + return fnv1.Resource( + resource=resource.dict_to_struct({"apiVersion": "helm.m.crossplane.io/v1beta1", "kind": "ProviderConfig"}) + ) -def _kai_queue(name: str, parent: str | None) -> dict: +def _kubernetes_provider_config(*, identity: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The composed provider-kubernetes ProviderConfig, authenticating as identity if there is one.""" spec: dict = { - "resources": { - "cpu": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, - "gpu": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, - "memory": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, + "credentials": { + "source": "Secret", + "secretRef": {"name": "kube-secret", "namespace": "test-ns", "key": "kubeconfig"}, }, } - if parent: - spec["parentQueue"] = parent - return {"apiVersion": "scheduling.run.ai/v2", "kind": "Queue", "metadata": {"name": name}, "spec": spec} - - -_MX_META = {"name": "modelexpress-server", "namespace": "default"} -_MX_SELECT = {"modelplane.ai/modelexpress": "modelexpress-server"} - - -def _existing_dynamo_stack() -> dict[str, fnv1.Resource]: - """Every component the Existing/Dynamo stack renders, as literals.""" - out: dict[str, fnv1.Resource] = {} - - # --- the Existing cloud half (hand-written Modelplane pins) --- - out["cert-manager"] = _release( - key="cert-manager", - release="mp-cert-manager", - namespace="cert-manager", - chart="cert-manager", - repository="https://charts.jetstack.io", - version="v1.20.2", - wait=True, - # clusterResourceNamespace and enableCertificateOwnerRef are forced by - # fn._helm_release for every cloud's cert-manager: the ClusterIssuer CA - # lives in modelplane-system, and a deleted ModelRoute's client - # certificate Secret must go with its Certificate. - values={ - "crds": {"enabled": True}, - "clusterResourceNamespace": "modelplane-system", - "enableCertificateOwnerRef": True, - }, - ) - out["kube-prometheus-stack"] = _release( - key="kube-prometheus-stack", - release="mp-kube-prometheus-stack", - namespace="monitoring", - chart="kube-prometheus-stack", - repository="https://prometheus-community.github.io/helm-charts", - version="84.4.0", - values={ - "fullnameOverride": "prometheus", - "prometheus": { - "prometheusSpec": { - "podMonitorSelectorNilUsesHelmValues": False, - "podMonitorNamespaceSelector": {}, - "additionalScrapeConfigs": [ - { - "job_name": "envoy-gateway-proxy", - "kubernetes_sd_configs": [ - {"role": "pod", "namespaces": {"names": ["envoy-gateway-system"]}}, - ], - "relabel_configs": [ - { - "source_labels": [ - "__meta_kubernetes_pod_label_app_kubernetes_io_component", - ], - "action": "keep", - "regex": "proxy", - }, - { - "source_labels": ["__address__"], - "action": "replace", - "regex": "([^:]+)(?::\\d+)?", - "replacement": "$1:19001", - "target_label": "__address__", - }, - ], - "metrics_path": "/stats/prometheus", - }, - ], - }, - }, - "grafana": {"enabled": False}, - "alertmanager": {"enabled": False}, - }, - ) - out["node-feature-discovery"] = _release( - key="node-feature-discovery", - release="mp-node-feature-discovery", - namespace="node-feature-discovery", - chart="node-feature-discovery", - repository="https://kubernetes-sigs.github.io/node-feature-discovery/charts", - version="0.19.0", - values={ - "worker": { - "tolerations": [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}], - }, - }, + if identity is not None: + spec["identity"] = identity + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ProviderConfig", + "metadata": {"name": "test-backend-cluster-63fde"}, + "spec": spec, + } + ), + ready=ready, ) - out["nvidia-dra-driver-gpu"] = _release( - key="nvidia-dra-driver-gpu", - release="mp-dra-driver-nvidia-gpu", - namespace="nvidia-dra-driver", - chart="dra-driver-nvidia-gpu", - repository="oci://registry.k8s.io/dra-driver-nvidia/charts", - version="0.4.1", - values={ - "gpuResourcesEnabledOverride": True, - "resources": {"computeDomains": {"enabled": False}}, + + +def _helm_provider_config(*, identity: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The composed provider-helm ProviderConfig, authenticating as identity if there is one.""" + spec: dict = { + "credentials": { + "source": "Secret", + "secretRef": {"name": "kube-secret", "namespace": "test-ns", "key": "kubeconfig"}, }, + } + if identity is not None: + spec["identity"] = identity + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "metadata": {"name": "test-backend-cluster-63fde"}, + "spec": spec, + } + ), + ready=ready, ) - # --- the common half --- - out["envoy-gateway"] = _release( - key="envoy-gateway", - release="mp-gateway-helm", - namespace="envoy-gateway-system", - chart="gateway-helm", - repository="oci://docker.io/envoyproxy", - version="v1.8.4", - values={ - "config": { - "envoyGateway": { - "extensionApis": {"enableBackend": True}, - "extensionManager": { - "hooks": { - "xdsTranslator": { - "translation": { - "listener": {"includeAll": True}, - "route": {"includeAll": True}, - "cluster": {"includeAll": True}, - "secret": {"includeAll": True}, - }, - "post": ["Translation", "Cluster", "Route"], - }, + +def _cert_manager(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed cert-manager Release the hand-written clouds pin, as the Existing and Civo cases compose it. + + GKE's generated half pins its own, _gke_cert_manager. + """ + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-cert-manager"}, + "labels": {"modelplane.ai/resource": "cert-manager"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "cert-manager", + "repository": "https://charts.jetstack.io", + "version": "v1.20.2", }, - "service": { - "fqdn": { - "hostname": "ai-gateway-controller.envoy-ai-gateway-system.svc.cluster.local", - "port": 1063, - }, + "namespace": "cert-manager", + "wait": True, + "waitTimeout": "10m", + # clusterResourceNamespace and enableCertificateOwnerRef are forced by + # fn._helm_release for every cloud's cert-manager: the ClusterIssuer CA + # lives in modelplane-system, and a deleted ModelRoute's client + # certificate Secret must go with its Certificate. + "values": { + "crds": {"enabled": True}, + "clusterResourceNamespace": "modelplane-system", + "enableCertificateOwnerRef": True, }, - "backendResources": [ - {"group": "inference.networking.k8s.io", "kind": "InferencePool", "version": "v1"}, - ], }, }, - }, - }, - ) - out["ai-gateway-crds"] = _release( - key="ai-gateway-crds", - release="mp-ai-gateway-crds-helm", - namespace="envoy-ai-gateway-system", - chart="ai-gateway-crds-helm", - repository="oci://docker.io/envoyproxy", - version="v1.1.0", - wait=True, - ) - out["ai-gateway"] = _release( - key="ai-gateway", - release="mp-ai-gateway-helm", - namespace="envoy-ai-gateway-system", - chart="ai-gateway-helm", - repository="oci://docker.io/envoyproxy", - version="v1.1.0", - values={"controller": {"logRequestHeaderAttributes": "x-modelplane-caller:caller"}}, - ) - for doc in _crds("gaie.yaml"): - key = f"gaie-crds-{doc['metadata']['name']}" - out[key] = _object(key, doc) - out["gateway-namespace"] = _object( - "gateway-namespace", - { - "apiVersion": "v1", - "kind": "Namespace", - "metadata": {"name": "modelplane-system", "labels": {"modelplane.ai/namespace": "modelplane-system"}}, - }, + } + ), + ready=ready, ) - out["gateway-proxy"] = _object( - "gateway-proxy", - { - "apiVersion": "gateway.envoyproxy.io/v1alpha1", - "kind": "EnvoyProxy", - "metadata": {"name": "cluster-gateway", "namespace": "modelplane-system"}, - "spec": { - "provider": { - "type": "Kubernetes", - "kubernetes": {"envoyService": {"externalTrafficPolicy": "Cluster"}}, + + +def _kube_prometheus_stack(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed kube-prometheus-stack Release.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-kube-prometheus-stack"}, + "labels": {"modelplane.ai/resource": "kube-prometheus-stack"}, }, - }, - }, - ) - out["dra-driver-critical-pods-quota"] = _object( - "dra-driver-critical-pods-quota", - { - "apiVersion": "v1", - "kind": "ResourceQuota", - "metadata": {"name": "allow-critical-pods", "namespace": "nvidia-dra-driver"}, - "spec": { - "hard": {"pods": "1000"}, - "scopeSelector": { - "matchExpressions": [ - { - "operator": "In", - "scopeName": "PriorityClass", - "values": ["system-node-critical", "system-cluster-critical"], + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "kube-prometheus-stack", + "repository": "https://prometheus-community.github.io/helm-charts", + "version": "84.4.0", }, - ], + "namespace": "monitoring", + "values": { + "fullnameOverride": "prometheus", + "prometheus": { + "prometheusSpec": { + "podMonitorSelectorNilUsesHelmValues": False, + "podMonitorNamespaceSelector": {}, + "additionalScrapeConfigs": [ + { + "job_name": "envoy-gateway-proxy", + "kubernetes_sd_configs": [ + { + "role": "pod", + "namespaces": {"names": ["envoy-gateway-system"]}, + } + ], + "relabel_configs": [ + { + "source_labels": [ + "__meta_kubernetes_pod_label_app_kubernetes_io_component" + ], + "action": "keep", + "regex": "proxy", + }, + { + "source_labels": ["__address__"], + "action": "replace", + "regex": "([^:]+)(?::\\d+)?", + "replacement": "$1:19001", + "target_label": "__address__", + }, + ], + "metrics_path": "/stats/prometheus", + } + ], + } + }, + "grafana": {"enabled": False}, + "alertmanager": {"enabled": False}, + }, + }, }, - }, - }, + } + ), + ready=ready, ) - # --- the Dynamo half --- - out["grove"] = _release( - key="grove", - release="mp-grove-charts", - namespace="grove-system", - chart="grove-charts", - repository="oci://ghcr.io/ai-dynamo/grove", - version="v0.1.0-alpha.12-rc2", - ) - out["kai-scheduler"] = _release( - key="kai-scheduler", - release="mp-kai-scheduler", - namespace="kai-scheduler", - chart="kai-scheduler", - repository="oci://ghcr.io/kai-scheduler/kai-scheduler", - version="v0.16.8", - wait=True, - ) - out["kai-queue-root"] = _object("kai-queue-root", _kai_queue("modelplane-root", None)) - out["kai-queue"] = _object("kai-queue", _kai_queue("modelplane", "modelplane-root")) - for doc in _crds("modelexpress.yaml"): - key = f"modelexpress-crds-{doc['metadata']['name']}" - out[key] = _object(key, doc) - out["modelexpress-server-sa"] = _object( - "modelexpress-server-sa", - {"apiVersion": "v1", "kind": "ServiceAccount", "metadata": _MX_META}, - ) - out["modelexpress-server-role"] = _object( - "modelexpress-server-role", - { - "apiVersion": "rbac.authorization.k8s.io/v1", - "kind": "Role", - "metadata": _MX_META, - "rules": [ - { - "apiGroups": ["modelexpress.nvidia.com"], - "resources": ["modelmetadatas", "modelmetadatas/status"], - "verbs": ["get", "list", "create", "update", "patch", "delete"], - }, - { - "apiGroups": [""], - "resources": ["configmaps"], - "verbs": ["get", "list", "create", "update", "patch", "delete"], - }, - { - "apiGroups": ["modelexpress.nvidia.com"], - "resources": ["modelcacheentries", "modelcacheentries/status"], - "verbs": ["get", "list", "create", "update", "patch", "delete"], - }, - ], - }, - ) - out["modelexpress-server-rolebinding"] = _object( - "modelexpress-server-rolebinding", - { - "apiVersion": "rbac.authorization.k8s.io/v1", - "kind": "RoleBinding", - "metadata": _MX_META, - "subjects": [{"kind": "ServiceAccount", "name": "modelexpress-server", "namespace": "default"}], - "roleRef": {"apiGroup": "rbac.authorization.k8s.io", "kind": "Role", "name": "modelexpress-server"}, + +def _node_feature_discovery(*, ready: fnv1.Ready, wait: bool) -> fnv1.Resource: + """The composed node-feature-discovery Release the hand-written clouds pin, waiting for health if wait is set. + + Civo's waits and Existing's doesn't. GKE's generated half pins its own, + _gke_node_feature_discovery. + """ + for_provider: dict = { + "chart": { + "name": "node-feature-discovery", + "repository": "https://kubernetes-sigs.github.io/node-feature-discovery/charts", + "version": "0.19.0", }, - ) - out["modelexpress-server-svc"] = _object( - "modelexpress-server-svc", - { - "apiVersion": "v1", - "kind": "Service", - "metadata": _MX_META, - "spec": { - "selector": _MX_SELECT, - "ports": [{"name": "grpc", "port": 8001, "targetPort": 8001}], - }, + "namespace": "node-feature-discovery", + "values": { + "worker": {"tolerations": [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}]} }, - ) - out["modelexpress-server"] = _object( - "modelexpress-server", - { - "apiVersion": "apps/v1", - "kind": "Deployment", - "metadata": _MX_META, - "spec": { - "replicas": 1, - "selector": {"matchLabels": _MX_SELECT}, - "template": { - "metadata": {"labels": _MX_SELECT}, - "spec": { - "serviceAccountName": "modelexpress-server", - "containers": [ - { - "name": "modelexpress-server", - "image": "nvcr.io/nvidia/ai-dynamo/modelexpress-server:0.4.1", - "ports": [{"containerPort": 8001}], - "env": [ - {"name": "MODEL_EXPRESS_CACHE_DIRECTORY", "value": "/mnt/models"}, - {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, - {"name": "MX_METADATA_BACKEND", "value": "kubernetes"}, - { - "name": "POD_NAMESPACE", - "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, - }, - ], - "volumeMounts": [{"name": "cache", "mountPath": "/mnt/models"}], - "readinessProbe": {"tcpSocket": {"port": 8001}, "periodSeconds": 10}, - "livenessProbe": {"tcpSocket": {"port": 8001}, "periodSeconds": 20}, - }, - ], - "volumes": [{"name": "cache", "emptyDir": {}}], - }, + } + if wait: + for_provider |= {"wait": True, "waitTimeout": "10m"} + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-node-feature-discovery"}, + "labels": {"modelplane.ai/resource": "node-feature-discovery"}, }, - }, - }, - cel=_MODELEXPRESS_READY_CEL, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": for_provider, + }, + } + ), + ready=ready, ) - # --- the hand-rendered gateway pair --- - out["gateway-class"] = _object( - "gateway-class", - { - "apiVersion": "gateway.networking.k8s.io/v1", - "kind": "GatewayClass", - "metadata": {"name": "envoy"}, - "spec": { - "controllerName": "gateway.envoyproxy.io/gatewayclass-controller", - "parametersRef": { - "group": "gateway.envoyproxy.io", - "kind": "EnvoyProxy", - "name": "cluster-gateway", - "namespace": "modelplane-system", + +def _nvidia_dra_driver_gpu(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed NVIDIA GPU DRA driver Release as Existing pins it, at the chart's default driver root. + + Civo's points at the gpu-operator's driver root, and is written inline in + NvLinkEnabled. + """ + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-dra-driver-nvidia-gpu"}, + "labels": {"modelplane.ai/resource": "nvidia-dra-driver-gpu"}, }, - }, - }, - ) - out["gateway"] = _object( - "gateway", - { - "apiVersion": "gateway.networking.k8s.io/v1", - "kind": "Gateway", - "metadata": {"name": "cluster-gateway", "namespace": "modelplane-system"}, - "spec": { - "gatewayClassName": "envoy", - "listeners": [ - { - "name": "https", - "protocol": "HTTPS", - "port": 443, - "hostname": _GATEWAY_HOSTNAME, - "tls": {"mode": "Terminate", "certificateRefs": [{"name": "cluster-gateway-serving"}]}, - "allowedRoutes": { - "namespaces": { - "from": "Selector", - "selector": { - "matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}] - }, - } + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "dra-driver-nvidia-gpu", + "repository": "oci://registry.k8s.io/dra-driver-nvidia/charts", + "version": "0.4.1", + }, + "namespace": "nvidia-dra-driver", + "values": { + "gpuResourcesEnabledOverride": True, + "resources": {"computeDomains": {"enabled": False}}, }, }, - ], - }, - }, - cel=_GATEWAY_READY_CEL, + }, + } + ), + ready=ready, ) - # --- the cluster gateway's PKI, issued for its hostname --- - out["gateway-ca-certificate"] = _object( - "gateway-ca-certificate", - { - "apiVersion": "cert-manager.io/v1", - "kind": "Certificate", - "metadata": {"name": "modelplane-cluster-ca", "namespace": "modelplane-system"}, - "spec": { - "isCA": True, - "commonName": f"modelplane cluster CA {_GATEWAY_HOSTNAME}", - "secretName": "modelplane-cluster-ca", - "duration": "87600h", - "renewBefore": "8760h", - "privateKey": {"algorithm": "ECDSA", "size": 256}, - "issuerRef": {"name": "modelplane-selfsigned", "kind": "Issuer", "group": "cert-manager.io"}, - }, - }, - cel=_CERTIFICATE_READY_CEL, - labeled=False, - ) - out["gateway-ca-issuer"] = _object( - "gateway-ca-issuer", - { - "apiVersion": "cert-manager.io/v1", - "kind": "Issuer", - "metadata": {"name": "modelplane-cluster-ca", "namespace": "modelplane-system"}, - "spec": {"ca": {"secretName": "modelplane-cluster-ca"}}, - }, - labeled=False, - ) - out["gateway-serving-certificate"] = _object( - "gateway-serving-certificate", - { - "apiVersion": "cert-manager.io/v1", - "kind": "Certificate", - "metadata": {"name": "cluster-gateway-serving", "namespace": "modelplane-system"}, - "spec": { - "secretName": "cluster-gateway-serving", - "dnsNames": [_GATEWAY_HOSTNAME], - "duration": "2160h", - "renewBefore": "720h", - "privateKey": {"algorithm": "ECDSA", "size": 256, "rotationPolicy": "Always"}, - "issuerRef": {"name": "modelplane-cluster-ca", "kind": "Issuer", "group": "cert-manager.io"}, - }, - }, - cel=_CERTIFICATE_READY_CEL, - labeled=False, - ) - out["gateway-ca-bundle"] = _object( - "gateway-ca-bundle", - { - "apiVersion": "trust.cert-manager.io/v1alpha1", - "kind": "Bundle", - "metadata": {"name": "modelplane-cluster-ca"}, - "spec": { - "sources": [{"secret": {"name": "modelplane-cluster-ca", "key": "ca.crt"}}], - "target": { - "configMap": {"key": "ca.crt"}, - "namespaceSelector": {"matchLabels": {"kubernetes.io/metadata.name": "modelplane-system"}}, + +def _ai_gateway_crds(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Envoy AI Gateway CRDs Release.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-ai-gateway-crds-helm"}, + "labels": {"modelplane.ai/resource": "ai-gateway-crds"}, }, - }, - }, - cel=_BUNDLE_SYNCED_CEL, - labeled=False, - ) - out["gateway-ca-configmap"] = _object( - "gateway-ca-configmap", - { - "apiVersion": "v1", - "kind": "ConfigMap", - "metadata": {"name": "modelplane-cluster-ca", "namespace": "modelplane-system"}, - }, - labeled=False, - management_policies=["Observe"], - ) - out["gateway-client-ca-bundle"] = _object( - "gateway-client-ca-bundle", - { - "apiVersion": "v1", - "kind": "ConfigMap", - "metadata": {"name": "modelplane-inference-gateway-cas", "namespace": "modelplane-system"}, - "data": {"ca.crt": _CLIENT_CA}, - }, - labeled=False, - ) - out["gateway-client-auth"] = _object( - "gateway-client-auth", - { - "apiVersion": "gateway.envoyproxy.io/v1alpha1", - "kind": "ClientTrafficPolicy", - "metadata": {"name": "cluster-gateway-client-auth", "namespace": "modelplane-system"}, - "spec": { - "targetRefs": [ - { - "group": "gateway.networking.k8s.io", - "kind": "Gateway", - "name": "cluster-gateway", - "sectionName": "https", - } - ], - "tls": { - "clientValidation": { - "caCertificateRefs": [ - {"kind": "ConfigMap", "group": "", "name": "modelplane-inference-gateway-cas"} - ] - } + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "ai-gateway-crds-helm", + "repository": "oci://docker.io/envoyproxy", + "version": "v1.1.0", + }, + "namespace": "envoy-ai-gateway-system", + "wait": True, + "waitTimeout": "10m", + }, }, - }, - }, - cel=_POLICY_ACCEPTED_CEL, - labeled=False, + } + ), + ready=ready, ) - # --- the gateway PKI trust anchor, common components on every cluster --- - out["gateway-selfsigned-issuer"] = _object( - "gateway-selfsigned-issuer", - { - "apiVersion": "cert-manager.io/v1", - "kind": "Issuer", - "metadata": {"name": "modelplane-selfsigned", "namespace": "modelplane-system"}, - "spec": {"selfSigned": {}}, - }, - ) - out["trust-manager"] = _release( - key="trust-manager", - release="mp-trust-manager", - namespace="modelplane-system", - chart="trust-manager", - repository="oci://quay.io/jetstack/charts", - version="v0.25.0", - values={ - "crds": {"enabled": True, "keep": True}, - "app": {"trust": {"namespace": "modelplane-system"}}, - "defaultPackage": {"enabled": False}, - }, + +def _gaie_crd(*, name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Object holding the Gateway API Inference Extension CRD called name, from the vendored bundle.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": f"gaie-crds-{name}"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": {"manifest": _crd(filename="gaie.yaml", name=name)}, + }, + } + ), + ready=ready, ) - return out +def _gateway_namespace(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed modelplane-system Namespace.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "gateway-namespace"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Namespace", + "metadata": { + "name": "modelplane-system", + "labels": {"modelplane.ai/namespace": "modelplane-system"}, + }, + } + }, + }, + } + ), + ready=ready, + ) -def _response(resources: dict[str, fnv1.Resource], status: dict | None = None) -> fnv1.RunFunctionResponse: - """A whole expected response: 60s TTL, empty context, the XR status. - Every response asks for the telemetry kinds, because the collector is - composed from them and the function cannot know whether any exist until - they resolve. - """ - return fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct({"status": status if status is not None else {}})), - resources=resources, - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "destinations": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" - ), - "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), +def _gateway_selfsigned_issuer() -> fnv1.Resource: + """The composed self-signed Issuer the cluster CA roots in, marked ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "gateway-selfsigned-issuer"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "Issuer", + "metadata": {"name": "modelplane-selfsigned", "namespace": "modelplane-system"}, + "spec": {"selfSigned": {}}, + } + }, + }, } ), + ready=fnv1.READY_TRUE, ) -@dataclasses.dataclass -class Case: - name: str - req: fnv1.RunFunctionRequest - want: fnv1.RunFunctionResponse +def _trust_manager(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed trust-manager Release.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-trust-manager"}, + "labels": {"modelplane.ai/resource": "trust-manager"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "trust-manager", + "repository": "oci://quay.io/jetstack/charts", + "version": "v0.25.0", + }, + "namespace": "modelplane-system", + "values": { + "crds": {"enabled": True, "keep": True}, + "app": {"trust": {"namespace": "modelplane-system"}}, + "defaultPackage": {"enabled": False}, + }, + }, + }, + } + ), + ready=ready, + ) -def _compose_cases() -> list[Case]: - """The test_compose cases, built from the Existing/Dynamo stack's resources.""" - full = _provider_configs() | _EXISTING_DYNAMO_USAGES | _existing_dynamo_stack() - - # Second pass: PCs observed. depends_on gates first creation, so - # only the dependency-free wave renders; each dependent waits for - # its dependency's Ready before it is first created. - dep_gated = { - "envoy-gateway", # -> cert-manager - "ai-gateway", # -> ai-gateway-crds - "gateway-proxy", # -> gateway-namespace - "kai-queue-root", # -> kai-scheduler - "kai-queue", # -> kai-scheduler - "modelexpress-server", # -> modelexpress-crds - "gateway-selfsigned-issuer", # -> cert-manager, gateway-namespace - "trust-manager", # -> gateway-selfsigned-issuer - } - first_wave = {k: v for k, v in full.items() if k not in dep_gated} - - # Third pass: every rendered resource observed Ready (the gateway - # with its address assigned), so everything is marked ready and - # the address lands in the XR status. - rendered = [k for k in _existing_dynamo_stack() if k != "gateway"] - observed_ready = _observed_pcs() - for key in rendered: - observed_ready[key] = fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ) - observed_ready["gateway"] = fnv1.Resource( +def _dra_driver_critical_pods_quota(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed ResourceQuota admitting the DRA driver's critical pods.""" + return fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "conditions": [{"type": "Ready", "status": "True"}], - "atProvider": { - "manifest": {"status": {"addresses": [{"type": "IPAddress", "value": "203.0.113.7"}]}}, + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "dra-driver-critical-pods-quota"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ResourceQuota", + "metadata": {"name": "allow-critical-pods", "namespace": "nvidia-dra-driver"}, + "spec": { + "hard": {"pods": "1000"}, + "scopeSelector": { + "matchExpressions": [ + { + "operator": "In", + "scopeName": "PriorityClass", + "values": ["system-node-critical", "system-cluster-critical"], + } + ] + }, + }, + } }, }, } - ) + ), + ready=ready, ) - # Every component observed Ready; PCs and Usages are ready on arrival. - all_ready = copy.deepcopy(full) - for res in all_ready.values(): - res.ready = fnv1.READY_TRUE - - return [ - Case( - name="first pass composes only the provider configs and usages", - req=_request("Existing", "Dynamo"), - # Everything targeting the remote cluster is gated on the - # ProviderConfigs having been observed; Usages reference - # nothing remote and compose immediately. The unready - # ProviderConfigs keep the composite unready until the - # stack actually renders. - want=_response(_provider_configs(ready=False) | _EXISTING_DYNAMO_USAGES), - ), - Case( - name="second pass renders the dependency-free wave", - req=_request("Existing", "Dynamo", observed=_observed_pcs()), - want=_response(first_wave), - ), - Case( - name="all dependencies ready renders the whole stack, marks it ready, and writes the gateway address", - req=_request("Existing", "Dynamo", observed=observed_ready), - want=_response(all_ready, status={"gateway": {"address": "203.0.113.7"}}), - ), - ] -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) -def test_compose(case: Case) -> None: - """RunFunction composes the Existing/Dynamo stack across the reconcile passes.""" - got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) - - -def test_identity_secret_type_flows_to_provider_configs() -> None: - """A non-GCP identity secret's type and namespace reach both ProviderConfigs verbatim.""" - # The type is stamped as is rather than being forced to - # GoogleApplicationCredentials, and the secret's own namespace wins over - # the XR's. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Nebius", - secrets=[ - v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), - v1alpha1.Secret( - type="NebiusServiceAccountCredentials", - name="nebius-secret", - key="credentials.json", - namespace="other-ns", - ), - ], - gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), - ), - ).model_dump(exclude_none=True, mode="json") - ), - ), +def _leader_worker_set() -> fnv1.Resource: + """The composed LeaderWorkerSet Release, not yet marked ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-lws"}, + "labels": {"modelplane.ai/resource": "leader-worker-set"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": {"name": "lws", "repository": "oci://registry.k8s.io/lws/charts", "version": "v0.8.0"}, + "namespace": "lws-system", + }, + }, + } ), + ready=fnv1.READY_UNSPECIFIED, ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - pc = resource.struct_to_dict(got.desired.resources["provider-config-kubernetes"].resource) - assert pc["spec"]["identity"]["type"] == "NebiusServiceAccountCredentials" - assert pc["spec"]["identity"]["secretRef"]["namespace"] == "other-ns" - helm_pc = resource.struct_to_dict(got.desired.resources["provider-config-helm"].resource) - assert helm_pc["spec"]["identity"]["type"] == "NebiusServiceAccountCredentials" - - -def test_gpu_pool_nvlink_disable_flows_to_components() -> None: - """A Civo ServingStack whose spec.gpu flags a pool for NVLink - disable composes the gpu-operator release in NVIDIADriver-CRD - mode, the kernel module ConfigMap, and a per-pool NVIDIADriver - selecting that pool's nodes - and only that pool's.""" - # The install gate composes a component once its dependencies are - # observed Ready; observe the chain up to the per-pool driver. - observed = _observed_pcs() - for key in ( - "cert-manager", - "node-feature-discovery", - "gpu-operator", - "nvlink-disable-config-gpu-operator", - "nvlink-disable-config-nvidia-kernel-config", - ): - observed[key] = fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ) - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Civo", - secrets=[ - v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), - ], - gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), - gpu=v1alpha1.Gpu( - pools=[v1alpha1.Pool(name="h100-pool", disableNvLink=True)], - ), - ), - ).model_dump(exclude_none=True, mode="json") - ), - ), - resources=observed, + + +def _gateway_class(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed envoy GatewayClass, parameterised by the gateway's EnvoyProxy.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "gateway-class"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "GatewayClass", + "metadata": {"name": "envoy"}, + "spec": { + "controllerName": "gateway.envoyproxy.io/gatewayclass-controller", + "parametersRef": { + "group": "gateway.envoyproxy.io", + "kind": "EnvoyProxy", + "name": "cluster-gateway", + "namespace": "modelplane-system", + }, + }, + } + }, + }, + } ), + ready=ready, ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - release = resource.struct_to_dict(got.desired.resources["gpu-operator"].resource) - assert release["spec"]["forProvider"]["values"]["driver"]["nvidiaDriverCRD"] == { - "enabled": True, - "deployDefaultCR": True, - } - config = resource.struct_to_dict(got.desired.resources["nvlink-disable-config-nvidia-kernel-config"].resource) - assert config["spec"]["forProvider"]["manifest"]["data"] == {"nvidia.conf": "options nvidia NVreg_NvLinkDisable=1"} +def _gateway(*, hostname: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Gateway: one HTTPS listener for hostname, terminating TLS with the serving certificate.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "gateway"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "Gateway", + "metadata": {"name": "cluster-gateway", "namespace": "modelplane-system"}, + "spec": { + "gatewayClassName": "envoy", + "listeners": [ + { + "name": "https", + "protocol": "HTTPS", + "port": 443, + "hostname": hostname, + "tls": { + "mode": "Terminate", + "certificateRefs": [{"name": "cluster-gateway-serving"}], + }, + "allowedRoutes": { + "namespaces": { + "from": "Selector", + "selector": { + "matchExpressions": [ + {"key": "modelplane.ai/namespace", "operator": "Exists"} + ] + }, + } + }, + } + ], + }, + } + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status.addresses) && object.status.addresses.size() > 0", + }, + }, + } + ), + ready=ready, + ) - driver = resource.struct_to_dict(got.desired.resources["nvlink-disabled-driver-h100-pool"].resource) - manifest = driver["spec"]["forProvider"]["manifest"] - assert manifest["kind"] == "NVIDIADriver" - assert manifest["spec"]["nodeSelector"] == {"modelplane.ai/pool": "h100-pool"} - assert manifest["spec"]["kernelModuleConfig"] == {"name": "nvidia-kernel-config"} - # The derived Usages hold the operator release and the ConfigMap - # until the per-pool driver is gone. - assert "usage-gpu-operator-by-nvlink-disabled-driver-h100-pool" in got.desired.resources +def _gateway_ca_certificate(*, common_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed cluster CA Certificate, issued by the self-signed Issuer.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "Certificate", + "metadata": {"name": "modelplane-cluster-ca", "namespace": "modelplane-system"}, + "spec": { + "isCA": True, + "commonName": common_name, + "secretName": "modelplane-cluster-ca", + "duration": "87600h", + "renewBefore": "8760h", + "privateKey": {"algorithm": "ECDSA", "size": 256}, + "issuerRef": { + "name": "modelplane-selfsigned", + "kind": "Issuer", + "group": "cert-manager.io", + }, + }, + } + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status) && has(object.status.conditions) && object.status.conditions.exists(c, c.type == 'Ready' && c.status == 'True')", + }, + }, + } + ), + ready=ready, + ) -def test_gpu_pool_without_nvlink_disable_changes_nothing() -> None: - """A Civo ServingStack whose spec.gpu flags no pool composes the - stock component list: ClusterPolicy-managed driver, no NVIDIADriver - or kernel module ConfigMap objects.""" - observed = _observed_pcs() - observed["gpu-operator"] = fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ) - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Civo", - secrets=[ - v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), - ], - gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), - gpu=v1alpha1.Gpu( - pools=[v1alpha1.Pool(name="l40s-pool", disableNvLink=False)], - ), - ), - ).model_dump(exclude_none=True, mode="json") - ), - ), - resources=observed, +def _gateway_ca_issuer(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Issuer that signs with the cluster CA.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "Issuer", + "metadata": {"name": "modelplane-cluster-ca", "namespace": "modelplane-system"}, + "spec": {"ca": {"secretName": "modelplane-cluster-ca"}}, + } + }, + }, + } ), + ready=ready, ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - release = resource.struct_to_dict(got.desired.resources["gpu-operator"].resource) - assert "nvidiaDriverCRD" not in release["spec"]["forProvider"]["values"]["driver"] - for key in got.desired.resources: - assert "nvlink" not in key - - -def test_cluster_gateway_composes_mtls_with_ca() -> None: - """A cluster with an InferenceGateway CA composes its own PKI and serves mTLS.""" - # It issues its own PKI, republishes the CA without its key, demands a - # client certificate on its HTTPS listener, and publishes the CA in status. - # - # The hostname is a full Service FQDN, so the CA certificate's commonName - # overflows the 64-byte X.509 limit and is truncated. - hostname = "gateway-test-backend-12345.modelplane-system.svc.cluster.local" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Existing", - stack="Standard", - secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], - gateway=v1alpha1.Gateway( - hostname=hostname, - # Deliberately out of name order, to prove the - # bundle sorts before concatenating. - clientCAs=[ - v1alpha1.ClientCA(name="fleet-b", certificate="BBB"), - v1alpha1.ClientCA(name="fleet-a", certificate="AAA"), - ], - ), - ), - ).model_dump(exclude_none=True, mode="json") - ), - ), - # PCs observed, the self-signed Issuer Ready (so trust-manager and - # the CA chain proceed), and the CA ConfigMap trust-manager syncs - # carrying the certificate back for status. - resources=_observed_pcs() - | { - "gateway-selfsigned-issuer": fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ), - "gateway-ca-configmap": fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}}} - ) - ), - }, + + +def _gateway_serving_certificate(*, hostname: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Certificate the gateway serves for hostname, issued by the cluster CA.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "Certificate", + "metadata": {"name": "cluster-gateway-serving", "namespace": "modelplane-system"}, + "spec": { + "secretName": "cluster-gateway-serving", + "dnsNames": [hostname], + "duration": "2160h", + "renewBefore": "720h", + "privateKey": {"algorithm": "ECDSA", "size": 256, "rotationPolicy": "Always"}, + "issuerRef": { + "name": "modelplane-cluster-ca", + "kind": "Issuer", + "group": "cert-manager.io", + }, + }, + } + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status) && has(object.status.conditions) && object.status.conditions.exists(c, c.type == 'Ready' && c.status == 'True')", + }, + }, + } ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - def manifest(key: str) -> dict: - return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] - - ca_cert = manifest("gateway-ca-certificate") - assert ca_cert["spec"]["commonName"] == "modelplane cluster CA gateway-test-backend-12345.modelplane-syst" - assert len(ca_cert["spec"]["commonName"]) <= 64 - assert ca_cert["spec"]["isCA"] - assert ca_cert["spec"]["issuerRef"]["name"] == "modelplane-selfsigned" - - serving = manifest("gateway-serving-certificate") - assert serving["spec"]["dnsNames"] == [hostname] - assert serving["spec"]["issuerRef"]["name"] == "modelplane-cluster-ca" - - bundle = manifest("gateway-ca-bundle") - assert bundle["apiVersion"] == "trust.cert-manager.io/v1alpha1" - assert bundle["spec"]["sources"] == [{"secret": {"name": "modelplane-cluster-ca", "key": "ca.crt"}}] - - # Observed only, never managed: trust-manager owns the ConfigMap. - ca_cm = got.desired.resources["gateway-ca-configmap"] - assert resource.struct_to_dict(ca_cm.resource)["spec"]["managementPolicies"] == ["Observe"] - - # Every InferenceGateway's CA, sorted by name and concatenated. - client_bundle = manifest("gateway-client-ca-bundle") - assert client_bundle["data"]["ca.crt"] == "AAA\nBBB\n" - - client_auth = manifest("gateway-client-auth") - assert client_auth["kind"] == "ClientTrafficPolicy" - assert client_auth["spec"]["targetRefs"][0]["sectionName"] == "https" - assert ( - client_auth["spec"]["tls"]["clientValidation"]["caCertificateRefs"][0]["name"] - == "modelplane-inference-gateway-cas" + ready=ready, ) - # One HTTPS listener, terminating TLS with the serving certificate. - gateway = manifest("gateway") - assert gateway["spec"]["listeners"] == [ - { - "name": "https", - "protocol": "HTTPS", - "port": 443, - "hostname": hostname, - "tls": {"mode": "Terminate", "certificateRefs": [{"name": "cluster-gateway-serving"}]}, - "allowedRoutes": { - "namespaces": { - "from": "Selector", - "selector": {"matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}]}, - } - }, - } - ] - - status = resource.struct_to_dict(got.desired.composite.resource)["status"] - assert status["gateway"]["caCertificate"] == "CLUSTERCA" - - # Every PKI resource must be tracked for readiness: - # compose_gateway_pki marks only the keys it returns, so one composed - # but not returned would silently hold the cluster un-Ready. Observe - # each Ready and assert it's marked ready, which fails if the key was - # dropped from the rendered list. (The self-signed Issuer and - # trust-manager are stack components, covered by the golden test.) - pki_keys = [ - "gateway-ca-certificate", - "gateway-ca-issuer", - "gateway-serving-certificate", - "gateway-ca-bundle", - "gateway-ca-configmap", - "gateway-client-ca-bundle", - "gateway-client-auth", - ] - for key in pki_keys: - req.observed.resources[key].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ) - ) - # Preserve the CA ConfigMap's data alongside its Ready condition. - req.observed.resources["gateway-ca-configmap"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "conditions": [{"type": "Ready", "status": "True"}], - "atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}, - } - } - ) - ) + +def _gateway_ca_bundle(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed trust-manager Bundle republishing the cluster CA's certificate without its key.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "trust.cert-manager.io/v1alpha1", + "kind": "Bundle", + "metadata": {"name": "modelplane-cluster-ca"}, + "spec": { + "sources": [{"secret": {"name": "modelplane-cluster-ca", "key": "ca.crt"}}], + "target": { + "configMap": {"key": "ca.crt"}, + "namespaceSelector": { + "matchLabels": {"kubernetes.io/metadata.name": "modelplane-system"} + }, + }, + }, + } + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status) && has(object.status.conditions) && object.status.conditions.exists(c, c.type == 'Synced' && c.status == 'True')", + }, + }, + } + ), + ready=ready, ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - for key in pki_keys: - assert got.desired.resources[key].ready == fnv1.READY_TRUE, f"{key} not marked ready" - - -def test_cluster_gateway_without_ca_serves_nothing() -> None: - """A cluster with no InferenceGateway CA withholds its Gateway, and warns.""" - # Withholding the Gateway entirely, rather than serving the engines - # unauthenticated. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Existing", - stack="Standard", - secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], - gateway=v1alpha1.Gateway(hostname="gw.clusters.example.com"), - ), - ).model_dump(exclude_none=True, mode="json") - ), - ), - resources=_observed_pcs(), + + +def _gateway_ca_configmap(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Object observing, never managing, the CA ConfigMap trust-manager owns.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "modelplane-cluster-ca", "namespace": "modelplane-system"}, + } + }, + "managementPolicies": ["Observe"], + }, + } ), + ready=ready, ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - # The GatewayClass and the cluster's own PKI are composed, so the CA is - # ready to publish when the first InferenceGateway's CA arrives. The - # Gateway, the client CA bundle and the policy demanding a client - # certificate aren't, and nor is the Usage protecting the Gateway. - gateway_keys = {k for k in got.desired.resources if k.startswith(("gateway", "usage-gateway"))} - assert gateway_keys == { - "gateway-class", - "gateway-ca-certificate", - "gateway-ca-issuer", - "gateway-serving-certificate", - "gateway-ca-bundle", - "gateway-ca-configmap", - "gateway-namespace", - "usage-gateway-namespace-by-gateway-proxy", - "usage-gateway-namespace-by-gateway-selfsigned-issuer", - "usage-gateway-selfsigned-issuer-by-trust-manager", - } - assert list(got.results) == [ - fnv1.Result( - severity=fnv1.SEVERITY_WARNING, - message=( - "Gateway gw.clusters.example.com not served: no InferenceGateway has published a client " - "CA for this cluster to trust, and serving without one would accept unauthenticated callers" - ), - ) - ] -# The composed-resource key a component renders under is its identity: -# renaming one deletes and recreates the remote resource (for an Object -# holding a CRD, the CRD and its CRs). This pins the full key set per -# cloud and stack, including the Usage keys derived from depends_on, as -# reviewed literals. A failure here means the stack data changed a key - -# make sure that's intended, then update the inventory and the release -# notes. +def _gateway_client_ca_bundle(*, ca_crt: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed ConfigMap holding ca_crt, the InferenceGateway CAs the gateway trusts.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": { + "name": "modelplane-inference-gateway-cas", + "namespace": "modelplane-system", + }, + "data": {"ca.crt": ca_crt}, + } + }, + }, + } + ), + ready=ready, + ) -# Every cloud and stack, for a stack with an InferenceGateway CA to trust (as -# _request builds), so the Gateway and its client-auth policy are included. -_ALWAYS = frozenset( - { - "provider-config-kubernetes", - "provider-config-helm", - "gateway", - "gateway-class", - "gateway-ca-certificate", - "gateway-ca-issuer", - "gateway-serving-certificate", - "gateway-ca-bundle", - "gateway-ca-configmap", - "gateway-client-ca-bundle", - "gateway-client-auth", - "usage-gateway-class-by-gateway", - "usage-envoy-gateway-by-gateway-class", - } -) - -_COMMON = frozenset( - { - "ai-gateway", - "ai-gateway-crds", - "dra-driver-critical-pods-quota", - "envoy-gateway", - "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", - "gaie-crds-inferencepools.inference.networking.k8s.io", - "gaie-crds-inferencepools.inference.networking.x-k8s.io", - "gateway-namespace", - "gateway-proxy", - "gateway-selfsigned-issuer", - "trust-manager", - "usage-ai-gateway-crds-by-ai-gateway", - "usage-cert-manager-by-envoy-gateway", - "usage-cert-manager-by-gateway-selfsigned-issuer", - "usage-gateway-namespace-by-gateway-proxy", - "usage-gateway-namespace-by-gateway-selfsigned-issuer", - "usage-gateway-selfsigned-issuer-by-trust-manager", - } -) -_STANDARD = frozenset( - { - "leader-worker-set", - } -) - -_DYNAMO = frozenset( - { - "grove", - "kai-queue", - "kai-queue-root", - "kai-scheduler", - "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", - "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", - "modelexpress-server", - "modelexpress-server-role", - "modelexpress-server-rolebinding", - "modelexpress-server-sa", - "modelexpress-server-svc", - "usage-kai-scheduler-by-kai-queue", - "usage-kai-scheduler-by-kai-queue-root", - "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", - "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", - } -) - -_EKS = frozenset( - { - "cert-manager", - "gpu-operator", - "k8s-ephemeral-storage-metrics", - "kube-prometheus-stack", - "node-feature-discovery", - "nodewright-operator", - "nvidia-dra-driver-gpu", - "nvsentinel", - "prometheus-adapter", - "prometheus-operator-crds", - "usage-cert-manager-by-gpu-operator", - "usage-cert-manager-by-nvsentinel", - "usage-gpu-operator-by-nvidia-dra-driver-gpu", - "usage-gpu-operator-by-nvsentinel", - "usage-kube-prometheus-stack-by-gpu-operator", - "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics", - "usage-kube-prometheus-stack-by-prometheus-adapter", - "usage-node-feature-discovery-by-gpu-operator", - "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics", - "usage-prometheus-operator-crds-by-kube-prometheus-stack", - "usage-prometheus-operator-crds-by-nvsentinel", - } -) +def _gateway_client_auth(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed ClientTrafficPolicy demanding a client certificate on the HTTPS listener.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "ClientTrafficPolicy", + "metadata": { + "name": "cluster-gateway-client-auth", + "namespace": "modelplane-system", + }, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "cluster-gateway", + "sectionName": "https", + } + ], + "tls": { + "clientValidation": { + "caCertificateRefs": [ + { + "kind": "ConfigMap", + "group": "", + "name": "modelplane-inference-gateway-cas", + } + ] + } + }, + }, + } + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status) && has(object.status.ancestors) && object.status.ancestors.exists(a, has(a.conditions) && a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))", + }, + }, + } + ), + ready=ready, + ) -# AKS additionally carries the gpu-operator's toolkit-hardening manifest. -_AKS = _EKS | frozenset( - { - "gpu-operator-manifests", - "usage-gpu-operator-by-gpu-operator-manifests", - } -) - -# GKE additionally carries the critical-pods ResourceQuota aicr's -# bundler synthesizes as a gpu-operator pre-manifest (GKE rejects -# system-node-critical pods in a namespace without one; aicr#915). -_GKE = _EKS | frozenset( - { - "gpu-operator-pre-manifests-gpu-operator", - "gpu-operator-pre-manifests-aicr-gke-critical-pods", - "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator", - "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator", - } -) - -_HAND_WRITTEN = frozenset( - { - "cert-manager", - "kube-prometheus-stack", - "node-feature-discovery", - "nvidia-dra-driver-gpu", - } -) - -# VKE pre-installs NFD via its managed GPU Operator add-on, so the -# Vultr half carries no node-feature-discovery of its own. -_VULTR = _HAND_WRITTEN - frozenset({"node-feature-discovery"}) - -# Civo pre-installs no GPU stack at all, so its hand-written half is the -# one that carries the GPU Operator (driver included), plus the Usages -# for its depends_on edges and the DRA driver's edge onto it. -_CIVO = _HAND_WRITTEN | frozenset( - { - "gpu-operator", - "usage-cert-manager-by-gpu-operator", - "usage-node-feature-discovery-by-gpu-operator", - "usage-gpu-operator-by-nvidia-dra-driver-gpu", - } -) - -_INVENTORY = { - "EKS": _EKS, - "AKS": _AKS, - "GKE": _GKE, - "Nebius": _HAND_WRITTEN, - "Vultr": _VULTR, - "Civo": _CIVO, - "Existing": _HAND_WRITTEN, -} - - -@pytest.mark.parametrize( - ("stack", "stack_keys"), [("Standard", _STANDARD), ("Dynamo", _DYNAMO)], ids=["Standard", "Dynamo"] -) -@pytest.mark.parametrize(("cloud", "cloud_keys"), list(_INVENTORY.items()), ids=list(_INVENTORY)) -def test_composed_resource_keys(cloud: str, cloud_keys: frozenset[str], stack: str, stack_keys: frozenset[str]) -> None: - """Every cloud and stack composes exactly its inventoried resource keys.""" - expected = _ALWAYS | _COMMON | cloud_keys | stack_keys - # Observe every expected key Ready so the depends_on - # install gate opens and the full stack renders; a - # key the function doesn't render still fails the - # comparison. - observed = _observed_pcs() - for key in expected: - observed[key] = fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(_request(cloud, stack, observed=observed), None)) - assert set(got.desired.resources.keys()) == expected - - -def _with_destination(req: fnv1.RunFunctionRequest, *, secret: str | None = None) -> fnv1.RunFunctionRequest: - sink: dict = {"name": "primary", "type": "otlphttp", "endpoint": "https://otel.acme.example"} - if secret: - sink |= {"secretRef": {"name": secret}, "auth": {"bearerTokenKey": "token"}} - req.required_resources["destinations"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "TelemetryDestination", - "metadata": {"name": "acme"}, - "spec": {"sinks": [sink]}, - } - ) - ) + +def _usage( + *, of_api_version: str, of_kind: str, of_key: str, by_api_version: str, by_kind: str, by_key: str +) -> fnv1.Resource: + """The composed Usage holding the resource labelled of_key until the one labelled by_key is gone, marked ready on arrival.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "spec": { + "of": { + "apiVersion": of_api_version, + "kind": of_kind, + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": of_key}, + }, + }, + "by": { + "apiVersion": by_api_version, + "kind": by_kind, + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": by_key}, + }, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, ) - req.required_resources["mappings"].items.extend([]) - if secret: - req.required_resources["collector-secret-primary"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": secret, "namespace": "modelplane-system"}, - "data": {"token": "c2hoaGg="}, - } - ) - ) - ) - return req -def test_the_collector_does_not_gate_the_stack() -> None: - """A collector nothing has observed yet is still Ready. +def _gke_cert_manager() -> fnv1.Resource: + """The composed cert-manager Release GKE's generated half pins, not yet marked ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-cert-manager"}, + "labels": {"modelplane.ai/resource": "cert-manager"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "cert-manager", + "repository": "https://charts.jetstack.io", + "version": "v1.20.2", + }, + "namespace": "cert-manager", + "wait": True, + "waitTimeout": "10m", + "values": { + "cainjector": { + "resources": { + "limits": {"cpu": "50m", "memory": "320Mi"}, + "requests": {"cpu": "50m", "memory": "320Mi"}, + }, + "tolerations": [], + }, + "crds": {"enabled": True}, + "fullnameOverride": "cert-manager", + "prometheus": {"servicemonitor": {"enabled": False}}, + "resources": { + "limits": {"cpu": "50m", "memory": "90Mi"}, + "requests": {"cpu": "50m", "memory": "90Mi"}, + }, + "startupapicheck": {"enabled": True, "tolerations": []}, + "tolerations": [], + "webhook": { + "resources": { + "limits": {"cpu": "50m", "memory": "40Mi"}, + "requests": {"cpu": "50m", "memory": "40Mi"}, + }, + "tolerations": [], + }, + # Forced by fn._helm_release, as on every cloud. + "clusterResourceNamespace": "modelplane-system", + "enableCertificateOwnerRef": True, + }, + }, + }, + } + ), + ready=fnv1.READY_UNSPECIFIED, + ) - Everything else the stack composes is Ready only once its observed - Ready condition says so, because the fleet cannot serve without it. - The collector only watches, so gating on it would put placing a - replica behind exporting a metric: one destination pointing at an - endpoint that has gone away would take every InferenceCluster in the - fleet out of Ready and stop the scheduler. - """ - req = _with_destination(_request("GKE", "Standard", observed=_observed_pcs())) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - collector_keys = [k for k in got.desired.resources if k == "collector" or k.startswith("collector-")] - # The Deployment, which is the one with a readiness CEL of its own and - # so the one that would have gated the stack. - assert "collector" in collector_keys - for key in collector_keys: - assert key not in req.observed.resources, key - assert got.desired.resources[key].ready == fnv1.READY_TRUE, key - - -def test_the_credential_reaches_the_cluster_that_mounts_it() -> None: - """The operator writes one Secret; the collector mounts it elsewhere. - - A TelemetryDestination is cluster-scoped on the control plane and the - collector runs on every workload cluster in the fleet. Resolving the - Secret and stopping there leaves the Deployment mounting a name - nothing out there creates, so the pod never starts and the fleet - exports nothing - the failure every destination with a credential - would hit, which is every destination that reaches a real backend. - """ - req = _with_destination(_request("GKE", "Standard", observed=_observed_pcs()), secret="telemetry-credentials") - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert "collector-secret-primary" in got.desired.resources - composed = resource.struct_to_dict(got.desired.resources["collector-secret-primary"].resource) - manifest = composed["spec"]["forProvider"]["manifest"] - assert manifest["kind"] == "Secret" - assert manifest["metadata"]["name"] == "telemetry-credentials" - assert manifest["metadata"]["namespace"] == "modelplane-system" - # Copied verbatim: re-encoding would corrupt a credential that is not - # text, and the mount reads the same key the sink's auth names. - assert manifest["data"] == {"token": "c2hoaGg="} - - -def test_a_destination_asks_for_its_credential_in_one_namespace() -> None: - """Unqualified, the requirement matches a Secret of that name anywhere.""" - req = _with_destination(_request("GKE", "Standard", observed=_observed_pcs()), secret="telemetry-credentials") - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - selector = got.requirements.resources["collector-secret-primary"] - assert selector.match_name == "telemetry-credentials" - assert selector.namespace == "modelplane-system" - - -def test_every_destination_contributes_its_sinks() -> None: - """A second backend is a second object, not an edit to a singleton. - - Picking one destination and warning about the rest means a team - adding an export has to edit an object another team owns, and gets - silence if they create their own instead. - """ - req = _request("GKE", "Standard", observed=_observed_pcs()) - for name, sink in ( - ("acme", {"name": "vendor", "type": "otlphttp", "endpoint": "https://otel.vendor.example"}), - ("zeta", {"name": "prom", "type": "prometheus_remote_write", "endpoint": "https://p.example/w"}), - ): - req.required_resources["destinations"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "TelemetryDestination", - "metadata": {"name": name}, - "spec": {"sinks": [sink]}, - } - ) - ) - ) - req.required_resources["mappings"].items.extend([]) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - config = yaml.safe_load( - resource.struct_to_dict(got.desired.resources["collector-config"].resource)["spec"]["forProvider"]["manifest"][ - "data" - ]["collector.yaml"] + +def _gke_node_feature_discovery() -> fnv1.Resource: + """The composed node-feature-discovery Release GKE's generated half pins, not yet marked ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-node-feature-discovery"}, + "labels": {"modelplane.ai/resource": "node-feature-discovery"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "node-feature-discovery", + "repository": "https://kubernetes-sigs.github.io/node-feature-discovery/charts", + "version": "0.19.0", + }, + "namespace": "node-feature-discovery", + "wait": True, + "waitTimeout": "10m", + "values": { + "gc": {"enable": True, "tolerations": []}, + "master": {"enable": True, "tolerations": []}, + "topologyUpdater": { + "createCRDs": True, + "enable": False, + "kubeletStateDir": "", + "resources": { + "limits": {"memory": "256Mi"}, + "requests": {"cpu": "50m", "memory": "128Mi"}, + }, + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ], + }, + "worker": { + "enable": True, + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ], + }, + }, + }, + }, + } + ), + ready=fnv1.READY_UNSPECIFIED, ) - assert sorted(config["exporters"]) == ["otlphttp/vendor", "prometheus_remote_write/prom"] - assert sorted(config["service"]["pipelines"]["metrics"]["exporters"]) == [ - "otlphttp/vendor", - "prometheus_remote_write/prom", - ] -def test_two_destinations_cannot_name_one_exporter() -> None: - """A sink names the collector's exporter instance. +def _nodewright_operator() -> fnv1.Resource: + """The composed nodewright operator Release, not yet marked ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-nodewright"}, + "labels": {"modelplane.ai/resource": "nodewright-operator"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "nodewright", + "repository": "oci://ghcr.io/nvidia/nodewright/charts", + "version": "v0.17.1", + }, + "namespace": "skyhook", + "values": { + "controllerManager": { + "manager": { + "env": {"copyDirRoot": "/etc/nodewright", "reapplyOnReboot": "true"}, + "resources": { + "limits": {"cpu": "1000m", "memory": "4000Mi"}, + "requests": {"cpu": "1000m", "memory": "2000Mi"}, + }, + }, + "tolerations": [], + }, + "fullnameOverride": "skyhook-operator", + "limitRange": { + "default": {"cpu": "1", "memory": "1Gi"}, + "defaultRequest": {"cpu": "500m", "memory": "512Mi"}, + }, + }, + }, + }, + } + ), + ready=fnv1.READY_UNSPECIFIED, + ) + - Two of them under one name is one exporter with two meanings. The - destination sorting first keeps it and the other is dropped with a - warning, rather than failing the whole fleet's telemetry over a name. - """ - req = _request("GKE", "Standard", observed=_observed_pcs()) - for name, endpoint in (("acme", "https://a.example"), ("zeta", "https://z.example")): - req.required_resources["destinations"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "TelemetryDestination", - "metadata": {"name": name}, - "spec": {"sinks": [{"name": "primary", "type": "otlphttp", "endpoint": endpoint}]}, - } - ) - ) - ) - req.required_resources["mappings"].items.extend([]) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - config = yaml.safe_load( - resource.struct_to_dict(got.desired.resources["collector-config"].resource)["spec"]["forProvider"]["manifest"][ - "data" - ]["collector.yaml"] +def _prometheus_operator_crds() -> fnv1.Resource: + """The composed Prometheus operator CRDs Release, not yet marked ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-prometheus-operator-crds"}, + "labels": {"modelplane.ai/resource": "prometheus-operator-crds"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "prometheus-operator-crds", + "repository": "https://prometheus-community.github.io/helm-charts", + "version": "28.0.1", + }, + "namespace": "monitoring", + "wait": True, + "waitTimeout": "10m", + "values": {"enabled": True}, + }, + }, + } + ), + ready=fnv1.READY_UNSPECIFIED, ) - assert list(config["exporters"]) == ["otlphttp/primary"] - assert config["exporters"]["otlphttp/primary"]["endpoint"] == "https://a.example" - assert [r for r in got.results if "zeta" in r.message] -def test_a_stale_mapping_does_not_break_the_stack() -> None: - """A CRD validates on write, not on what it already stored. +def _gpu_operator_pre_manifests_gpu_operator() -> fnv1.Resource: + """The composed gpu-operator Namespace the GPU operator's pre-manifests create, not yet marked ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "gpu-operator-pre-manifests-gpu-operator"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": {"apiVersion": "v1", "kind": "Namespace", "metadata": {"name": "gpu-operator"}} + }, + }, + } + ), + ready=fnv1.READY_UNSPECIFIED, + ) + - A MetricMapping written against an older schema comes back on read - exactly as it was stored, so a value the enum no longer carries - reaches the parser. Parsing it raises, and raising fails the whole - pipeline step - so the serving stack composes nothing and the fleet - stops placing replicas, because one telemetry object is out of date. - Seen on a real cluster, where a mapping predating a required field - did it. - """ - req = _with_destination(_request("GKE", "Standard", observed=_observed_pcs())) - req.required_resources["mappings"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "MetricMapping", - "metadata": {"name": "stale"}, - # A unit the enum no longer carries. - "spec": { - "metrics": [ - { - "from": "old_engine_transfer", - "to": "modelplane_request_kv_transfer_seconds", - "fromUnit": "Centiseconds", - } - ] +def _gpu_operator_pre_manifests_aicr_gke_critical_pods() -> fnv1.Resource: + """The composed ResourceQuota admitting the GPU operator's critical pods on GKE, not yet marked ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "gpu-operator-pre-manifests-aicr-gke-critical-pods"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ResourceQuota", + "metadata": {"name": "aicr-gke-critical-pods", "namespace": "gpu-operator"}, + "spec": { + "hard": {"pods": "32"}, + "scopeSelector": { + "matchExpressions": [ + { + "operator": "In", + "scopeName": "PriorityClass", + "values": ["system-node-critical", "system-cluster-critical"], + } + ] + }, + }, + } }, - } - ) - ) + }, + } + ), + ready=fnv1.READY_UNSPECIFIED, ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - # The stack still composes, and says what it dropped. - assert "collector" in got.desired.resources - assert [r for r in got.results if "stale" in r.message] - config = yaml.safe_load( - resource.struct_to_dict(got.desired.resources["collector-config"].resource)["spec"]["forProvider"]["manifest"][ - "data" - ]["collector.yaml"] + + +def _collector_service_account() -> fnv1.Resource: + """The composed collector ServiceAccount, marked ready on arrival.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "collector-serviceaccount"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ServiceAccount", + "metadata": { + "name": "modelplane-collector", + "namespace": "modelplane-system", + "labels": { + "app.kubernetes.io/name": "modelplane-collector", + "app.kubernetes.io/managed-by": "modelplane", + }, + }, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, ) - assert "old_engine_waiting" not in yaml.safe_dump(config) + + +def _collector_cluster_role() -> fnv1.Resource: + """The composed collector ClusterRole, marked ready on arrival.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "collector-clusterrole"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "ClusterRole", + "metadata": { + "name": "modelplane-collector", + "labels": { + "app.kubernetes.io/name": "modelplane-collector", + "app.kubernetes.io/managed-by": "modelplane", + }, + }, + "rules": [ + { + "apiGroups": [""], + "resources": ["pods", "services", "endpoints", "nodes", "nodes/metrics"], + "verbs": ["get", "list", "watch"], + }, + {"nonResourceURLs": ["/metrics"], "verbs": ["get"]}, + ], + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _collector_cluster_role_binding() -> fnv1.Resource: + """The composed collector ClusterRoleBinding, marked ready on arrival.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "collector-clusterrolebinding"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "ClusterRoleBinding", + "metadata": { + "name": "modelplane-collector", + "labels": { + "app.kubernetes.io/name": "modelplane-collector", + "app.kubernetes.io/managed-by": "modelplane", + }, + }, + "roleRef": { + "apiGroup": "rbac.authorization.k8s.io", + "kind": "ClusterRole", + "name": "modelplane-collector", + }, + "subjects": [ + { + "kind": "ServiceAccount", + "name": "modelplane-collector", + "namespace": "modelplane-system", + } + ], + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _collector_config( + *, exporters: dict, pipeline_exporters: list[str], extensions: dict | None, service_extensions: list[str] | None +) -> fnv1.Resource: + """The composed ConfigMap holding the collector's config, marked ready on arrival. + + The ConfigMap holds the config as YAML text. This writes it as the dict the + function dumps, in the order the function builds it, so it reads as config + rather than as wrapped YAML, and PyYAML dumps it on both sides. + test_collector.py says what each part of it is for. + """ + service: dict = { + "pipelines": { + "metrics": { + "receivers": ["prometheus"], + "processors": [ + "memory_limiter", + "resource/cluster", + "transform/identity", + "transform/modelplane", + "groupbyattrs/identity", + "filter/modelplane", + "batch", + ], + "exporters": pipeline_exporters, + } + }, + } + if service_extensions is not None: + service["extensions"] = service_extensions + config: dict = { + "receivers": { + "prometheus": { + "config": { + "scrape_configs": [ + { + "job_name": "modelplane-engines", + "scrape_interval": "15s", + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + { + "source_labels": ["__meta_kubernetes_pod_label_modelplane_ai_deployment"], + "action": "keep", + "regex": ".+", + }, + { + "source_labels": ["__meta_kubernetes_pod_container_port_name"], + "action": "keep", + "regex": "http", + }, + { + "source_labels": ["__meta_kubernetes_pod_label_modelplane_ai_deployment"], + "target_label": "deployment", + }, + { + "source_labels": ["__meta_kubernetes_pod_label_modelplane_ai_replica"], + "target_label": "replica", + }, + { + "source_labels": ["__meta_kubernetes_pod_label_modelplane_ai_engine"], + "target_label": "engine", + }, + { + "source_labels": ["__meta_kubernetes_pod_label_modelplane_ai_role"], + "target_label": "role", + }, + {"source_labels": ["__meta_kubernetes_namespace"], "target_label": "namespace"}, + ], + }, + { + "job_name": "modelplane-gateway", + "scrape_interval": "15s", + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + { + "source_labels": [ + "__meta_kubernetes_pod_label_gateway_envoyproxy_io_owning_gateway_name" + ], + "action": "keep", + "regex": ".+", + }, + { + "source_labels": ["__meta_kubernetes_pod_container_port_name"], + "action": "keep", + "regex": "metrics", + }, + { + "source_labels": ["__meta_kubernetes_pod_annotation_prometheus_io_path"], + "action": "replace", + "target_label": "__metrics_path__", + "regex": "(.+)", + }, + ], + }, + { + "job_name": "modelplane-gateway-genai", + "scrape_interval": "15s", + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + { + "source_labels": [ + "__meta_kubernetes_pod_label_gateway_envoyproxy_io_owning_gateway_name" + ], + "action": "keep", + "regex": ".+", + }, + { + "source_labels": ["__meta_kubernetes_pod_container_port_name"], + "action": "keep", + "regex": "aigw-admin", + }, + ], + }, + { + "job_name": "modelplane-gpu", + "scrape_interval": "15s", + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + { + "source_labels": [ + "__meta_kubernetes_pod_label_app_kubernetes_io_name", + "__meta_kubernetes_pod_label_app", + ], + "action": "keep", + "regex": ".*dcgm.*", + }, + { + "source_labels": ["__meta_kubernetes_pod_container_port_name"], + "action": "keep", + "regex": "metrics", + }, + {"source_labels": ["__meta_kubernetes_pod_node_name"], "target_label": "node"}, + ], + }, + { + "job_name": "modelplane-substrate", + "scrape_interval": "30s", + "kubernetes_sd_configs": [{"role": "pod"}], + "relabel_configs": [ + { + "source_labels": ["__meta_kubernetes_pod_annotation_prometheus_io_scrape"], + "action": "keep", + "regex": "true", + }, + { + "source_labels": [ + "__meta_kubernetes_pod_label_gateway_envoyproxy_io_owning_gateway_name" + ], + "action": "drop", + "regex": ".+", + }, + { + "source_labels": ["__meta_kubernetes_pod_label_modelplane_ai_deployment"], + "action": "drop", + "regex": ".+", + }, + { + "source_labels": [ + "__meta_kubernetes_pod_label_app_kubernetes_io_name", + "__meta_kubernetes_pod_label_app", + ], + "action": "drop", + "regex": ".*dcgm.*", + }, + { + "source_labels": ["__meta_kubernetes_pod_annotation_prometheus_io_path"], + "action": "replace", + "target_label": "__metrics_path__", + "regex": "(.+)", + }, + { + "source_labels": [ + "__address__", + "__meta_kubernetes_pod_annotation_prometheus_io_port", + ], + "action": "replace", + "target_label": "__address__", + "regex": r"(\[.+\]|[^:]+)(?::\d+)?;(\d+)", + "replacement": "$1:$2", + }, + {"source_labels": ["__meta_kubernetes_namespace"], "target_label": "namespace"}, + ], + }, + ] + } + } + }, + "processors": { + "memory_limiter": {"check_interval": "1s", "limit_percentage": 80, "spike_limit_percentage": 25}, + "resource/cluster": {"attributes": [{"key": "cluster", "value": "test-backend", "action": "upsert"}]}, + "transform/identity": { + "metric_statements": [ + { + "context": "resource", + "statements": [ + 'keep_keys(resource.attributes, ["cluster", "deployment", "engine", "namespace", "node", "replica", "role", "service.instance.id", "service.name"])' + ], + } + ] + }, + "transform/modelplane": { + "metric_statements": [ + { + "context": "metric", + "error_mode": "ignore", + "statements": [ + 'scale_metric(1048576.0) where metric.name == "DCGM_FI_DEV_FB_USED"', + 'scale_metric(0.001) where metric.name == "DCGM_FI_DEV_TOTAL_ENERGY_CONSUMPTION"', + ], + }, + { + "context": "metric", + "statements": [ + 'set(metric.name, "modelplane_frontend_request_duration_seconds") where metric.name == "gen_ai_server_request_duration_seconds"', + 'set(metric.name, "modelplane_frontend_ttft_seconds") where metric.name == "gen_ai_server_time_to_first_token_seconds"', + 'set(metric.name, "modelplane_frontend_tpot_seconds") where metric.name == "gen_ai_server_time_per_output_token_seconds"', + 'set(metric.name, "modelplane_request_ttft_seconds") where metric.name == "vllm:time_to_first_token_seconds"', + 'set(metric.name, "modelplane_request_duration_seconds") where metric.name == "vllm:e2e_request_latency_seconds"', + 'set(metric.name, "modelplane_request_queue_seconds") where metric.name == "vllm:request_queue_time_seconds"', + 'set(metric.name, "modelplane_request_prefill_seconds") where metric.name == "vllm:request_prefill_time_seconds"', + 'set(metric.name, "modelplane_request_decode_seconds") where metric.name == "vllm:request_decode_time_seconds"', + 'set(metric.name, "modelplane_request_input_tokens") where metric.name == "vllm:request_prompt_tokens"', + 'set(metric.name, "modelplane_request_output_tokens") where metric.name == "vllm:request_generation_tokens"', + 'set(metric.name, "modelplane_requests_running") where metric.name == "vllm:num_requests_running"', + 'set(metric.name, "modelplane_requests_waiting") where metric.name == "vllm:num_requests_waiting"', + 'set(metric.name, "modelplane_kv_cache_utilization_ratio") where metric.name == "vllm:kv_cache_usage_perc"', + 'set(metric.name, "modelplane_requests_preempted_total") where metric.name == "vllm:num_preemptions_total"', + 'set(metric.name, "modelplane_prefix_cache_hits_total") where metric.name == "vllm:prefix_cache_hits_total"', + 'set(metric.name, "modelplane_prefix_cache_lookups_total") where metric.name == "vllm:prefix_cache_queries_total"', + 'set(metric.name, "modelplane_requests_running") where metric.name == "sglang:num_running_reqs"', + 'set(metric.name, "modelplane_requests_waiting") where metric.name == "sglang:num_queue_reqs"', + 'set(metric.name, "modelplane_kv_cache_utilization_ratio") where metric.name == "sglang:token_usage"', + 'set(metric.name, "modelplane_request_input_tokens") where metric.name == "sglang:prompt_tokens_histogram"', + 'set(metric.name, "modelplane_request_output_tokens") where metric.name == "sglang:generation_tokens_histogram"', + 'set(metric.name, "modelplane_route_decision_seconds") where metric.name == "llm_d_epp_scheduler_e2e_duration_seconds"', + 'set(metric.name, "modelplane_gpu_memory_used_bytes") where metric.name == "DCGM_FI_DEV_FB_USED"', + 'set(metric.name, "modelplane_gpu_compute_active_ratio") where metric.name == "DCGM_FI_PROF_GR_ENGINE_ACTIVE"', + 'set(metric.name, "modelplane_gpu_tensor_active_ratio") where metric.name == "DCGM_FI_PROF_PIPE_TENSOR_ACTIVE"', + 'set(metric.name, "modelplane_gpu_memory_bandwidth_ratio") where metric.name == "DCGM_FI_PROF_DRAM_ACTIVE"', + 'set(metric.name, "modelplane_gpu_temperature_celsius") where metric.name == "DCGM_FI_DEV_GPU_TEMP"', + 'set(metric.name, "modelplane_gpu_power_watts") where metric.name == "DCGM_FI_DEV_POWER_USAGE"', + 'set(metric.name, "modelplane_energy_joules_total") where metric.name == "DCGM_FI_DEV_TOTAL_ENERGY_CONSUMPTION"', + ], + }, + ] + }, + "groupbyattrs/identity": { + "keys": [ + "cluster", + "namespace", + "deployment", + "replica", + "engine", + "role", + "node", + "service.name", + "service.instance.id", + ] + }, + "filter/modelplane": {"metrics": {"metric": ['not IsMatch(name, "^modelplane_.*")']}}, + "batch": {"timeout": "10s"}, + }, + "exporters": exporters, + "service": service, + } + if extensions is not None: + config["extensions"] = extensions + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "collector-config"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": { + "name": "modelplane-collector", + "namespace": "modelplane-system", + "labels": { + "app.kubernetes.io/name": "modelplane-collector", + "app.kubernetes.io/managed-by": "modelplane", + }, + }, + "data": {"collector.yaml": yaml.safe_dump(config, sort_keys=False)}, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _collector( + *, config_hash: str, volumes: list[dict], volume_mounts: list[dict], env_from: list[dict] | None +) -> fnv1.Resource: + """The composed collector Deployment, restarted by its config's hash and marked ready on arrival.""" + container: dict = { + "name": "collector", + "image": "otel/opentelemetry-collector-contrib:0.161.0", + "args": ["--config=/conf/collector.yaml"], + "volumeMounts": volume_mounts, + "resources": {"requests": {"cpu": "100m", "memory": "256Mi"}, "limits": {"memory": "512Mi"}}, + } + if env_from is not None: + container["envFrom"] = env_from + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "collector"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": { + "name": "modelplane-collector", + "namespace": "modelplane-system", + "labels": { + "app.kubernetes.io/name": "modelplane-collector", + "app.kubernetes.io/managed-by": "modelplane", + }, + }, + "spec": { + "replicas": 1, + "selector": {"matchLabels": {"app.kubernetes.io/name": "modelplane-collector"}}, + "template": { + "metadata": { + "labels": {"app.kubernetes.io/name": "modelplane-collector"}, + "annotations": {"modelplane.ai/config-hash": config_hash}, + }, + "spec": { + "serviceAccountName": "modelplane-collector", + "containers": [container], + "volumes": volumes, + }, + }, + }, + } + }, + "readiness": {"policy": "DeriveFromCelQuery", "celQuery": "object.status.readyReplicas > 0"}, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _telemetry_destination(*, name: str, sinks: list[dict]) -> fnv1.Resource: + """A required TelemetryDestination exporting to sinks.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "TelemetryDestination", + "metadata": {"name": name}, + "spec": {"sinks": sinks}, + } + ) + ) + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +COMPOSE_CASES = [ + # Everything targeting the remote cluster is gated on the ProviderConfigs + # having been observed; Usages reference nothing remote and compose + # immediately. The unready ProviderConfigs keep the composite unready until + # the stack actually renders. + ComposeCase( + name="NothingObserved", + reason="Before its ProviderConfigs are observed, a stack composes only them, unready, and its Usages.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-helm": _helm_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_UNSPECIFIED, + ), + # One Usage per depends_on edge in the joined stack data, then the + # two hand-written edges of the gateway chain. + "usage-cert-manager-by-envoy-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="envoy-gateway", + ), + "usage-ai-gateway-crds-by-ai-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="ai-gateway-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="ai-gateway", + ), + "usage-gateway-namespace-by-gateway-proxy": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-proxy", + ), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-selfsigned-issuer", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="trust-manager", + ), + "usage-kai-scheduler-by-kai-queue-root": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kai-scheduler", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="kai-queue-root", + ), + "usage-kai-scheduler-by-kai-queue": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kai-scheduler", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="kai-queue", + ), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="modelexpress-server", + ), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="modelexpress-server", + ), + "usage-gateway-class-by-gateway": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-class", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway", + ), + "usage-envoy-gateway-by-gateway-class": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="envoy-gateway", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-class", + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + } + ), + ), + ), + # The ProviderConfigs are observed, which opens the gate on the rest of the + # stack. depends_on gates first creation too, so only the dependency-free + # wave renders. Each dependent waits for its dependencies' Ready before it's + # first created: envoy-gateway on cert-manager, ai-gateway on + # ai-gateway-crds, gateway-proxy on gateway-namespace, kai-queue-root and + # kai-queue on kai-scheduler, modelexpress-server on modelexpress-crds, + # gateway-selfsigned-issuer on cert-manager and gateway-namespace, and + # trust-manager on gateway-selfsigned-issuer. + ComposeCase( + name="ProviderConfigsObserved", + reason="Once its ProviderConfigs are observed, a stack renders the components with no dependencies.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": _helm_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "cert-manager": _cert_manager(ready=fnv1.READY_UNSPECIFIED), + "kube-prometheus-stack": _kube_prometheus_stack(ready=fnv1.READY_UNSPECIFIED), + "node-feature-discovery": _node_feature_discovery(ready=fnv1.READY_UNSPECIFIED, wait=False), + "nvidia-dra-driver-gpu": _nvidia_dra_driver_gpu(ready=fnv1.READY_UNSPECIFIED), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crd( + name="inferenceobjectives.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crd( + name="inferencepools.inference.networking.k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crd( + name="inferencepools.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "grove": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-grove-charts"}, + "labels": {"modelplane.ai/resource": "grove"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "grove-charts", + "repository": "oci://ghcr.io/ai-dynamo/grove", + "version": "v0.1.0-alpha.12-rc2", + }, + "namespace": "grove-system", + }, + }, + } + ) + ), + "kai-scheduler": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-kai-scheduler"}, + "labels": {"modelplane.ai/resource": "kai-scheduler"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "kai-scheduler", + "repository": "oci://ghcr.io/kai-scheduler/kai-scheduler", + "version": "v0.16.8", + }, + "namespace": "kai-scheduler", + "wait": True, + "waitTimeout": "10m", + }, + }, + } + ) + ), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "labels": { + "modelplane.ai/resource": "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com" + } + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": _crd( + filename="modelexpress.yaml", name="modelmetadatas.modelexpress.nvidia.com" + ) + }, + }, + } + ) + ), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "labels": { + "modelplane.ai/resource": "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com" + } + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": _crd( + filename="modelexpress.yaml", + name="modelcacheentries.modelexpress.nvidia.com", + ) + }, + }, + } + ) + ), + "modelexpress-server-sa": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-sa"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ServiceAccount", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + } + }, + }, + } + ) + ), + "modelexpress-server-role": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-role"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "Role", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + "rules": [ + { + "apiGroups": ["modelexpress.nvidia.com"], + "resources": ["modelmetadatas", "modelmetadatas/status"], + "verbs": ["get", "list", "create", "update", "patch", "delete"], + }, + { + "apiGroups": [""], + "resources": ["configmaps"], + "verbs": ["get", "list", "create", "update", "patch", "delete"], + }, + { + "apiGroups": ["modelexpress.nvidia.com"], + "resources": ["modelcacheentries", "modelcacheentries/status"], + "verbs": ["get", "list", "create", "update", "patch", "delete"], + }, + ], + } + }, + }, + } + ) + ), + "modelexpress-server-rolebinding": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-rolebinding"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "RoleBinding", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + "subjects": [ + { + "kind": "ServiceAccount", + "name": "modelexpress-server", + "namespace": "default", + } + ], + "roleRef": { + "apiGroup": "rbac.authorization.k8s.io", + "kind": "Role", + "name": "modelexpress-server", + }, + } + }, + }, + } + ) + ), + "modelexpress-server-svc": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-svc"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Service", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + "spec": { + "selector": {"modelplane.ai/modelexpress": "modelexpress-server"}, + "ports": [{"name": "grpc", "port": 8001, "targetPort": 8001}], + }, + } + }, + }, + } + ) + ), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway": _gateway(hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA test-backend.gateways.example.com", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_UNSPECIFIED), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_UNSPECIFIED), + "gateway-client-ca-bundle": _gateway_client_ca_bundle( + ca_crt="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-client-auth": _gateway_client_auth(ready=fnv1.READY_UNSPECIFIED), + "usage-cert-manager-by-envoy-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="envoy-gateway", + ), + "usage-ai-gateway-crds-by-ai-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="ai-gateway-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="ai-gateway", + ), + "usage-gateway-namespace-by-gateway-proxy": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-proxy", + ), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-selfsigned-issuer", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="trust-manager", + ), + "usage-kai-scheduler-by-kai-queue-root": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kai-scheduler", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="kai-queue-root", + ), + "usage-kai-scheduler-by-kai-queue": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kai-scheduler", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="kai-queue", + ), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="modelexpress-server", + ), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="modelexpress-server", + ), + "usage-gateway-class-by-gateway": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-class", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway", + ), + "usage-envoy-gateway-by-gateway-class": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="envoy-gateway", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-class", + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + } + ), + ), + ), + # The ProviderConfigs are ready because they're observed, and the Usages on + # arrival. + ComposeCase( + name="AllReady", + reason="With every rendered resource observed Ready and the gateway's address assigned, the whole stack renders Ready and the address lands in the XR's status.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "envoy-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "ai-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "grove": _observed_ready(), + "kai-scheduler": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-queue": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "modelexpress-server": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway": fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "conditions": [{"type": "Ready", "status": "True"}], + "atProvider": { + "manifest": { + "status": {"addresses": [{"type": "IPAddress", "value": "203.0.113.7"}]} + } + }, + } + } + ) + ), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway={"address": "203.0.113.7"}), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": _helm_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "cert-manager": _cert_manager(ready=fnv1.READY_TRUE), + "kube-prometheus-stack": _kube_prometheus_stack(ready=fnv1.READY_TRUE), + "node-feature-discovery": _node_feature_discovery(ready=fnv1.READY_TRUE, wait=False), + "nvidia-dra-driver-gpu": _nvidia_dra_driver_gpu(ready=fnv1.READY_TRUE), + "envoy-gateway": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-gateway-helm"}, + "labels": {"modelplane.ai/resource": "envoy-gateway"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "gateway-helm", + "repository": "oci://docker.io/envoyproxy", + "version": "v1.8.4", + }, + "namespace": "envoy-gateway-system", + "values": { + "config": { + "envoyGateway": { + "extensionApis": {"enableBackend": True}, + "extensionManager": { + "hooks": { + "xdsTranslator": { + "translation": { + "listener": {"includeAll": True}, + "route": {"includeAll": True}, + "cluster": {"includeAll": True}, + "secret": {"includeAll": True}, + }, + "post": ["Translation", "Cluster", "Route"], + } + }, + "service": { + "fqdn": { + "hostname": "ai-gateway-controller.envoy-ai-gateway-system.svc.cluster.local", + "port": 1063, + } + }, + "backendResources": [ + { + "group": "inference.networking.k8s.io", + "kind": "InferencePool", + "version": "v1", + } + ], + }, + } + } + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_TRUE), + "ai-gateway": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-ai-gateway-helm"}, + "labels": {"modelplane.ai/resource": "ai-gateway"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "ai-gateway-helm", + "repository": "oci://docker.io/envoyproxy", + "version": "v1.1.0", + }, + "namespace": "envoy-ai-gateway-system", + "values": { + "controller": {"logRequestHeaderAttributes": "x-modelplane-caller:caller"} + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crd( + name="inferenceobjectives.inference.networking.x-k8s.io", ready=fnv1.READY_TRUE + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crd( + name="inferencepools.inference.networking.k8s.io", ready=fnv1.READY_TRUE + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crd( + name="inferencepools.inference.networking.x-k8s.io", ready=fnv1.READY_TRUE + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_TRUE), + "gateway-proxy": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "gateway-proxy"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "EnvoyProxy", + "metadata": {"name": "cluster-gateway", "namespace": "modelplane-system"}, + "spec": { + "provider": { + "type": "Kubernetes", + "kubernetes": { + "envoyService": {"externalTrafficPolicy": "Cluster"} + }, + } + }, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "gateway-selfsigned-issuer": _gateway_selfsigned_issuer(), + "trust-manager": _trust_manager(ready=fnv1.READY_TRUE), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_TRUE), + "grove": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-grove-charts"}, + "labels": {"modelplane.ai/resource": "grove"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "grove-charts", + "repository": "oci://ghcr.io/ai-dynamo/grove", + "version": "v0.1.0-alpha.12-rc2", + }, + "namespace": "grove-system", + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "kai-scheduler": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-kai-scheduler"}, + "labels": {"modelplane.ai/resource": "kai-scheduler"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "kai-scheduler", + "repository": "oci://ghcr.io/kai-scheduler/kai-scheduler", + "version": "v0.16.8", + }, + "namespace": "kai-scheduler", + "wait": True, + "waitTimeout": "10m", + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "kai-queue-root": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "kai-queue-root"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "scheduling.run.ai/v2", + "kind": "Queue", + "metadata": {"name": "modelplane-root"}, + "spec": { + "resources": { + "cpu": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, + "gpu": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, + "memory": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, + } + }, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "kai-queue": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "kai-queue"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "scheduling.run.ai/v2", + "kind": "Queue", + "metadata": {"name": "modelplane"}, + "spec": { + "resources": { + "cpu": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, + "gpu": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, + "memory": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, + }, + "parentQueue": "modelplane-root", + }, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "labels": { + "modelplane.ai/resource": "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com" + } + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": _crd( + filename="modelexpress.yaml", name="modelmetadatas.modelexpress.nvidia.com" + ) + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "labels": { + "modelplane.ai/resource": "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com" + } + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": _crd( + filename="modelexpress.yaml", + name="modelcacheentries.modelexpress.nvidia.com", + ) + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "modelexpress-server-sa": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-sa"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ServiceAccount", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "modelexpress-server-role": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-role"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "Role", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + "rules": [ + { + "apiGroups": ["modelexpress.nvidia.com"], + "resources": ["modelmetadatas", "modelmetadatas/status"], + "verbs": ["get", "list", "create", "update", "patch", "delete"], + }, + { + "apiGroups": [""], + "resources": ["configmaps"], + "verbs": ["get", "list", "create", "update", "patch", "delete"], + }, + { + "apiGroups": ["modelexpress.nvidia.com"], + "resources": ["modelcacheentries", "modelcacheentries/status"], + "verbs": ["get", "list", "create", "update", "patch", "delete"], + }, + ], + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "modelexpress-server-rolebinding": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-rolebinding"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "RoleBinding", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + "subjects": [ + { + "kind": "ServiceAccount", + "name": "modelexpress-server", + "namespace": "default", + } + ], + "roleRef": { + "apiGroup": "rbac.authorization.k8s.io", + "kind": "Role", + "name": "modelexpress-server", + }, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "modelexpress-server-svc": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-svc"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Service", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + "spec": { + "selector": {"modelplane.ai/modelexpress": "modelexpress-server"}, + "ports": [{"name": "grpc", "port": 8001, "targetPort": 8001}], + }, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "modelexpress-server": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + "spec": { + "replicas": 1, + "selector": { + "matchLabels": {"modelplane.ai/modelexpress": "modelexpress-server"} + }, + "template": { + "metadata": { + "labels": {"modelplane.ai/modelexpress": "modelexpress-server"} + }, + "spec": { + "serviceAccountName": "modelexpress-server", + "containers": [ + { + "name": "modelexpress-server", + "image": "nvcr.io/nvidia/ai-dynamo/modelexpress-server:0.4.1", + "ports": [{"containerPort": 8001}], + "env": [ + { + "name": "MODEL_EXPRESS_CACHE_DIRECTORY", + "value": "/mnt/models", + }, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_METADATA_BACKEND", + "value": "kubernetes", + }, + { + "name": "POD_NAMESPACE", + "valueFrom": { + "fieldRef": { + "fieldPath": "metadata.namespace" + } + }, + }, + ], + "volumeMounts": [ + {"name": "cache", "mountPath": "/mnt/models"} + ], + "readinessProbe": { + "tcpSocket": {"port": 8001}, + "periodSeconds": 10, + }, + "livenessProbe": { + "tcpSocket": {"port": 8001}, + "periodSeconds": 20, + }, + } + ], + "volumes": [{"name": "cache", "emptyDir": {}}], + }, + }, + }, + } + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "gateway-class": _gateway_class(ready=fnv1.READY_TRUE), + "gateway": _gateway(hostname="test-backend.gateways.example.com", ready=fnv1.READY_TRUE), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA test-backend.gateways.example.com", ready=fnv1.READY_TRUE + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_TRUE), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="test-backend.gateways.example.com", ready=fnv1.READY_TRUE + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_TRUE), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_TRUE), + "gateway-client-ca-bundle": _gateway_client_ca_bundle( + ca_crt="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", ready=fnv1.READY_TRUE + ), + "gateway-client-auth": _gateway_client_auth(ready=fnv1.READY_TRUE), + "usage-cert-manager-by-envoy-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="envoy-gateway", + ), + "usage-ai-gateway-crds-by-ai-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="ai-gateway-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="ai-gateway", + ), + "usage-gateway-namespace-by-gateway-proxy": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-proxy", + ), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-selfsigned-issuer", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="trust-manager", + ), + "usage-kai-scheduler-by-kai-queue-root": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kai-scheduler", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="kai-queue-root", + ), + "usage-kai-scheduler-by-kai-queue": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kai-scheduler", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="kai-queue", + ), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="modelexpress-server", + ), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="modelexpress-server", + ), + "usage-gateway-class-by-gateway": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-class", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway", + ), + "usage-envoy-gateway-by-gateway-class": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="envoy-gateway", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-class", + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + } + ), + ), + ), + # The type isn't forced to GoogleApplicationCredentials, and the secret's own + # namespace wins over the XR's. + ComposeCase( + name="NebiusIdentity", + reason="A Nebius identity secret's type and namespace reach both ProviderConfigs as written.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Nebius", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret( + type="NebiusServiceAccountCredentials", + name="nebius-secret", + key="credentials.json", + namespace="other-ns", + ), + ], + gateway=v1alpha1.Gateway(hostname="test-backend.gateways.example.com"), + gpu=None, + ), + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config( + identity={ + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": {"name": "nebius-secret", "namespace": "other-ns", "key": "credentials.json"}, + }, + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-helm": _helm_provider_config( + identity={ + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": {"name": "nebius-secret", "namespace": "other-ns", "key": "credentials.json"}, + }, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-cert-manager-by-envoy-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="envoy-gateway", + ), + "usage-ai-gateway-crds-by-ai-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="ai-gateway-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="ai-gateway", + ), + "usage-gateway-namespace-by-gateway-proxy": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-proxy", + ), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-selfsigned-issuer", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="trust-manager", + ), + "usage-envoy-gateway-by-gateway-class": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="envoy-gateway", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-class", + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="Gateway test-backend.gateways.example.com not served: no InferenceGateway has published a client CA for this cluster to trust, and serving without one would accept unauthenticated callers", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + } + ), + ), + ), + # The ProviderConfigs are observed, the self-signed Issuer trust-manager + # depends on is Ready, and the CA ConfigMap trust-manager syncs carries the + # certificate back for status. + ComposeCase( + name="ClientCAs", + reason="A cluster with InferenceGateway CAs to trust composes its own PKI, serves mTLS with their bundle, and publishes its own CA.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Standard", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway( + # A full Service FQDN, so the CA certificate's commonName overflows + # the 64-byte X.509 limit. + hostname="gateway-test-backend-12345.modelplane-system.svc.cluster.local", + # Deliberately out of name order, to prove the bundle sorts before concatenating. + clientCAs=[ + v1alpha1.ClientCA(name="fleet-b", certificate="BBB"), + v1alpha1.ClientCA(name="fleet-a", certificate="AAA"), + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + "gateway-selfsigned-issuer": _observed_ready(), + "gateway-ca-configmap": fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}}} + ) + ), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + # The cluster's CA, published for InferenceGateways to trust. + composite=_desired_serving_stack(gateway={"caCertificate": "CLUSTERCA"}), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config(identity=None, ready=fnv1.READY_TRUE), + "provider-config-helm": _helm_provider_config(identity=None, ready=fnv1.READY_TRUE), + "cert-manager": _cert_manager(ready=fnv1.READY_UNSPECIFIED), + "kube-prometheus-stack": _kube_prometheus_stack(ready=fnv1.READY_UNSPECIFIED), + "node-feature-discovery": _node_feature_discovery(ready=fnv1.READY_UNSPECIFIED, wait=False), + "nvidia-dra-driver-gpu": _nvidia_dra_driver_gpu(ready=fnv1.READY_UNSPECIFIED), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crd( + name="inferenceobjectives.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crd( + name="inferencepools.inference.networking.k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crd( + name="inferencepools.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "gateway-selfsigned-issuer": _gateway_selfsigned_issuer(), + "trust-manager": _trust_manager(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "leader-worker-set": _leader_worker_set(), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway": _gateway( + hostname="gateway-test-backend-12345.modelplane-system.svc.cluster.local", + ready=fnv1.READY_UNSPECIFIED, + ), + # The commonName is truncated to the 64-byte X.509 limit. + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA gateway-test-backend-12345.modelplane-syst", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_UNSPECIFIED), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="gateway-test-backend-12345.modelplane-system.svc.cluster.local", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_UNSPECIFIED), + # Every InferenceGateway's CA, sorted by name and concatenated. + "gateway-client-ca-bundle": _gateway_client_ca_bundle( + ca_crt="AAA\nBBB\n", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-client-auth": _gateway_client_auth(ready=fnv1.READY_UNSPECIFIED), + "usage-cert-manager-by-envoy-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="envoy-gateway", + ), + "usage-ai-gateway-crds-by-ai-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="ai-gateway-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="ai-gateway", + ), + "usage-gateway-namespace-by-gateway-proxy": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-proxy", + ), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-selfsigned-issuer", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="trust-manager", + ), + "usage-gateway-class-by-gateway": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-class", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway", + ), + "usage-envoy-gateway-by-gateway-class": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="envoy-gateway", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-class", + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + } + ), + ), + ), + # Every PKI resource must be tracked for readiness: mark_readiness marks only + # the keys compose_gateway_pki returns, so one composed but not returned + # would silently hold the cluster un-Ready. Each is observed Ready here, so a + # key dropped from the rendered list fails this case. + ComposeCase( + name="PKIReady", + reason="Each gateway PKI resource observed Ready is marked ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Standard", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway( + hostname="gateway-test-backend-12345.modelplane-system.svc.cluster.local", + clientCAs=[ + v1alpha1.ClientCA(name="fleet-b", certificate="BBB"), + v1alpha1.ClientCA(name="fleet-a", certificate="AAA"), + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + "gateway-selfsigned-issuer": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + # The CA ConfigMap's data, alongside its Ready condition. + "gateway-ca-configmap": fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "conditions": [{"type": "Ready", "status": "True"}], + "atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}, + } + } + ) + ), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway={"caCertificate": "CLUSTERCA"}), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config(identity=None, ready=fnv1.READY_TRUE), + "provider-config-helm": _helm_provider_config(identity=None, ready=fnv1.READY_TRUE), + "cert-manager": _cert_manager(ready=fnv1.READY_UNSPECIFIED), + "kube-prometheus-stack": _kube_prometheus_stack(ready=fnv1.READY_UNSPECIFIED), + "node-feature-discovery": _node_feature_discovery(ready=fnv1.READY_UNSPECIFIED, wait=False), + "nvidia-dra-driver-gpu": _nvidia_dra_driver_gpu(ready=fnv1.READY_UNSPECIFIED), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crd( + name="inferenceobjectives.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crd( + name="inferencepools.inference.networking.k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crd( + name="inferencepools.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "gateway-selfsigned-issuer": _gateway_selfsigned_issuer(), + "trust-manager": _trust_manager(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "leader-worker-set": _leader_worker_set(), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway": _gateway( + hostname="gateway-test-backend-12345.modelplane-system.svc.cluster.local", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA gateway-test-backend-12345.modelplane-syst", + ready=fnv1.READY_TRUE, + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_TRUE), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="gateway-test-backend-12345.modelplane-system.svc.cluster.local", ready=fnv1.READY_TRUE + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_TRUE), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_TRUE), + "gateway-client-ca-bundle": _gateway_client_ca_bundle(ca_crt="AAA\nBBB\n", ready=fnv1.READY_TRUE), + "gateway-client-auth": _gateway_client_auth(ready=fnv1.READY_TRUE), + "usage-cert-manager-by-envoy-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="envoy-gateway", + ), + "usage-ai-gateway-crds-by-ai-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="ai-gateway-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="ai-gateway", + ), + "usage-gateway-namespace-by-gateway-proxy": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-proxy", + ), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-selfsigned-issuer", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="trust-manager", + ), + "usage-gateway-class-by-gateway": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-class", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway", + ), + "usage-envoy-gateway-by-gateway-class": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="envoy-gateway", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-class", + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + } + ), + ), + ), + # The GatewayClass and the cluster's own PKI are composed, so the CA is ready + # to publish when the first InferenceGateway's CA arrives. The Gateway, the + # client CA bundle and the policy demanding a client certificate aren't, and + # nor is the Usage holding the GatewayClass for the Gateway. + ComposeCase( + name="NoClientCAs", + reason="A cluster with no InferenceGateway CA to trust withholds its Gateway and warns.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Standard", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway(hostname="gw.clusters.example.com"), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config(identity=None, ready=fnv1.READY_TRUE), + "provider-config-helm": _helm_provider_config(identity=None, ready=fnv1.READY_TRUE), + "cert-manager": _cert_manager(ready=fnv1.READY_UNSPECIFIED), + "kube-prometheus-stack": _kube_prometheus_stack(ready=fnv1.READY_UNSPECIFIED), + "node-feature-discovery": _node_feature_discovery(ready=fnv1.READY_UNSPECIFIED, wait=False), + "nvidia-dra-driver-gpu": _nvidia_dra_driver_gpu(ready=fnv1.READY_UNSPECIFIED), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crd( + name="inferenceobjectives.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crd( + name="inferencepools.inference.networking.k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crd( + name="inferencepools.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "leader-worker-set": _leader_worker_set(), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA gw.clusters.example.com", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_UNSPECIFIED), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="gw.clusters.example.com", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_UNSPECIFIED), + "usage-cert-manager-by-envoy-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="envoy-gateway", + ), + "usage-ai-gateway-crds-by-ai-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="ai-gateway-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="ai-gateway", + ), + "usage-gateway-namespace-by-gateway-proxy": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-proxy", + ), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-selfsigned-issuer", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="trust-manager", + ), + "usage-envoy-gateway-by-gateway-class": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="envoy-gateway", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-class", + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="Gateway gw.clusters.example.com not served: no InferenceGateway has published a client CA for this cluster to trust, and serving without one would accept unauthenticated callers", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + } + ), + ), + ), + # The install gate composes a component once its dependencies are observed + # Ready, so the chain is observed up to the per-pool driver, which the DRA + # driver waits on. The derived Usages hold the operator release and the + # ConfigMap until the per-pool driver is gone. The stack has only the one + # pool, so this can't show a pool that isn't flagged going without an + # NVIDIADriver. + ComposeCase( + name="NvLinkDisabled", + reason=( + "A Civo stack flagging a pool for NVLink disable composes gpu-operator in NVIDIADriver-CRD mode, the " + "kernel module ConfigMap, and an NVIDIADriver selecting that pool's nodes." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Civo", + stack="Standard", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway(hostname="test-backend.gateways.example.com"), + gpu=v1alpha1.Gpu(pools=[v1alpha1.Pool(name="h100-pool", disableNvLink=True)]), + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + "cert-manager": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "gpu-operator": _observed_ready(), + "nvlink-disable-config-gpu-operator": _observed_ready(), + "nvlink-disable-config-nvidia-kernel-config": _observed_ready(), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config(identity=None, ready=fnv1.READY_TRUE), + "provider-config-helm": _helm_provider_config(identity=None, ready=fnv1.READY_TRUE), + "cert-manager": _cert_manager(ready=fnv1.READY_TRUE), + "kube-prometheus-stack": _kube_prometheus_stack(ready=fnv1.READY_UNSPECIFIED), + "node-feature-discovery": _node_feature_discovery(ready=fnv1.READY_TRUE, wait=True), + "gpu-operator": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-gpu-operator"}, + "labels": {"modelplane.ai/resource": "gpu-operator"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "gpu-operator", + "repository": "https://helm.ngc.nvidia.com/nvidia", + "version": "v26.3.3", + }, + "namespace": "gpu-operator", + "wait": True, + "waitTimeout": "10m", + "values": { + "ccManager": {"enabled": False}, + "cdi": {"default": True, "enabled": True}, + "daemonsets": { + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ] + }, + "dcgm": {"enabled": False}, + "dcgmExporter": {"enabled": False}, + "devicePlugin": {"enabled": False}, + "driver": { + "enabled": True, + "maxParallelUpgrades": 5, + "rdma": {"enabled": False}, + "useOpenKernelModules": True, + "version": "580.173.02", + "nvidiaDriverCRD": {"enabled": True, "deployDefaultCR": True}, + }, + "fullnameOverride": "gpu-operator", + "gdrcopy": {"enabled": False}, + "gfd": {"enabled": True}, + "kataSandboxDevicePlugin": {"enabled": False}, + "migManager": {"enabled": False}, + "nfd": {"enabled": False}, + "operator": { + "resources": { + "limits": {"cpu": "500m", "memory": "700Mi"}, + "requests": {"cpu": "200m", "memory": "300Mi"}, + }, + "tolerations": [], + "upgradeCRD": True, + }, + "toolkit": {"enabled": False}, + "validator": { + "plugin": {"env": [{"name": "WITH_WORKLOAD", "value": "false"}]} + }, + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "envoy-gateway": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-gateway-helm"}, + "labels": {"modelplane.ai/resource": "envoy-gateway"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "gateway-helm", + "repository": "oci://docker.io/envoyproxy", + "version": "v1.8.4", + }, + "namespace": "envoy-gateway-system", + "values": { + "config": { + "envoyGateway": { + "extensionApis": {"enableBackend": True}, + "extensionManager": { + "hooks": { + "xdsTranslator": { + "translation": { + "listener": {"includeAll": True}, + "route": {"includeAll": True}, + "cluster": {"includeAll": True}, + "secret": {"includeAll": True}, + }, + "post": ["Translation", "Cluster", "Route"], + } + }, + "service": { + "fqdn": { + "hostname": "ai-gateway-controller.envoy-ai-gateway-system.svc.cluster.local", + "port": 1063, + } + }, + "backendResources": [ + { + "group": "inference.networking.k8s.io", + "kind": "InferencePool", + "version": "v1", + } + ], + }, + } + } + }, + }, + }, + } + ) + ), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crd( + name="inferenceobjectives.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crd( + name="inferencepools.inference.networking.k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crd( + name="inferencepools.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "leader-worker-set": _leader_worker_set(), + "nvlink-disable-config-gpu-operator": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "labels": {"modelplane.ai/resource": "nvlink-disable-config-gpu-operator"} + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Namespace", + "metadata": {"name": "gpu-operator"}, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "nvlink-disable-config-nvidia-kernel-config": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "labels": {"modelplane.ai/resource": "nvlink-disable-config-nvidia-kernel-config"} + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "nvidia-kernel-config", "namespace": "gpu-operator"}, + "data": {"nvidia.conf": "options nvidia NVreg_NvLinkDisable=1"}, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "nvlink-disabled-driver-h100-pool": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "nvlink-disabled-driver-h100-pool"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "nvidia.com/v1alpha1", + "kind": "NVIDIADriver", + "metadata": {"name": "nvlink-disabled-h100-pool"}, + "spec": { + "driverType": "gpu", + "version": "580.173.02", + "useOpenKernelModules": True, + "nodeSelector": {"modelplane.ai/pool": "h100-pool"}, + "kernelModuleConfig": {"name": "nvidia-kernel-config"}, + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + } + }, + "readiness": { + "celQuery": 'object.status.state == "ready"', + "policy": "DeriveFromCelQuery", + }, + }, + } + ) + ), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA test-backend.gateways.example.com", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_UNSPECIFIED), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_UNSPECIFIED), + "usage-node-feature-discovery-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="node-feature-discovery", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-cert-manager-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvidia-dra-driver-gpu", + ), + "usage-nvlink-disabled-driver-h100-pool-by-nvidia-dra-driver-gpu": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="nvlink-disabled-driver-h100-pool", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvidia-dra-driver-gpu", + ), + "usage-cert-manager-by-envoy-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="envoy-gateway", + ), + "usage-ai-gateway-crds-by-ai-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="ai-gateway-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="ai-gateway", + ), + "usage-gateway-namespace-by-gateway-proxy": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-proxy", + ), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-selfsigned-issuer", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="trust-manager", + ), + "usage-gpu-operator-by-nvlink-disabled-driver-h100-pool": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="gpu-operator", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="nvlink-disabled-driver-h100-pool", + ), + "usage-nvlink-disable-config-gpu-operator-by-nvlink-disabled-driver-h100-pool": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="nvlink-disable-config-gpu-operator", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="nvlink-disabled-driver-h100-pool", + ), + "usage-nvlink-disable-config-nvidia-kernel-config-by-nvlink-disabled-driver-h100-pool": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="nvlink-disable-config-nvidia-kernel-config", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="nvlink-disabled-driver-h100-pool", + ), + "usage-envoy-gateway-by-gateway-class": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="envoy-gateway", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-class", + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="Gateway test-backend.gateways.example.com not served: no InferenceGateway has published a client CA for this cluster to trust, and serving without one would accept unauthenticated callers", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + } + ), + ), + ), + # The gpu-operator release keeps its ClusterPolicy-managed driver. + ComposeCase( + name="NvLinkEnabled", + reason=( + "A Civo stack flagging no pool for NVLink disable composes the stock GPU operator, with no NVIDIADriver " + "or kernel module ConfigMap." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Civo", + stack="Standard", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway(hostname="test-backend.gateways.example.com"), + gpu=v1alpha1.Gpu(pools=[v1alpha1.Pool(name="l40s-pool", disableNvLink=False)]), + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + "gpu-operator": _observed_ready(), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config(identity=None, ready=fnv1.READY_TRUE), + "provider-config-helm": _helm_provider_config(identity=None, ready=fnv1.READY_TRUE), + "cert-manager": _cert_manager(ready=fnv1.READY_UNSPECIFIED), + "kube-prometheus-stack": _kube_prometheus_stack(ready=fnv1.READY_UNSPECIFIED), + "node-feature-discovery": _node_feature_discovery(ready=fnv1.READY_UNSPECIFIED, wait=True), + "gpu-operator": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-gpu-operator"}, + "labels": {"modelplane.ai/resource": "gpu-operator"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "gpu-operator", + "repository": "https://helm.ngc.nvidia.com/nvidia", + "version": "v26.3.3", + }, + "namespace": "gpu-operator", + "wait": True, + "waitTimeout": "10m", + "values": { + "ccManager": {"enabled": False}, + "cdi": {"default": True, "enabled": True}, + "daemonsets": { + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ] + }, + "dcgm": {"enabled": False}, + "dcgmExporter": {"enabled": False}, + "devicePlugin": {"enabled": False}, + "driver": { + "enabled": True, + "maxParallelUpgrades": 5, + "rdma": {"enabled": False}, + "useOpenKernelModules": True, + "version": "580.173.02", + }, + "fullnameOverride": "gpu-operator", + "gdrcopy": {"enabled": False}, + "gfd": {"enabled": True}, + "kataSandboxDevicePlugin": {"enabled": False}, + "migManager": {"enabled": False}, + "nfd": {"enabled": False}, + "operator": { + "resources": { + "limits": {"cpu": "500m", "memory": "700Mi"}, + "requests": {"cpu": "200m", "memory": "300Mi"}, + }, + "tolerations": [], + "upgradeCRD": True, + }, + "toolkit": {"enabled": False}, + "validator": { + "plugin": {"env": [{"name": "WITH_WORKLOAD", "value": "false"}]} + }, + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + # Not _nvidia_dra_driver_gpu: Civo's DRA driver points + # nvidiaDriverRoot at the gpu-operator's driver, and this is + # the only case that writes it out. + "nvidia-dra-driver-gpu": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-dra-driver-nvidia-gpu"}, + "labels": {"modelplane.ai/resource": "nvidia-dra-driver-gpu"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "dra-driver-nvidia-gpu", + "repository": "oci://registry.k8s.io/dra-driver-nvidia/charts", + "version": "0.4.1", + }, + "namespace": "nvidia-dra-driver", + "values": { + "gpuResourcesEnabledOverride": True, + "nvidiaDriverRoot": "/run/nvidia/driver", + "resources": {"computeDomains": {"enabled": False}}, + }, + }, + }, + } + ) + ), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crd( + name="inferenceobjectives.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crd( + name="inferencepools.inference.networking.k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crd( + name="inferencepools.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "leader-worker-set": _leader_worker_set(), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA test-backend.gateways.example.com", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_UNSPECIFIED), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_UNSPECIFIED), + "usage-node-feature-discovery-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="node-feature-discovery", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-cert-manager-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvidia-dra-driver-gpu", + ), + "usage-cert-manager-by-envoy-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="envoy-gateway", + ), + "usage-ai-gateway-crds-by-ai-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="ai-gateway-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="ai-gateway", + ), + "usage-gateway-namespace-by-gateway-proxy": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-proxy", + ), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-selfsigned-issuer", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="trust-manager", + ), + "usage-envoy-gateway-by-gateway-class": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="envoy-gateway", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-class", + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="Gateway test-backend.gateways.example.com not served: no InferenceGateway has published a client CA for this cluster to trust, and serving without one would accept unauthenticated callers", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + } + ), + ), + ), + # Everything else the stack composes is Ready only once its observed Ready + # condition says so, because the fleet can't serve without it. The collector + # only watches, so gating on it would put placing a replica behind exporting + # a metric: one destination pointing at an endpoint that has gone away would + # take every InferenceCluster in the fleet out of Ready and stop the + # scheduler. + ComposeCase( + name="CollectorNotGated", + reason="With a TelemetryDestination, a stack composes the collector, Ready before anything has observed it.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="GKE", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + }, + ), + required_resources={ + "destinations": fnv1.Resources( + items=[ + _telemetry_destination( + name="acme", + sinks=[{"name": "primary", "type": "otlphttp", "endpoint": "https://otel.acme.example"}], + ) + ] + ), + "mappings": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": _helm_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "cert-manager": _gke_cert_manager(), + "node-feature-discovery": _gke_node_feature_discovery(), + "nodewright-operator": _nodewright_operator(), + "prometheus-operator-crds": _prometheus_operator_crds(), + "gpu-operator-pre-manifests-gpu-operator": _gpu_operator_pre_manifests_gpu_operator(), + "gpu-operator-pre-manifests-aicr-gke-critical-pods": _gpu_operator_pre_manifests_aicr_gke_critical_pods(), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crd( + name="inferenceobjectives.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crd( + name="inferencepools.inference.networking.k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crd( + name="inferencepools.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "leader-worker-set": _leader_worker_set(), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway": _gateway(hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA test-backend.gateways.example.com", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_UNSPECIFIED), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_UNSPECIFIED), + "gateway-client-ca-bundle": _gateway_client_ca_bundle( + ca_crt="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-client-auth": _gateway_client_auth(ready=fnv1.READY_UNSPECIFIED), + "collector-serviceaccount": _collector_service_account(), + "collector-clusterrole": _collector_cluster_role(), + "collector-clusterrolebinding": _collector_cluster_role_binding(), + "collector-config": _collector_config( + exporters={"otlphttp/primary": {"endpoint": "https://otel.acme.example"}}, + pipeline_exporters=["otlphttp/primary"], + extensions=None, + service_extensions=None, + ), + "collector": _collector( + config_hash="80bb34e74b135d14", + volumes=[{"name": "config", "configMap": {"name": "modelplane-collector"}}], + volume_mounts=[{"name": "config", "mountPath": "/conf"}], + env_from=None, + ), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="kube-prometheus-stack", + ), + "usage-node-feature-discovery-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="node-feature-discovery", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-cert-manager-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-kube-prometheus-stack-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gpu-operator-pre-manifests-gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gpu-operator-pre-manifests-aicr-gke-critical-pods", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="k8s-ephemeral-storage-metrics", + ), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="k8s-ephemeral-storage-metrics", + ), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvidia-dra-driver-gpu", + ), + "usage-cert-manager-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-gpu-operator-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-prometheus-operator-crds-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-kube-prometheus-stack-by-prometheus-adapter": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="prometheus-adapter", + ), + "usage-cert-manager-by-envoy-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="envoy-gateway", + ), + "usage-ai-gateway-crds-by-ai-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="ai-gateway-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="ai-gateway", + ), + "usage-gateway-namespace-by-gateway-proxy": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-proxy", + ), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-selfsigned-issuer", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="trust-manager", + ), + "usage-gateway-class-by-gateway": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-class", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway", + ), + "usage-envoy-gateway-by-gateway-class": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="envoy-gateway", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-class", + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + } + ), + ), + ), + # A TelemetryDestination is cluster-scoped on the control plane, and the + # collector runs on every workload cluster in the fleet. Resolving the Secret + # and stopping there would leave the Deployment mounting a Secret nothing out + # there creates, so the pod would never start. The Secret is required from + # modelplane-system because, unqualified, the requirement would match a + # Secret of that name anywhere. Its data is copied verbatim: re-encoding + # would corrupt a credential that isn't text. + ComposeCase( + name="CollectorCredential", + reason=( + "A destination's credential is required from modelplane-system and composed onto the cluster whose " + "collector mounts it." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="GKE", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + }, + ), + required_resources={ + "destinations": fnv1.Resources( + items=[ + _telemetry_destination( + name="acme", + sinks=[ + { + "name": "primary", + "type": "otlphttp", + "endpoint": "https://otel.acme.example", + "secretRef": {"name": "telemetry-credentials"}, + "auth": {"bearerTokenKey": "token"}, + } + ], + ) + ] + ), + "mappings": fnv1.Resources(), + "collector-secret-primary": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "telemetry-credentials", "namespace": "modelplane-system"}, + "data": {"token": "c2hoaGg="}, + } + ) + ) + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": _helm_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "cert-manager": _gke_cert_manager(), + "node-feature-discovery": _gke_node_feature_discovery(), + "nodewright-operator": _nodewright_operator(), + "prometheus-operator-crds": _prometheus_operator_crds(), + "gpu-operator-pre-manifests-gpu-operator": _gpu_operator_pre_manifests_gpu_operator(), + "gpu-operator-pre-manifests-aicr-gke-critical-pods": _gpu_operator_pre_manifests_aicr_gke_critical_pods(), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crd( + name="inferenceobjectives.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crd( + name="inferencepools.inference.networking.k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crd( + name="inferencepools.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "leader-worker-set": _leader_worker_set(), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway": _gateway(hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA test-backend.gateways.example.com", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_UNSPECIFIED), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_UNSPECIFIED), + "gateway-client-ca-bundle": _gateway_client_ca_bundle( + ca_crt="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-client-auth": _gateway_client_auth(ready=fnv1.READY_UNSPECIFIED), + "collector-serviceaccount": _collector_service_account(), + "collector-clusterrole": _collector_cluster_role(), + "collector-clusterrolebinding": _collector_cluster_role_binding(), + "collector-config": _collector_config( + exporters={ + "otlphttp/primary": { + "endpoint": "https://otel.acme.example", + "auth": {"authenticator": "bearertokenauth/primary"}, + } + }, + pipeline_exporters=["otlphttp/primary"], + extensions={"bearertokenauth/primary": {"filename": "/etc/modelplane/telemetry/primary/token"}}, + service_extensions=["bearertokenauth/primary"], + ), + "collector-secret-primary": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "collector-secret-primary"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Secret", + "metadata": { + "name": "telemetry-credentials", + "namespace": "modelplane-system", + }, + "data": {"token": "c2hoaGg="}, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "collector": _collector( + config_hash="173c93b5f6048d87", + volumes=[ + {"name": "config", "configMap": {"name": "modelplane-collector"}}, + {"name": "credentials-primary", "secret": {"secretName": "telemetry-credentials"}}, + ], + volume_mounts=[ + {"name": "config", "mountPath": "/conf"}, + { + "name": "credentials-primary", + "mountPath": "/etc/modelplane/telemetry/primary", + "readOnly": True, + }, + ], + env_from=[{"secretRef": {"name": "telemetry-credentials"}}], + ), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="kube-prometheus-stack", + ), + "usage-node-feature-discovery-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="node-feature-discovery", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-cert-manager-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-kube-prometheus-stack-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gpu-operator-pre-manifests-gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gpu-operator-pre-manifests-aicr-gke-critical-pods", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="k8s-ephemeral-storage-metrics", + ), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="k8s-ephemeral-storage-metrics", + ), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvidia-dra-driver-gpu", + ), + "usage-cert-manager-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-gpu-operator-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-prometheus-operator-crds-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-kube-prometheus-stack-by-prometheus-adapter": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="prometheus-adapter", + ), + "usage-cert-manager-by-envoy-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="envoy-gateway", + ), + "usage-ai-gateway-crds-by-ai-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="ai-gateway-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="ai-gateway", + ), + "usage-gateway-namespace-by-gateway-proxy": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-proxy", + ), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-selfsigned-issuer", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="trust-manager", + ), + "usage-gateway-class-by-gateway": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-class", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway", + ), + "usage-envoy-gateway-by-gateway-class": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="envoy-gateway", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-class", + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + "collector-secret-primary": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="telemetry-credentials", + namespace="modelplane-system", + ), + } + ), + ), + ), + # A second backend is a second object, not an edit to a singleton. Picking + # one destination and warning about the rest would mean a team adding an + # export has to edit an object another team owns, and gets silence if they + # create their own instead. + ComposeCase( + name="TwoDestinations", + reason="Every TelemetryDestination contributes its sinks to the one collector.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="GKE", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + }, + ), + required_resources={ + "destinations": fnv1.Resources( + items=[ + _telemetry_destination( + name="acme", + sinks=[{"name": "vendor", "type": "otlphttp", "endpoint": "https://otel.vendor.example"}], + ), + _telemetry_destination( + name="zeta", + sinks=[ + {"name": "prom", "type": "prometheus_remote_write", "endpoint": "https://p.example/w"} + ], + ), + ] + ), + "mappings": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": _helm_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "cert-manager": _gke_cert_manager(), + "node-feature-discovery": _gke_node_feature_discovery(), + "nodewright-operator": _nodewright_operator(), + "prometheus-operator-crds": _prometheus_operator_crds(), + "gpu-operator-pre-manifests-gpu-operator": _gpu_operator_pre_manifests_gpu_operator(), + "gpu-operator-pre-manifests-aicr-gke-critical-pods": _gpu_operator_pre_manifests_aicr_gke_critical_pods(), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crd( + name="inferenceobjectives.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crd( + name="inferencepools.inference.networking.k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crd( + name="inferencepools.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "leader-worker-set": _leader_worker_set(), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway": _gateway(hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA test-backend.gateways.example.com", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_UNSPECIFIED), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_UNSPECIFIED), + "gateway-client-ca-bundle": _gateway_client_ca_bundle( + ca_crt="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-client-auth": _gateway_client_auth(ready=fnv1.READY_UNSPECIFIED), + "collector-serviceaccount": _collector_service_account(), + "collector-clusterrole": _collector_cluster_role(), + "collector-clusterrolebinding": _collector_cluster_role_binding(), + "collector-config": _collector_config( + exporters={ + "otlphttp/vendor": {"endpoint": "https://otel.vendor.example"}, + "prometheus_remote_write/prom": { + "resource_to_telemetry_conversion": {"enabled": True}, + "endpoint": "https://p.example/w", + }, + }, + pipeline_exporters=["otlphttp/vendor", "prometheus_remote_write/prom"], + extensions=None, + service_extensions=None, + ), + "collector": _collector( + config_hash="5676f43101e4b6b1", + volumes=[{"name": "config", "configMap": {"name": "modelplane-collector"}}], + volume_mounts=[{"name": "config", "mountPath": "/conf"}], + env_from=None, + ), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="kube-prometheus-stack", + ), + "usage-node-feature-discovery-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="node-feature-discovery", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-cert-manager-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-kube-prometheus-stack-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gpu-operator-pre-manifests-gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gpu-operator-pre-manifests-aicr-gke-critical-pods", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="k8s-ephemeral-storage-metrics", + ), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="k8s-ephemeral-storage-metrics", + ), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvidia-dra-driver-gpu", + ), + "usage-cert-manager-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-gpu-operator-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-prometheus-operator-crds-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-kube-prometheus-stack-by-prometheus-adapter": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="prometheus-adapter", + ), + "usage-cert-manager-by-envoy-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="envoy-gateway", + ), + "usage-ai-gateway-crds-by-ai-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="ai-gateway-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="ai-gateway", + ), + "usage-gateway-namespace-by-gateway-proxy": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-proxy", + ), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-selfsigned-issuer", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="trust-manager", + ), + "usage-gateway-class-by-gateway": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-class", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway", + ), + "usage-envoy-gateway-by-gateway-class": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="envoy-gateway", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-class", + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + } + ), + ), + ), + # A sink names the collector's exporter instance, so two of them under one + # name would be one exporter with two meanings. Dropping one beats failing + # the whole fleet's telemetry over a name. acme is listed first and sorts + # first, so this can't show the destinations being sorted before they merge. + ComposeCase( + name="SinkNameClash", + reason="Of two destinations naming one sink, the first keeps it and the other is dropped with a warning.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="GKE", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + }, + ), + required_resources={ + "destinations": fnv1.Resources( + items=[ + _telemetry_destination( + name="acme", + sinks=[{"name": "primary", "type": "otlphttp", "endpoint": "https://a.example"}], + ), + _telemetry_destination( + name="zeta", + sinks=[{"name": "primary", "type": "otlphttp", "endpoint": "https://z.example"}], + ), + ] + ), + "mappings": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": _helm_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "cert-manager": _gke_cert_manager(), + "node-feature-discovery": _gke_node_feature_discovery(), + "nodewright-operator": _nodewright_operator(), + "prometheus-operator-crds": _prometheus_operator_crds(), + "gpu-operator-pre-manifests-gpu-operator": _gpu_operator_pre_manifests_gpu_operator(), + "gpu-operator-pre-manifests-aicr-gke-critical-pods": _gpu_operator_pre_manifests_aicr_gke_critical_pods(), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crd( + name="inferenceobjectives.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crd( + name="inferencepools.inference.networking.k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crd( + name="inferencepools.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "leader-worker-set": _leader_worker_set(), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway": _gateway(hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA test-backend.gateways.example.com", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_UNSPECIFIED), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_UNSPECIFIED), + "gateway-client-ca-bundle": _gateway_client_ca_bundle( + ca_crt="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-client-auth": _gateway_client_auth(ready=fnv1.READY_UNSPECIFIED), + "collector-serviceaccount": _collector_service_account(), + "collector-clusterrole": _collector_cluster_role(), + "collector-clusterrolebinding": _collector_cluster_role_binding(), + "collector-config": _collector_config( + exporters={"otlphttp/primary": {"endpoint": "https://a.example"}}, + pipeline_exporters=["otlphttp/primary"], + extensions=None, + service_extensions=None, + ), + "collector": _collector( + config_hash="460bb87917962297", + volumes=[{"name": "config", "configMap": {"name": "modelplane-collector"}}], + volume_mounts=[{"name": "config", "mountPath": "/conf"}], + env_from=None, + ), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="kube-prometheus-stack", + ), + "usage-node-feature-discovery-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="node-feature-discovery", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-cert-manager-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-kube-prometheus-stack-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gpu-operator-pre-manifests-gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gpu-operator-pre-manifests-aicr-gke-critical-pods", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="k8s-ephemeral-storage-metrics", + ), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="k8s-ephemeral-storage-metrics", + ), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvidia-dra-driver-gpu", + ), + "usage-cert-manager-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-gpu-operator-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-prometheus-operator-crds-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-kube-prometheus-stack-by-prometheus-adapter": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="prometheus-adapter", + ), + "usage-cert-manager-by-envoy-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="envoy-gateway", + ), + "usage-ai-gateway-crds-by-ai-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="ai-gateway-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="ai-gateway", + ), + "usage-gateway-namespace-by-gateway-proxy": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-proxy", + ), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-selfsigned-issuer", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="trust-manager", + ), + "usage-gateway-class-by-gateway": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-class", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway", + ), + "usage-envoy-gateway-by-gateway-class": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="envoy-gateway", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-class", + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="Ignored sink primary in TelemetryDestination zeta: already defined by a TelemetryDestination sorting earlier.", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + } + ), + ), + ), + # A CRD validates on write, not on what it already stored, so a mapping + # written against an older schema comes back on read as it was stored. + # Parsing it raises, and raising would fail the whole pipeline step: the + # stack would compose nothing and the fleet would stop placing replicas, + # because one telemetry object is out of date. Seen on a real cluster, where + # a mapping predating a required field did it. The collector's config has + # the built-in mappings alone. + ComposeCase( + name="StaleMapping", + reason=( + "A MetricMapping that no longer matches the schema is dropped with a warning, and the stack still composes." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="GKE", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + }, + ), + required_resources={ + "destinations": fnv1.Resources( + items=[ + _telemetry_destination( + name="acme", + sinks=[{"name": "primary", "type": "otlphttp", "endpoint": "https://otel.acme.example"}], + ) + ] + ), + "mappings": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "MetricMapping", + "metadata": {"name": "stale"}, + "spec": { + "metrics": [ + { + "from": "old_engine_transfer", + "to": "modelplane_request_kv_transfer_seconds", + "fromUnit": "Centiseconds", # A unit the enum no longer carries. + } + ] + }, + } + ) + ) + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": _helm_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "cert-manager": _gke_cert_manager(), + "node-feature-discovery": _gke_node_feature_discovery(), + "nodewright-operator": _nodewright_operator(), + "prometheus-operator-crds": _prometheus_operator_crds(), + "gpu-operator-pre-manifests-gpu-operator": _gpu_operator_pre_manifests_gpu_operator(), + "gpu-operator-pre-manifests-aicr-gke-critical-pods": _gpu_operator_pre_manifests_aicr_gke_critical_pods(), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crd( + name="inferenceobjectives.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crd( + name="inferencepools.inference.networking.k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crd( + name="inferencepools.inference.networking.x-k8s.io", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "leader-worker-set": _leader_worker_set(), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway": _gateway(hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA test-backend.gateways.example.com", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_UNSPECIFIED), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_UNSPECIFIED), + "gateway-client-ca-bundle": _gateway_client_ca_bundle( + ca_crt="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-client-auth": _gateway_client_auth(ready=fnv1.READY_UNSPECIFIED), + "collector-serviceaccount": _collector_service_account(), + "collector-clusterrole": _collector_cluster_role(), + "collector-clusterrolebinding": _collector_cluster_role_binding(), + "collector-config": _collector_config( + exporters={"otlphttp/primary": {"endpoint": "https://otel.acme.example"}}, + pipeline_exporters=["otlphttp/primary"], + extensions=None, + service_extensions=None, + ), + "collector": _collector( + config_hash="80bb34e74b135d14", + volumes=[{"name": "config", "configMap": {"name": "modelplane-collector"}}], + volume_mounts=[{"name": "config", "mountPath": "/conf"}], + env_from=None, + ), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="kube-prometheus-stack", + ), + "usage-node-feature-discovery-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="node-feature-discovery", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-cert-manager-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-kube-prometheus-stack-by-gpu-operator": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gpu-operator-pre-manifests-gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gpu-operator-pre-manifests-aicr-gke-critical-pods", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="gpu-operator", + ), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="k8s-ephemeral-storage-metrics", + ), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="k8s-ephemeral-storage-metrics", + ), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvidia-dra-driver-gpu", + ), + "usage-cert-manager-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-gpu-operator-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="gpu-operator", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-prometheus-operator-crds-by-nvsentinel": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="prometheus-operator-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="nvsentinel", + ), + "usage-kube-prometheus-stack-by-prometheus-adapter": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="kube-prometheus-stack", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="prometheus-adapter", + ), + "usage-cert-manager-by-envoy-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="envoy-gateway", + ), + "usage-ai-gateway-crds-by-ai-gateway": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="ai-gateway-crds", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="ai-gateway", + ), + "usage-gateway-namespace-by-gateway-proxy": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-proxy", + ), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="cert-manager", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-namespace", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-selfsigned-issuer", + ), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-selfsigned-issuer", + by_api_version="helm.m.crossplane.io/v1beta1", + by_kind="Release", + by_key="trust-manager", + ), + "usage-gateway-class-by-gateway": _usage( + of_api_version="kubernetes.m.crossplane.io/v1alpha1", + of_kind="Object", + of_key="gateway-class", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway", + ), + "usage-envoy-gateway-by-gateway-class": _usage( + of_api_version="helm.m.crossplane.io/v1beta1", + of_kind="Release", + of_key="envoy-gateway", + by_api_version="kubernetes.m.crossplane.io/v1alpha1", + by_kind="Object", + by_key="gateway-class", + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="Ignoring MetricMapping stale: it does not match the current schema (1 problems), so nothing it asks for is collected.", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "destinations": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="TelemetryDestination" + ), + "mappings": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="MetricMapping"), + } + ), + ), + ), +] + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: ComposeCase) -> None: + """RunFunction composes the serving stack, gated on what's observed.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want), case.reason + + +# The composed-resource key a component renders under is its identity: +# renaming one deletes and recreates the remote resource (for an Object +# holding a CRD, the CRD and its CRs). This pins the full key set per +# cloud and stack, including the Usage keys derived from depends_on, as +# reviewed literals. A failure here means the stack data changed a key - +# make sure that's intended, then update that case's want and the release +# notes. +# +# Each case's XR has an InferenceGateway CA to trust, so the Gateway and its +# client-auth policy are included. Every key the case expects is observed +# Ready, so the depends_on install gate opens and the full stack renders; a +# key the function doesn't render still fails the comparison. The identity +# secret every XR carries, and the observed Usages, don't affect the keys. +# The cases compare keys rather than whole responses because every cloud's +# rendered half would restate its stack data, much of it generated, where +# COMPOSE_CASES already covers how components render. +COMPOSED_RESOURCE_KEYS_CASES = [ + ComposedResourceKeysCase( + name="EKSStandard", + reason=( + "On EKS, the Standard stack composes the gateway, the common components, EKS's generated half " + "and the Standard stack's own components, each under its pinned key." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="EKS", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "gpu-operator": _observed_ready(), + "k8s-ephemeral-storage-metrics": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nodewright-operator": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "nvsentinel": _observed_ready(), + "prometheus-adapter": _observed_ready(), + "prometheus-operator-crds": _observed_ready(), + "usage-cert-manager-by-gpu-operator": _observed_ready(), + "usage-cert-manager-by-nvsentinel": _observed_ready(), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _observed_ready(), + "usage-gpu-operator-by-nvsentinel": _observed_ready(), + "usage-kube-prometheus-stack-by-gpu-operator": _observed_ready(), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-kube-prometheus-stack-by-prometheus-adapter": _observed_ready(), + "usage-node-feature-discovery-by-gpu-operator": _observed_ready(), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _observed_ready(), + "usage-prometheus-operator-crds-by-nvsentinel": _observed_ready(), + "leader-worker-set": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The EKS half, generated from aicr. + "cert-manager", + "gpu-operator", + "k8s-ephemeral-storage-metrics", + "kube-prometheus-stack", + "node-feature-discovery", + "nodewright-operator", + "nvidia-dra-driver-gpu", + "nvsentinel", + "prometheus-adapter", + "prometheus-operator-crds", + "usage-cert-manager-by-gpu-operator", + "usage-cert-manager-by-nvsentinel", + "usage-gpu-operator-by-nvidia-dra-driver-gpu", + "usage-gpu-operator-by-nvsentinel", + "usage-kube-prometheus-stack-by-gpu-operator", + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics", + "usage-kube-prometheus-stack-by-prometheus-adapter", + "usage-node-feature-discovery-by-gpu-operator", + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics", + "usage-prometheus-operator-crds-by-kube-prometheus-stack", + "usage-prometheus-operator-crds-by-nvsentinel", + # The Standard stack. + "leader-worker-set", + }, + ), + ComposedResourceKeysCase( + name="EKSDynamo", + reason=( + "On EKS, the Dynamo stack composes the gateway, the common components, EKS's generated half " + "and the Dynamo stack's own components, each under its pinned key." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="EKS", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "gpu-operator": _observed_ready(), + "k8s-ephemeral-storage-metrics": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nodewright-operator": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "nvsentinel": _observed_ready(), + "prometheus-adapter": _observed_ready(), + "prometheus-operator-crds": _observed_ready(), + "usage-cert-manager-by-gpu-operator": _observed_ready(), + "usage-cert-manager-by-nvsentinel": _observed_ready(), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _observed_ready(), + "usage-gpu-operator-by-nvsentinel": _observed_ready(), + "usage-kube-prometheus-stack-by-gpu-operator": _observed_ready(), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-kube-prometheus-stack-by-prometheus-adapter": _observed_ready(), + "usage-node-feature-discovery-by-gpu-operator": _observed_ready(), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _observed_ready(), + "usage-prometheus-operator-crds-by-nvsentinel": _observed_ready(), + "grove": _observed_ready(), + "kai-queue": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-scheduler": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "usage-kai-scheduler-by-kai-queue": _observed_ready(), + "usage-kai-scheduler-by-kai-queue-root": _observed_ready(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The EKS half, generated from aicr. + "cert-manager", + "gpu-operator", + "k8s-ephemeral-storage-metrics", + "kube-prometheus-stack", + "node-feature-discovery", + "nodewright-operator", + "nvidia-dra-driver-gpu", + "nvsentinel", + "prometheus-adapter", + "prometheus-operator-crds", + "usage-cert-manager-by-gpu-operator", + "usage-cert-manager-by-nvsentinel", + "usage-gpu-operator-by-nvidia-dra-driver-gpu", + "usage-gpu-operator-by-nvsentinel", + "usage-kube-prometheus-stack-by-gpu-operator", + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics", + "usage-kube-prometheus-stack-by-prometheus-adapter", + "usage-node-feature-discovery-by-gpu-operator", + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics", + "usage-prometheus-operator-crds-by-kube-prometheus-stack", + "usage-prometheus-operator-crds-by-nvsentinel", + # The Dynamo stack. + "grove", + "kai-queue", + "kai-queue-root", + "kai-scheduler", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-server", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-sa", + "modelexpress-server-svc", + "usage-kai-scheduler-by-kai-queue", + "usage-kai-scheduler-by-kai-queue-root", + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", + }, + ), + ComposedResourceKeysCase( + name="AKSStandard", + reason=( + "On AKS, the Standard stack composes the gateway, the common components, AKS's generated half " + "and the Standard stack's own components, each under its pinned key." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="AKS", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "gpu-operator": _observed_ready(), + "k8s-ephemeral-storage-metrics": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nodewright-operator": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "nvsentinel": _observed_ready(), + "prometheus-adapter": _observed_ready(), + "prometheus-operator-crds": _observed_ready(), + "usage-cert-manager-by-gpu-operator": _observed_ready(), + "usage-cert-manager-by-nvsentinel": _observed_ready(), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _observed_ready(), + "usage-gpu-operator-by-nvsentinel": _observed_ready(), + "usage-kube-prometheus-stack-by-gpu-operator": _observed_ready(), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-kube-prometheus-stack-by-prometheus-adapter": _observed_ready(), + "usage-node-feature-discovery-by-gpu-operator": _observed_ready(), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _observed_ready(), + "usage-prometheus-operator-crds-by-nvsentinel": _observed_ready(), + "gpu-operator-manifests": _observed_ready(), + "usage-gpu-operator-by-gpu-operator-manifests": _observed_ready(), + "leader-worker-set": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The AKS half, generated from aicr. + "cert-manager", + "gpu-operator", + "k8s-ephemeral-storage-metrics", + "kube-prometheus-stack", + "node-feature-discovery", + "nodewright-operator", + "nvidia-dra-driver-gpu", + "nvsentinel", + "prometheus-adapter", + "prometheus-operator-crds", + "usage-cert-manager-by-gpu-operator", + "usage-cert-manager-by-nvsentinel", + "usage-gpu-operator-by-nvidia-dra-driver-gpu", + "usage-gpu-operator-by-nvsentinel", + "usage-kube-prometheus-stack-by-gpu-operator", + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics", + "usage-kube-prometheus-stack-by-prometheus-adapter", + "usage-node-feature-discovery-by-gpu-operator", + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics", + "usage-prometheus-operator-crds-by-kube-prometheus-stack", + "usage-prometheus-operator-crds-by-nvsentinel", + # AKS additionally carries the gpu-operator's toolkit-hardening manifest. + "gpu-operator-manifests", + "usage-gpu-operator-by-gpu-operator-manifests", + # The Standard stack. + "leader-worker-set", + }, + ), + ComposedResourceKeysCase( + name="AKSDynamo", + reason=( + "On AKS, the Dynamo stack composes the gateway, the common components, AKS's generated half " + "and the Dynamo stack's own components, each under its pinned key." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="AKS", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "gpu-operator": _observed_ready(), + "k8s-ephemeral-storage-metrics": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nodewright-operator": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "nvsentinel": _observed_ready(), + "prometheus-adapter": _observed_ready(), + "prometheus-operator-crds": _observed_ready(), + "usage-cert-manager-by-gpu-operator": _observed_ready(), + "usage-cert-manager-by-nvsentinel": _observed_ready(), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _observed_ready(), + "usage-gpu-operator-by-nvsentinel": _observed_ready(), + "usage-kube-prometheus-stack-by-gpu-operator": _observed_ready(), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-kube-prometheus-stack-by-prometheus-adapter": _observed_ready(), + "usage-node-feature-discovery-by-gpu-operator": _observed_ready(), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _observed_ready(), + "usage-prometheus-operator-crds-by-nvsentinel": _observed_ready(), + "gpu-operator-manifests": _observed_ready(), + "usage-gpu-operator-by-gpu-operator-manifests": _observed_ready(), + "grove": _observed_ready(), + "kai-queue": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-scheduler": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "usage-kai-scheduler-by-kai-queue": _observed_ready(), + "usage-kai-scheduler-by-kai-queue-root": _observed_ready(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The AKS half, generated from aicr. + "cert-manager", + "gpu-operator", + "k8s-ephemeral-storage-metrics", + "kube-prometheus-stack", + "node-feature-discovery", + "nodewright-operator", + "nvidia-dra-driver-gpu", + "nvsentinel", + "prometheus-adapter", + "prometheus-operator-crds", + "usage-cert-manager-by-gpu-operator", + "usage-cert-manager-by-nvsentinel", + "usage-gpu-operator-by-nvidia-dra-driver-gpu", + "usage-gpu-operator-by-nvsentinel", + "usage-kube-prometheus-stack-by-gpu-operator", + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics", + "usage-kube-prometheus-stack-by-prometheus-adapter", + "usage-node-feature-discovery-by-gpu-operator", + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics", + "usage-prometheus-operator-crds-by-kube-prometheus-stack", + "usage-prometheus-operator-crds-by-nvsentinel", + # AKS additionally carries the gpu-operator's toolkit-hardening manifest. + "gpu-operator-manifests", + "usage-gpu-operator-by-gpu-operator-manifests", + # The Dynamo stack. + "grove", + "kai-queue", + "kai-queue-root", + "kai-scheduler", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-server", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-sa", + "modelexpress-server-svc", + "usage-kai-scheduler-by-kai-queue", + "usage-kai-scheduler-by-kai-queue-root", + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", + }, + ), + ComposedResourceKeysCase( + name="GKEStandard", + reason=( + "On GKE, the Standard stack composes the gateway, the common components, GKE's generated half " + "and the Standard stack's own components, each under its pinned key." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="GKE", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "gpu-operator": _observed_ready(), + "k8s-ephemeral-storage-metrics": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nodewright-operator": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "nvsentinel": _observed_ready(), + "prometheus-adapter": _observed_ready(), + "prometheus-operator-crds": _observed_ready(), + "usage-cert-manager-by-gpu-operator": _observed_ready(), + "usage-cert-manager-by-nvsentinel": _observed_ready(), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _observed_ready(), + "usage-gpu-operator-by-nvsentinel": _observed_ready(), + "usage-kube-prometheus-stack-by-gpu-operator": _observed_ready(), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-kube-prometheus-stack-by-prometheus-adapter": _observed_ready(), + "usage-node-feature-discovery-by-gpu-operator": _observed_ready(), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _observed_ready(), + "usage-prometheus-operator-crds-by-nvsentinel": _observed_ready(), + "gpu-operator-pre-manifests-aicr-gke-critical-pods": _observed_ready(), + "gpu-operator-pre-manifests-gpu-operator": _observed_ready(), + "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator": _observed_ready(), + "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator": _observed_ready(), + "leader-worker-set": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The GKE half, generated from aicr. + "cert-manager", + "gpu-operator", + "k8s-ephemeral-storage-metrics", + "kube-prometheus-stack", + "node-feature-discovery", + "nodewright-operator", + "nvidia-dra-driver-gpu", + "nvsentinel", + "prometheus-adapter", + "prometheus-operator-crds", + "usage-cert-manager-by-gpu-operator", + "usage-cert-manager-by-nvsentinel", + "usage-gpu-operator-by-nvidia-dra-driver-gpu", + "usage-gpu-operator-by-nvsentinel", + "usage-kube-prometheus-stack-by-gpu-operator", + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics", + "usage-kube-prometheus-stack-by-prometheus-adapter", + "usage-node-feature-discovery-by-gpu-operator", + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics", + "usage-prometheus-operator-crds-by-kube-prometheus-stack", + "usage-prometheus-operator-crds-by-nvsentinel", + # GKE additionally carries the critical-pods ResourceQuota aicr's + # bundler synthesizes as a gpu-operator pre-manifest (GKE rejects + # system-node-critical pods in a namespace without one; aicr#915). + "gpu-operator-pre-manifests-aicr-gke-critical-pods", + "gpu-operator-pre-manifests-gpu-operator", + "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator", + "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator", + # The Standard stack. + "leader-worker-set", + }, + ), + ComposedResourceKeysCase( + name="GKEDynamo", + reason=( + "On GKE, the Dynamo stack composes the gateway, the common components, GKE's generated half " + "and the Dynamo stack's own components, each under its pinned key." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="GKE", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "gpu-operator": _observed_ready(), + "k8s-ephemeral-storage-metrics": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nodewright-operator": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "nvsentinel": _observed_ready(), + "prometheus-adapter": _observed_ready(), + "prometheus-operator-crds": _observed_ready(), + "usage-cert-manager-by-gpu-operator": _observed_ready(), + "usage-cert-manager-by-nvsentinel": _observed_ready(), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _observed_ready(), + "usage-gpu-operator-by-nvsentinel": _observed_ready(), + "usage-kube-prometheus-stack-by-gpu-operator": _observed_ready(), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-kube-prometheus-stack-by-prometheus-adapter": _observed_ready(), + "usage-node-feature-discovery-by-gpu-operator": _observed_ready(), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _observed_ready(), + "usage-prometheus-operator-crds-by-nvsentinel": _observed_ready(), + "gpu-operator-pre-manifests-aicr-gke-critical-pods": _observed_ready(), + "gpu-operator-pre-manifests-gpu-operator": _observed_ready(), + "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator": _observed_ready(), + "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator": _observed_ready(), + "grove": _observed_ready(), + "kai-queue": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-scheduler": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "usage-kai-scheduler-by-kai-queue": _observed_ready(), + "usage-kai-scheduler-by-kai-queue-root": _observed_ready(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The GKE half, generated from aicr. + "cert-manager", + "gpu-operator", + "k8s-ephemeral-storage-metrics", + "kube-prometheus-stack", + "node-feature-discovery", + "nodewright-operator", + "nvidia-dra-driver-gpu", + "nvsentinel", + "prometheus-adapter", + "prometheus-operator-crds", + "usage-cert-manager-by-gpu-operator", + "usage-cert-manager-by-nvsentinel", + "usage-gpu-operator-by-nvidia-dra-driver-gpu", + "usage-gpu-operator-by-nvsentinel", + "usage-kube-prometheus-stack-by-gpu-operator", + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics", + "usage-kube-prometheus-stack-by-prometheus-adapter", + "usage-node-feature-discovery-by-gpu-operator", + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics", + "usage-prometheus-operator-crds-by-kube-prometheus-stack", + "usage-prometheus-operator-crds-by-nvsentinel", + # GKE additionally carries the critical-pods ResourceQuota aicr's + # bundler synthesizes as a gpu-operator pre-manifest (GKE rejects + # system-node-critical pods in a namespace without one; aicr#915). + "gpu-operator-pre-manifests-aicr-gke-critical-pods", + "gpu-operator-pre-manifests-gpu-operator", + "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator", + "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator", + # The Dynamo stack. + "grove", + "kai-queue", + "kai-queue-root", + "kai-scheduler", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-server", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-sa", + "modelexpress-server-svc", + "usage-kai-scheduler-by-kai-queue", + "usage-kai-scheduler-by-kai-queue-root", + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", + }, + ), + ComposedResourceKeysCase( + name="NebiusStandard", + reason=( + "On Nebius, the Standard stack composes the gateway, the common components, Nebius's hand-written half " + "and the Standard stack's own components, each under its pinned key." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Nebius", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "leader-worker-set": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The hand-written Nebius half. + "cert-manager", + "kube-prometheus-stack", + "node-feature-discovery", + "nvidia-dra-driver-gpu", + # The Standard stack. + "leader-worker-set", + }, + ), + ComposedResourceKeysCase( + name="NebiusDynamo", + reason=( + "On Nebius, the Dynamo stack composes the gateway, the common components, Nebius's hand-written half " + "and the Dynamo stack's own components, each under its pinned key." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Nebius", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "grove": _observed_ready(), + "kai-queue": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-scheduler": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "usage-kai-scheduler-by-kai-queue": _observed_ready(), + "usage-kai-scheduler-by-kai-queue-root": _observed_ready(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The hand-written Nebius half. + "cert-manager", + "kube-prometheus-stack", + "node-feature-discovery", + "nvidia-dra-driver-gpu", + # The Dynamo stack. + "grove", + "kai-queue", + "kai-queue-root", + "kai-scheduler", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-server", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-sa", + "modelexpress-server-svc", + "usage-kai-scheduler-by-kai-queue", + "usage-kai-scheduler-by-kai-queue-root", + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", + }, + ), + ComposedResourceKeysCase( + name="VultrStandard", + reason=( + "On Vultr, the Standard stack composes the gateway, the common components, Vultr's hand-written half " + "and the Standard stack's own components, each under its pinned key." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Vultr", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "leader-worker-set": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The hand-written Vultr half. VKE pre-installs NFD via its managed GPU + # Operator add-on, so it carries no node-feature-discovery of its own. + "cert-manager", + "kube-prometheus-stack", + "nvidia-dra-driver-gpu", + # The Standard stack. + "leader-worker-set", + }, + ), + ComposedResourceKeysCase( + name="VultrDynamo", + reason=( + "On Vultr, the Dynamo stack composes the gateway, the common components, Vultr's hand-written half " + "and the Dynamo stack's own components, each under its pinned key." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Vultr", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "grove": _observed_ready(), + "kai-queue": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-scheduler": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "usage-kai-scheduler-by-kai-queue": _observed_ready(), + "usage-kai-scheduler-by-kai-queue-root": _observed_ready(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The hand-written Vultr half. VKE pre-installs NFD via its managed GPU + # Operator add-on, so it carries no node-feature-discovery of its own. + "cert-manager", + "kube-prometheus-stack", + "nvidia-dra-driver-gpu", + # The Dynamo stack. + "grove", + "kai-queue", + "kai-queue-root", + "kai-scheduler", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-server", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-sa", + "modelexpress-server-svc", + "usage-kai-scheduler-by-kai-queue", + "usage-kai-scheduler-by-kai-queue-root", + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", + }, + ), + ComposedResourceKeysCase( + name="CivoStandard", + reason=( + "On Civo, the Standard stack composes the gateway, the common components, Civo's hand-written half " + "and the Standard stack's own components, each under its pinned key." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Civo", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "gpu-operator": _observed_ready(), + "usage-cert-manager-by-gpu-operator": _observed_ready(), + "usage-node-feature-discovery-by-gpu-operator": _observed_ready(), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _observed_ready(), + "leader-worker-set": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The hand-written Civo half. Civo pre-installs no GPU stack at all, + # so it's the one that carries the GPU Operator, driver included, + # with the Usages for its depends_on edges and the DRA driver's edge + # onto it. + "cert-manager", + "kube-prometheus-stack", + "node-feature-discovery", + "nvidia-dra-driver-gpu", + "gpu-operator", + "usage-cert-manager-by-gpu-operator", + "usage-node-feature-discovery-by-gpu-operator", + "usage-gpu-operator-by-nvidia-dra-driver-gpu", + # The Standard stack. + "leader-worker-set", + }, + ), + ComposedResourceKeysCase( + name="CivoDynamo", + reason=( + "On Civo, the Dynamo stack composes the gateway, the common components, Civo's hand-written half " + "and the Dynamo stack's own components, each under its pinned key." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Civo", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "gpu-operator": _observed_ready(), + "usage-cert-manager-by-gpu-operator": _observed_ready(), + "usage-node-feature-discovery-by-gpu-operator": _observed_ready(), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _observed_ready(), + "grove": _observed_ready(), + "kai-queue": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-scheduler": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "usage-kai-scheduler-by-kai-queue": _observed_ready(), + "usage-kai-scheduler-by-kai-queue-root": _observed_ready(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The hand-written Civo half. Civo pre-installs no GPU stack at all, + # so it's the one that carries the GPU Operator, driver included, + # with the Usages for its depends_on edges and the DRA driver's edge + # onto it. + "cert-manager", + "kube-prometheus-stack", + "node-feature-discovery", + "nvidia-dra-driver-gpu", + "gpu-operator", + "usage-cert-manager-by-gpu-operator", + "usage-node-feature-discovery-by-gpu-operator", + "usage-gpu-operator-by-nvidia-dra-driver-gpu", + # The Dynamo stack. + "grove", + "kai-queue", + "kai-queue-root", + "kai-scheduler", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-server", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-sa", + "modelexpress-server-svc", + "usage-kai-scheduler-by-kai-queue", + "usage-kai-scheduler-by-kai-queue-root", + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", + }, + ), + ComposedResourceKeysCase( + name="ExistingStandard", + reason=( + "On Existing, the Standard stack composes the gateway, the common components, the hand-written half " + "and the Standard stack's own components, each under its pinned key." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "leader-worker-set": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The hand-written Existing half. + "cert-manager", + "kube-prometheus-stack", + "node-feature-discovery", + "nvidia-dra-driver-gpu", + # The Standard stack. + "leader-worker-set", + }, + ), + ComposedResourceKeysCase( + name="ExistingDynamo", + reason=( + "On Existing, the Dynamo stack composes the gateway, the common components, the hand-written half " + "and the Dynamo stack's own components, each under its pinned key." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + gpu=None, + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "grove": _observed_ready(), + "kai-queue": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-scheduler": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "usage-kai-scheduler-by-kai-queue": _observed_ready(), + "usage-kai-scheduler-by-kai-queue-root": _observed_ready(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The hand-written Existing half. + "cert-manager", + "kube-prometheus-stack", + "node-feature-discovery", + "nvidia-dra-driver-gpu", + # The Dynamo stack. + "grove", + "kai-queue", + "kai-queue-root", + "kai-scheduler", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-server", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-sa", + "modelexpress-server-svc", + "usage-kai-scheduler-by-kai-queue", + "usage-kai-scheduler-by-kai-queue-root", + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", + }, + ), +] + + +@pytest.mark.parametrize("case", COMPOSED_RESOURCE_KEYS_CASES, ids=lambda case: case.name) +def test_composed_resource_keys(case: ComposedResourceKeysCase) -> None: + """RunFunction composes exactly the pinned resource keys for a cloud and stack.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert set(got.desired.resources) == case.want, case.reason + + +# These call fn._cluster_name rather than RunFunction. Through RunFunction the +# name surfaces only in the collector config's resource/cluster processor, and +# reaching that takes a TelemetryDestination and a whole stack's response. +CLUSTER_NAME_CASES = [ + ClusterNameCase( + name="CompositeLabel", + reason="A stack Crossplane labels with its composite is named for the composite an operator named.", + xr=v1alpha1.ServingStack( + metadata=metav1.ObjectMeta(name="local-serving-stack-d4206", labels={"crossplane.io/composite": "local"}), + spec=v1alpha1.Spec( + cloud="Existing", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway(hostname="test-backend.gateways.example.com"), + ), + ), + want="local", + ), + # Better a generated name on the series than none at all. + ClusterNameCase( + name="NoCompositeLabel", + reason="A stack with no composite label is named for itself, generated suffix and all.", + xr=v1alpha1.ServingStack( + metadata=metav1.ObjectMeta(name="local-serving-stack-d4206"), + spec=v1alpha1.Spec( + cloud="Existing", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway(hostname="test-backend.gateways.example.com"), + ), + ), + want="local-serving-stack-d4206", + ), +] + + +@pytest.mark.parametrize("case", CLUSTER_NAME_CASES, ids=lambda case: case.name) +def test_cluster_name(case: ClusterNameCase) -> None: + """The cluster every exported series is stamped with is named as its operator named it.""" + assert fn._cluster_name(case.xr) == case.want, case.reason diff --git a/functions/compose-serving-stack/tests/test_stacks.py b/functions/compose-serving-stack/tests/test_stacks.py index 592a6ec28..d8e5a3e8c 100644 --- a/functions/compose-serving-stack/tests/test_stacks.py +++ b/functions/compose-serving-stack/tests/test_stacks.py @@ -14,74 +14,109 @@ """Tests for the serving stack component lists. -The join itself is the assertion: components() fails closed on -duplicate keys and on depends_on edges the join didn't produce, so -iterating every cloud and stack pair gates every list - including the -generated ones, once mapped - without involving fn.py. +The join itself is the assertion: join() fails closed on duplicate +keys and on depends_on edges the join didn't produce, so iterating +every cloud and stack pair gates every list - including the generated +ones, once mapped - without involving fn.py. The tests after it are +properties every joined stack must hold, each checked over the whole +stack at once so a failure lists every component that breaks it. +JOIN_FAILS_CLOSED_CASES then checks that join rejects a cloud or stack +it doesn't know. The last two tests cover Civo's NVLink transform, +which rewrites a joined stack to disable NVLink on the pools that need +it. """ +import dataclasses + import pytest from function import stacks from function.stacks.clouds import civo +@dataclasses.dataclass +class JoinFailsClosedCase: + """A test case for stacks.join rejecting a cloud or stack.""" + + name: str + reason: str + cloud: str + stack: str + want: str + + +@dataclasses.dataclass +class WithNvLinkDisabledCase: + """A test case for civo.with_nvlink_disabled.""" + + name: str + reason: str + components: list[stacks.Component] + pools: list[str] + want: list[stacks.Component] + + @pytest.mark.parametrize("stack", stacks.stacks()) @pytest.mark.parametrize("cloud", stacks.clouds()) -def test_every_cloud_and_stack_joins(cloud: stacks.Cloud, stack: stacks.Stack) -> None: +def test_join_not_empty(cloud: stacks.Cloud, stack: stacks.Stack) -> None: """Every cloud and stack pair joins into a non-empty stack.""" - got = stacks.join(cloud, stack) - assert got, "a joined stack can't be empty" + assert stacks.join(cloud, stack), "a joined stack can't be empty" @pytest.mark.parametrize("stack", stacks.stacks()) @pytest.mark.parametrize("cloud", stacks.clouds()) -def test_charts_have_reserved_release_names(cloud: stacks.Cloud, stack: stacks.Stack) -> None: +def test_release_names(cloud: stacks.Cloud, stack: stacks.Stack) -> None: """Every Chart's Helm release is named mp-.""" - for c in stacks.join(cloud, stack): - if isinstance(c, stacks.Chart): - assert c.release == f"mp-{c.chart}", ( - f"{c.key}: release names are mp-: stable across upgrades, reserved to Modelplane" - ) + charts = [c for c in stacks.join(cloud, stack) if isinstance(c, stacks.Chart)] + assert {c.key: c.release for c in charts} == {c.key: f"mp-{c.chart}" for c in charts}, ( + "release names are mp-: stable across upgrades, reserved to Modelplane" + ) @pytest.mark.parametrize("stack", stacks.stacks()) @pytest.mark.parametrize("cloud", stacks.clouds()) -def test_manifests_are_populated(cloud: stacks.Cloud, stack: stacks.Stack) -> None: +def test_manifests_populated(cloud: stacks.Cloud, stack: stacks.Stack) -> None: """Every Manifests entry carries at least one manifest.""" - for c in stacks.join(cloud, stack): - if isinstance(c, stacks.Manifests): - assert c.manifests, f"{c.key}: a Manifests entry can't be empty" + empty = [c.key for c in stacks.join(cloud, stack) if isinstance(c, stacks.Manifests) and not c.manifests] + assert empty == [], "a Manifests entry can't be empty" @pytest.mark.parametrize("stack", stacks.stacks()) @pytest.mark.parametrize("cloud", stacks.clouds()) -def test_multi_doc_manifests_derive_per_doc_keys(cloud: stacks.Cloud, stack: stacks.Stack) -> None: - """A multi-doc Manifests entry renders a key per doc, and anything else renders its own key.""" - for c in stacks.join(cloud, stack): - keys = stacks.components.doc_keys(c) +def test_doc_keys(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """doc_keys agrees with a restatement of its own rule.""" + # This restates doc_keys line for line, so it only catches one copy + # changing without the other. COMPOSED_RESOURCE_KEYS_CASES in test_fn.py + # pins the real keys, as literals. + joined = stacks.join(cloud, stack) + want = {} + for c in joined: if isinstance(c, stacks.Chart) or len(c.manifests) == 1: - assert keys == [c.key] - continue - assert keys == [f"{c.key}-{doc['metadata']['name']}" for doc in c.manifests], ( - f"{c.key}: a multi-doc bundle renders one Object per doc, keyed -" - ) + want[c.key] = [c.key] + else: + want[c.key] = [f"{c.key}-{doc['metadata']['name']}" for doc in c.manifests] + assert {c.key: stacks.components.doc_keys(c) for c in joined} == want, ( + "a multi-doc bundle renders one Object per doc, keyed -" + ) @pytest.mark.parametrize("stack", stacks.stacks()) @pytest.mark.parametrize("cloud", stacks.clouds()) -def test_ready_entries_are_single_doc(cloud: stacks.Cloud, stack: stacks.Stack) -> None: +def test_ready_single_doc(cloud: stacks.Cloud, stack: stacks.Stack) -> None: """A Manifests entry with a readiness query carries a single manifest.""" # A readiness CEL query applies to every doc in an entry, so an # entry carrying one keeps to a single manifest - a Service or # ServiceAccount has no status conditions to satisfy it. - for c in stacks.join(cloud, stack): - if isinstance(c, stacks.Manifests) and c.ready is not None: - assert len(c.manifests) == 1, c.key + not_single = [ + c.key + for c in stacks.join(cloud, stack) + if isinstance(c, stacks.Manifests) and c.ready is not None and len(c.manifests) != 1 + ] + assert not_single == [], "a readiness query applies to every doc, so an entry carrying one has one manifest" @pytest.mark.parametrize("stack", stacks.stacks()) @pytest.mark.parametrize("cloud", stacks.clouds()) -def test_depended_on_charts_wait(cloud: stacks.Cloud, stack: stacks.Stack) -> None: +def test_dependencies_wait(cloud: stacks.Cloud, stack: stacks.Stack) -> None: """Every Chart another component depends on sets wait.""" # A chart another component depends on renders with helm --wait, # so its Ready means healthy and the install gate orders @@ -89,16 +124,14 @@ def test_depended_on_charts_wait(cloud: stacks.Cloud, stack: stacks.Stack) -> No # gate would open the moment Helm accepted the manifests. joined = stacks.join(cloud, stack) depended_on = {dep for c in joined for dep in c.depends_on} - for c in joined: - if isinstance(c, stacks.Chart) and c.key in depended_on: - assert c.wait, f"{c.key}: a depended-on chart must set wait" + not_waiting = [c.key for c in joined if isinstance(c, stacks.Chart) and c.key in depended_on and not c.wait] + assert not_waiting == [], "a depended-on chart must set wait" @pytest.mark.parametrize("stack", stacks.stacks()) @pytest.mark.parametrize("cloud", stacks.clouds()) def test_no_wildcard_tolerations(cloud: stacks.Cloud, stack: stacks.Stack) -> None: """No component of a joined stack carries a keyless toleration.""" - # A keyless toleration tolerates every taint, so the pod lands # on tainted GPU nodes: control-plane charts squat on # accelerated capacity and their eviction stalls autoscaler @@ -107,13 +140,13 @@ def test_no_wildcard_tolerations(cloud: stacks.Cloud, stack: stacks.Stack) -> No # (TOLERATIONS in generate.py). This pins that no keyless # toleration survives in any joined stack, chart values and # manifests alike. + wildcards = [] + def check(node: object, where: str) -> None: if isinstance(node, dict): for key, val in node.items(): if key == "tolerations" and isinstance(val, list): - for toleration in val: - assert isinstance(toleration, dict), f"keyless (wildcard) toleration in {where}" - assert "key" in toleration, f"keyless (wildcard) toleration in {where}" + wildcards.extend((where, t) for t in val if not (isinstance(t, dict) and "key" in t)) else: check(val, where) elif isinstance(node, list): @@ -122,81 +155,293 @@ def check(node: object, where: str) -> None: for c in stacks.join(cloud, stack): check(c.values if isinstance(c, stacks.Chart) else c.manifests, c.key) - - -def test_unknown_cloud_and_stack_fail_closed() -> None: - """join rejects an unknown cloud or stack.""" - # The Literal types reject these at type-checking time; this - # exercises the runtime guard behind them, which catches the API - # and the stacks package disagreeing on a value. - with pytest.raises(ValueError, match="unknown cloud 'Mars'"): - stacks.join("Mars", "Standard") # ty: ignore[invalid-argument-type] - with pytest.raises(ValueError, match="unknown stack 'Turbo'"): - stacks.join("Nebius", "Turbo") # ty: ignore[invalid-argument-type] - - -def test_gpu_operator_switches_to_nvidia_driver_crd() -> None: - """with_nvlink_disabled switches the gpu-operator chart to NVIDIADriver-CRD mode.""" - # The chart's default NVIDIADriver (deployDefaultCR) keeps driving - # pools the transform doesn't name, so flipping modes changes - # nothing for them. - got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["h100-pool"]) - op = next(c for c in got if isinstance(c, stacks.Chart) and c.key == "gpu-operator") - assert op.values is not None - assert op.values["driver"]["nvidiaDriverCRD"] == {"enabled": True, "deployDefaultCR": True} - - -def test_each_pool_gets_its_own_driver() -> None: - """with_nvlink_disabled adds one NVIDIADriver per named pool, selecting that pool's nodes.""" - got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["pool-a", "pool-b"]) - drivers = [c for c in got if isinstance(c, stacks.Manifests) and c.key.startswith("nvlink-disabled-driver-")] - assert [c.key for c in drivers] == ["nvlink-disabled-driver-pool-a", "nvlink-disabled-driver-pool-b"] - for c, pool in zip(drivers, ["pool-a", "pool-b"], strict=True): - # Ready entries keep to a single doc, and gate on the - # operator-populated state so the DRA driver's install - # gate orders on driver health. - assert len(c.manifests) == 1, pool - assert c.ready is not None, pool - assert c.depends_on == ["gpu-operator", "nvlink-disable-config"], pool - spec = c.manifests[0]["spec"] - assert spec["nodeSelector"] == {"modelplane.ai/pool": pool}, pool - assert spec["kernelModuleConfig"] == {"name": "nvidia-kernel-config"}, pool - - -def test_driver_pin_mirrors_the_chart() -> None: - """A per-pool NVIDIADriver installs the same driver as the chart's default CR.""" - # One review moves both: the per-pool NVIDIADriver must install - # the same driver the chart's default CR does. - got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["h100-pool"]) - op = next(c for c in got if isinstance(c, stacks.Chart) and c.key == "gpu-operator") - driver = next(c for c in got if isinstance(c, stacks.Manifests) and c.key == "nvlink-disabled-driver-h100-pool") - assert op.values is not None - spec = driver.manifests[0]["spec"] - assert spec["version"] == op.values["driver"]["version"] - assert spec["useOpenKernelModules"] == op.values["driver"]["useOpenKernelModules"] - - -def test_configmap_carries_the_module_option() -> None: - """with_nvlink_disabled adds the kernel module ConfigMap, in its own Namespace.""" - got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["h100-pool"]) - config = next(c for c in got if isinstance(c, stacks.Manifests) and c.key == "nvlink-disable-config") - kinds = [doc["kind"] for doc in config.manifests] - assert kinds == ["Namespace", "ConfigMap"] - assert config.manifests[1]["data"] == {"nvidia.conf": "options nvidia NVreg_NvLinkDisable=1"} - - -def test_dra_driver_gates_on_pool_drivers() -> None: - """The DRA driver depends on every per-pool NVIDIADriver.""" - got = civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["pool-a", "pool-b"]) - dra = next(c for c in got if isinstance(c, stacks.Chart) and c.key == "nvidia-dra-driver-gpu") - assert dra.depends_on == ["gpu-operator", "nvlink-disabled-driver-pool-a", "nvlink-disabled-driver-pool-b"] - - -def test_join_is_not_mutated() -> None: + assert wildcards == [], "keyless (wildcard) tolerations, by component" + + +# The Literal types reject these at type-checking time; these exercise the +# runtime guard behind them, which catches the API and the stacks package +# disagreeing on a value. +JOIN_FAILS_CLOSED_CASES = [ + JoinFailsClosedCase( + name="UnknownCloud", + reason="A cloud the stacks package doesn't know fails the join.", + cloud="Mars", + stack="Standard", + want="unknown cloud 'Mars'", + ), + JoinFailsClosedCase( + name="UnknownStack", + reason="A stack the stacks package doesn't know fails the join.", + cloud="Nebius", + stack="Turbo", + want="unknown stack 'Turbo'", + ), +] + + +@pytest.mark.parametrize("case", JOIN_FAILS_CLOSED_CASES, ids=lambda case: case.name) +def test_join_fails_closed(case: JoinFailsClosedCase) -> None: + """join rejects a cloud or stack it doesn't know.""" + with pytest.raises(ValueError, match=case.want): + stacks.join(case.cloud, case.stack) # ty: ignore[invalid-argument-type] # cases pass values outside the Literals + + +# Each case's input is the real joined Civo stack rather than a literal one. +# The transform finds the gpu-operator and DRA driver charts by key, and where +# a key matches nothing it rewrites nothing, without complaint. Only the real +# stack shows the keys still match the charts Civo composes. The cost is that +# both wants restate those two charts as Civo pins them, so bumping either +# chart breaks both cases. +WITH_NVLINK_DISABLED_CASES = [ + WithNvLinkDisabledCase( + name="OnePool", + reason=( + "Disabling NVLink on one pool switches the gpu-operator chart to NVIDIADriver-CRD mode, adds the kernel " + "module ConfigMap and that pool's NVIDIADriver, and gates the DRA driver on it." + ), + components=stacks.join("Civo", "Standard"), + pools=["h100-pool"], + want=[ + stacks.Chart( + key="gpu-operator", + release="mp-gpu-operator", + namespace="gpu-operator", + chart="gpu-operator", + repository="https://helm.ngc.nvidia.com/nvidia", + version="v26.3.3", + wait=True, + depends_on=["node-feature-discovery", "cert-manager"], + values={ + "ccManager": {"enabled": False}, + "cdi": {"default": True, "enabled": True}, + "daemonsets": { + "tolerations": [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}] + }, + "dcgm": {"enabled": False}, + "dcgmExporter": {"enabled": False}, + "devicePlugin": {"enabled": False}, + "driver": { + "enabled": True, + "maxParallelUpgrades": 5, + "rdma": {"enabled": False}, + "useOpenKernelModules": True, + "version": "580.173.02", + # deployDefaultCR keeps the chart's default NVIDIADriver + # driving the pools the transform doesn't name, so + # flipping modes changes nothing for them. + "nvidiaDriverCRD": {"enabled": True, "deployDefaultCR": True}, + }, + "fullnameOverride": "gpu-operator", + "gdrcopy": {"enabled": False}, + "gfd": {"enabled": True}, + "kataSandboxDevicePlugin": {"enabled": False}, + "migManager": {"enabled": False}, + "nfd": {"enabled": False}, + "operator": { + "resources": { + "limits": {"cpu": "500m", "memory": "700Mi"}, + "requests": {"cpu": "200m", "memory": "300Mi"}, + }, + "tolerations": [], + "upgradeCRD": True, + }, + "toolkit": {"enabled": False}, + "validator": {"plugin": {"env": [{"name": "WITH_WORKLOAD", "value": "false"}]}}, + }, + ), + stacks.Chart( + key="nvidia-dra-driver-gpu", + release="mp-dra-driver-nvidia-gpu", + namespace="nvidia-dra-driver", + chart="dra-driver-nvidia-gpu", + repository="oci://registry.k8s.io/dra-driver-nvidia/charts", + version="0.4.1", + depends_on=["gpu-operator", "nvlink-disabled-driver-h100-pool"], + values={ + "gpuResourcesEnabledOverride": True, + "nvidiaDriverRoot": "/run/nvidia/driver", + "resources": {"computeDomains": {"enabled": False}}, + }, + ), + stacks.Manifests( + key="nvlink-disable-config", + manifests=[ + {"apiVersion": "v1", "kind": "Namespace", "metadata": {"name": "gpu-operator"}}, + { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "nvidia-kernel-config", "namespace": "gpu-operator"}, + "data": {"nvidia.conf": "options nvidia NVreg_NvLinkDisable=1"}, + }, + ], + ), + stacks.Manifests( + key="nvlink-disabled-driver-h100-pool", + manifests=[ + { + "apiVersion": "nvidia.com/v1alpha1", + "kind": "NVIDIADriver", + "metadata": {"name": "nvlink-disabled-h100-pool"}, + "spec": { + "driverType": "gpu", + # The chart's driver pin and module flavor: one + # review moves both. + "version": "580.173.02", + "useOpenKernelModules": True, + "nodeSelector": {"modelplane.ai/pool": "h100-pool"}, + "kernelModuleConfig": {"name": "nvidia-kernel-config"}, + "tolerations": [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}], + }, + } + ], + depends_on=["gpu-operator", "nvlink-disable-config"], + # The operator-populated state, so the DRA driver's install + # gate orders on driver health. + ready='object.status.state == "ready"', + ), + ], + ), + WithNvLinkDisabledCase( + name="TwoPools", + reason="Disabling NVLink on two pools adds an NVIDIADriver for each, and gates the DRA driver on both.", + components=stacks.join("Civo", "Standard"), + pools=["pool-a", "pool-b"], + want=[ + stacks.Chart( + key="gpu-operator", + release="mp-gpu-operator", + namespace="gpu-operator", + chart="gpu-operator", + repository="https://helm.ngc.nvidia.com/nvidia", + version="v26.3.3", + wait=True, + depends_on=["node-feature-discovery", "cert-manager"], + values={ + "ccManager": {"enabled": False}, + "cdi": {"default": True, "enabled": True}, + "daemonsets": { + "tolerations": [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}] + }, + "dcgm": {"enabled": False}, + "dcgmExporter": {"enabled": False}, + "devicePlugin": {"enabled": False}, + "driver": { + "enabled": True, + "maxParallelUpgrades": 5, + "rdma": {"enabled": False}, + "useOpenKernelModules": True, + "version": "580.173.02", + "nvidiaDriverCRD": {"enabled": True, "deployDefaultCR": True}, + }, + "fullnameOverride": "gpu-operator", + "gdrcopy": {"enabled": False}, + "gfd": {"enabled": True}, + "kataSandboxDevicePlugin": {"enabled": False}, + "migManager": {"enabled": False}, + "nfd": {"enabled": False}, + "operator": { + "resources": { + "limits": {"cpu": "500m", "memory": "700Mi"}, + "requests": {"cpu": "200m", "memory": "300Mi"}, + }, + "tolerations": [], + "upgradeCRD": True, + }, + "toolkit": {"enabled": False}, + "validator": {"plugin": {"env": [{"name": "WITH_WORKLOAD", "value": "false"}]}}, + }, + ), + stacks.Chart( + key="nvidia-dra-driver-gpu", + release="mp-dra-driver-nvidia-gpu", + namespace="nvidia-dra-driver", + chart="dra-driver-nvidia-gpu", + repository="oci://registry.k8s.io/dra-driver-nvidia/charts", + version="0.4.1", + depends_on=["gpu-operator", "nvlink-disabled-driver-pool-a", "nvlink-disabled-driver-pool-b"], + values={ + "gpuResourcesEnabledOverride": True, + "nvidiaDriverRoot": "/run/nvidia/driver", + "resources": {"computeDomains": {"enabled": False}}, + }, + ), + stacks.Manifests( + key="nvlink-disable-config", + manifests=[ + {"apiVersion": "v1", "kind": "Namespace", "metadata": {"name": "gpu-operator"}}, + { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "nvidia-kernel-config", "namespace": "gpu-operator"}, + "data": {"nvidia.conf": "options nvidia NVreg_NvLinkDisable=1"}, + }, + ], + ), + stacks.Manifests( + key="nvlink-disabled-driver-pool-a", + manifests=[ + { + "apiVersion": "nvidia.com/v1alpha1", + "kind": "NVIDIADriver", + "metadata": {"name": "nvlink-disabled-pool-a"}, + "spec": { + "driverType": "gpu", + "version": "580.173.02", + "useOpenKernelModules": True, + "nodeSelector": {"modelplane.ai/pool": "pool-a"}, + "kernelModuleConfig": {"name": "nvidia-kernel-config"}, + "tolerations": [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}], + }, + } + ], + depends_on=["gpu-operator", "nvlink-disable-config"], + ready='object.status.state == "ready"', + ), + stacks.Manifests( + key="nvlink-disabled-driver-pool-b", + manifests=[ + { + "apiVersion": "nvidia.com/v1alpha1", + "kind": "NVIDIADriver", + "metadata": {"name": "nvlink-disabled-pool-b"}, + "spec": { + "driverType": "gpu", + "version": "580.173.02", + "useOpenKernelModules": True, + "nodeSelector": {"modelplane.ai/pool": "pool-b"}, + "kernelModuleConfig": {"name": "nvidia-kernel-config"}, + "tolerations": [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}], + }, + } + ], + depends_on=["gpu-operator", "nvlink-disable-config"], + ready='object.status.state == "ready"', + ), + ], + ), +] + + +@pytest.mark.parametrize("case", WITH_NVLINK_DISABLED_CASES, ids=lambda case: case.name) +def test_with_nvlink_disabled(case: WithNvLinkDisabledCase) -> None: + """with_nvlink_disabled rewrites a joined stack to disable NVLink on the named pools.""" + got = civo.with_nvlink_disabled(case.components, case.pools) + # Only what the transform adds or rewrites, in order. The rest of the list + # is the joined stack passed through, vendored CRDs and all, and restating + # it would bury the components each case is about. A component the + # transform dropped wouldn't show up here. + changed = [c for c in got if c not in case.components] + assert changed == case.want, case.reason + + +def test_join_not_mutated() -> None: """with_nvlink_disabled leaves the joined stack it was given unchanged.""" # The transform must copy: the joined lists share the module-level # component objects, and mutating them would leak NVLink disable - # into every later request. + # into every later request. This checks the two fields the transform + # rewrites against their stock values rather than comparing the stack + # before and after, because a before-and-after comparison would miss an + # edit an earlier test's call had already made. civo.with_nvlink_disabled(stacks.join("Civo", "Standard"), ["h100-pool"]) joined = stacks.join("Civo", "Standard") op = next(c for c in joined if isinstance(c, stacks.Chart) and c.key == "gpu-operator") diff --git a/functions/compose-telemetry-destination/tests/test_fn.py b/functions/compose-telemetry-destination/tests/test_fn.py index 14d1d4cc6..ae2b2fa23 100644 --- a/functions/compose-telemetry-destination/tests/test_fn.py +++ b/functions/compose-telemetry-destination/tests/test_fn.py @@ -17,6 +17,7 @@ import asyncio import dataclasses import json +from typing import Any import pytest from crossplane.function import resource @@ -25,6 +26,8 @@ from google.protobuf import duration_pb2 as durationpb from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb +from models.ai.modelplane.telemetrydestination import v1alpha1 +from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @dataclasses.dataclass @@ -32,215 +35,340 @@ class Case: """A test case for compose-telemetry-destination.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse +def _telemetry_destination(*, sinks: list[v1alpha1.Sink], extensions: dict[str, Any] | None) -> fnv1.Resource: + """The TelemetryDestination XR named default, exporting through sinks, with extensions unless they're None.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.TelemetryDestination( + metadata=metav1.ObjectMeta(name="default"), + spec=v1alpha1.Spec(sinks=sinks, extensions=extensions), + ).model_dump(exclude_none=True, mode="json", by_alias=True) + ), + ) + + +def _desired_telemetry_destination(*, status: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The desired TelemetryDestination XR, carrying status, or only its readiness if status is None.""" + if status is None: + return fnv1.Resource(ready=ready) + return fnv1.Resource(resource=resource.dict_to_struct({"status": status}), ready=ready) + + def _to_dict(msg: message.Message) -> dict: """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" return json.loads(json_format.MessageToJson(msg, sort_keys=True)) -def _compose_cases() -> list[Case]: - def sink(name: str = "primary", type_: str = "otlphttp", secret: str | None = None) -> dict: - """A sink wiring its own authenticator, which is the case worth validating.""" - return { - "name": name, - "type": type_, - "endpoint": "https://otel.acme.example", - "config": {"auth": {"authenticator": "oauth2client/acme"}}, - **({"secretRef": {"name": secret}} if secret else {}), - } - - sinks = [sink()] - extensions = {"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}} - - def xr(spec: dict) -> dict: - return { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "TelemetryDestination", - "metadata": {"name": "default"}, - "spec": spec, - } - - def req(spec: dict, secrets: list | None = None) -> fnv1.RunFunctionRequest: - r = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(xr(spec)))), - ) - if secrets is not None: - r.required_resources["secret-primary"].items.extend([fnv1.Resource(resource=s) for s in secrets]) - return r - - def want( - ready: fnv1.Ready, status: dict | None, cond: fnv1.Condition, secret: str | None = None - ) -> fnv1.RunFunctionResponse: - composite = fnv1.Resource(ready=ready) - if status is not None: - composite.resource.CopyFrom(resource.dict_to_struct(status)) - rsp = fnv1.RunFunctionResponse( +# A sink's credential requirement names modelplane-system: unqualified, it would +# resolve a Secret of that name in any namespace, and accept the wrong +# credential. +COMPOSE_CASES = [ + Case( + name="AuthenticatorDefined", + reason=( + "With its sink's authenticator defined under extensions, the destination is Accepted and Ready, " + "naming the sink it exports through." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_telemetry_destination( + sinks=[ + v1alpha1.Sink( + name="primary", + type="otlphttp", + endpoint="https://otel.acme.example", + config={"auth": {"authenticator": "oauth2client/acme"}}, + ), + ], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + ), + ), + ), + want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=composite), - conditions=[cond], + desired=fnv1.State(composite=_desired_telemetry_destination(status={}, ready=fnv1.READY_TRUE)), context=structpb.Struct(), - ) - if secret is not None: - rsp.requirements.resources["secret-primary"].api_version = "v1" - rsp.requirements.resources["secret-primary"].kind = "Secret" - rsp.requirements.resources["secret-primary"].match_name = secret - # Qualified: unqualified it would resolve a Secret of that - # name in any namespace, and accept the wrong credential. - rsp.requirements.resources["secret-primary"].namespace = "modelplane-system" - return rsp - - return [ - Case( - name="ready, naming the sinks it sends through", - req=req({"sinks": sinks, "extensions": extensions}), - want=want( - fnv1.READY_TRUE, - {"status": {}}, + conditions=[ fnv1.Condition( type="Accepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Available", message="Exporting through otlphttp/primary", ), - ), + ], ), - Case( - name="ready with an exporter that references no authenticator at all", - req=req( - { - "sinks": [ - { - "name": "prom", - "type": "prometheusremotewrite", - "endpoint": "https://prom.acme.example/api/v1/write", - } - ] - } + ), + Case( + name="NoAuthenticator", + reason="A sink that references no authenticator needs no extensions, so the destination is Ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_telemetry_destination( + sinks=[ + v1alpha1.Sink( + name="prom", + type="prometheusremotewrite", + endpoint="https://prom.acme.example/api/v1/write", + ), + ], + extensions=None, + ), ), - want=want( - fnv1.READY_TRUE, - {"status": {}}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_telemetry_destination(status={}, ready=fnv1.READY_TRUE)), + context=structpb.Struct(), + conditions=[ fnv1.Condition( type="Accepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Available", message="Exporting through prometheusremotewrite/prom", ), + ], + ), + ), + Case( + name="BearerTokenAuth", + reason=( + "With a sink that sets auth and a secretRef, and names no authenticator in its config, " + "the destination is Ready once its credential Secret is present." + ), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_telemetry_destination( + sinks=[ + v1alpha1.Sink( + name="primary", + type="otlphttp", + endpoint="https://otel.acme.example", + secretRef=v1alpha1.SecretRef(name="telemetry-credentials"), + auth=v1alpha1.Auth(bearerTokenKey="token"), + ), + ], + extensions=None, + ), ), + required_resources={ + "secret-primary": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + {"apiVersion": "v1", "kind": "Secret", "metadata": {"name": "telemetry-credentials"}} + ), + ), + ], + ), + }, ), - Case( - name="ready with no extensions, because Modelplane composes the authenticator", - req=req( - { - "sinks": [ - { - "name": "primary", - "type": "otlphttp", - "endpoint": "https://otel.acme.example", - "secretRef": {"name": "telemetry-credentials"}, - "auth": {"bearerTokenKey": "token"}, - } - ] + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_telemetry_destination(status={}, ready=fnv1.READY_TRUE)), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "secret-primary": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="telemetry-credentials", + namespace="modelplane-system", + ), }, - secrets=[ - resource.dict_to_struct( - {"apiVersion": "v1", "kind": "Secret", "metadata": {"name": "telemetry-credentials"}} - ) - ], ), - want=want( - fnv1.READY_TRUE, - {"status": {}}, + conditions=[ fnv1.Condition( type="Accepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Available", message="Exporting through otlphttp/primary", ), - secret="telemetry-credentials", + ], + ), + ), + Case( + name="SecretExists", + reason="Once its sink's credential Secret exists, the destination is Accepted and Ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_telemetry_destination( + sinks=[ + v1alpha1.Sink( + name="primary", + type="otlphttp", + endpoint="https://otel.acme.example", + config={"auth": {"authenticator": "oauth2client/acme"}}, + secretRef=v1alpha1.SecretRef(name="telemetry-credentials"), + ), + ], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + ), ), + required_resources={ + "secret-primary": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + {"apiVersion": "v1", "kind": "Secret", "metadata": {"name": "telemetry-credentials"}} + ), + ), + ], + ), + }, ), - Case( - name="ready once the credential Secret exists", - req=req( - {"sinks": [sink(secret="telemetry-credentials")], "extensions": extensions}, - secrets=[ - resource.dict_to_struct( - {"apiVersion": "v1", "kind": "Secret", "metadata": {"name": "telemetry-credentials"}} - ) - ], + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_telemetry_destination(status={}, ready=fnv1.READY_TRUE)), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "secret-primary": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="telemetry-credentials", + namespace="modelplane-system", + ), + }, ), - want=want( - fnv1.READY_TRUE, - {"status": {}}, + conditions=[ fnv1.Condition( type="Accepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Available", message="Exporting through otlphttp/primary", ), - secret="telemetry-credentials", + ], + ), + ), + Case( + name="SecretUnresolved", + reason="Until its sink's credential Secret requirement resolves, the destination waits for it and isn't Ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_telemetry_destination( + sinks=[ + v1alpha1.Sink( + name="primary", + type="otlphttp", + endpoint="https://otel.acme.example", + config={"auth": {"authenticator": "oauth2client/acme"}}, + secretRef=v1alpha1.SecretRef(name="telemetry-credentials"), + ), + ], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + ), ), ), - Case( - name="waits for the credential Secret to resolve", - req=req( - {"sinks": [sink(secret="telemetry-credentials")], "extensions": extensions}, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_telemetry_destination(status=None, ready=fnv1.READY_FALSE)), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "secret-primary": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="telemetry-credentials", + namespace="modelplane-system", + ), + }, ), - want=want( - fnv1.READY_FALSE, - None, + conditions=[ fnv1.Condition( type="Accepted", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForSecret", message="Waiting for the credential Secret to resolve", ), - secret="telemetry-credentials", + ], + ), + ), + Case( + name="UnknownAuthenticator", + reason="A sink naming an authenticator no extension defines leaves the destination not Ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_telemetry_destination( + sinks=[ + v1alpha1.Sink( + name="primary", + type="otlphttp", + endpoint="https://otel.acme.example", + config={"auth": {"authenticator": "oauth2client/acme"}}, + ), + ], + extensions=None, + ), ), ), - Case( - name="not ready when a sink names an authenticator nothing defines", - req=req({"sinks": sinks}), - want=want( - fnv1.READY_FALSE, - None, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_telemetry_destination(status=None, ready=fnv1.READY_FALSE)), + context=structpb.Struct(), + conditions=[ fnv1.Condition( type="Accepted", status=fnv1.STATUS_CONDITION_FALSE, reason="UnknownAuthenticator", message="No extension defines oauth2client/acme, so the collector would refuse to start", ), + ], + ), + ), + Case( + name="SecretMissing", + reason="When its sink's credential Secret doesn't exist, the destination isn't Ready and names the Secret.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_telemetry_destination( + sinks=[ + v1alpha1.Sink( + name="primary", + type="otlphttp", + endpoint="https://otel.acme.example", + config={"auth": {"authenticator": "oauth2client/acme"}}, + secretRef=v1alpha1.SecretRef(name="telemetry-credentials"), + ), + ], + extensions={"oauth2client/acme": {"token_url": "https://issuer.acme.example/token"}}, + ), ), + required_resources={"secret-primary": fnv1.Resources()}, ), - Case( - name="not ready when the credential Secret is missing", - req=req( - {"sinks": [sink(secret="telemetry-credentials")], "extensions": extensions}, - secrets=[], + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_telemetry_destination(status=None, ready=fnv1.READY_FALSE)), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "secret-primary": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="telemetry-credentials", + namespace="modelplane-system", + ), + }, ), - want=want( - fnv1.READY_FALSE, - None, + conditions=[ fnv1.Condition( type="Accepted", status=fnv1.STATUS_CONDITION_FALSE, reason="SecretNotFound", - message=( - "Secret telemetry-credentials does not exist, so sink primary has no credential to send with" - ), + message="Secret telemetry-credentials does not exist, so sink primary has no credential to send with", ), - secret="telemetry-credentials", - ), + ], ), - ] + ), +] -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: - """The function reports whether a destination can actually be sent through.""" + """RunFunction reports whether the destination can be sent through.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-usages/tests/test_fn.py b/functions/compose-usages/tests/test_fn.py index 912f59a70..75c5c7c69 100644 --- a/functions/compose-usages/tests/test_fn.py +++ b/functions/compose-usages/tests/test_fn.py @@ -26,166 +26,283 @@ from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb -_NAMESPACE = "test-ns" -_PC = "test-cluster" - -_RELEASE = { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "Release", - "metadata": {"namespace": _NAMESPACE}, - "spec": { - "providerConfigRef": {"kind": "ProviderConfig", "name": _PC}, - "forProvider": {"chart": {"name": "cert-manager"}}, - }, -} - -_OBJECT = { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": _NAMESPACE}, - "spec": { - "providerConfigRef": {"kind": "ProviderConfig", "name": _PC}, - "forProvider": {"manifest": {"apiVersion": "v1", "kind": "Namespace"}}, - }, -} - -# Not a consumer kind: a ProviderConfig gets no Usage of its own. -_PROVIDER_CONFIG = { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "ProviderConfig", - "metadata": {"name": _PC, "namespace": _NAMESPACE}, - "spec": {}, -} - -# A consumer kind (Object) that references no ProviderConfig: gets no Usage. -_OBJECT_NO_PC = { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": _NAMESPACE}, - "spec": {"forProvider": {"manifest": {"apiVersion": "v1", "kind": "ConfigMap"}}}, -} - -# A Release that already carries a label, to check relabeling preserves it. -_RELEASE_WITH_LABEL = { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "Release", - "metadata": {"namespace": _NAMESPACE, "labels": {"existing": "keep"}}, - "spec": { - "providerConfigRef": {"kind": "ProviderConfig", "name": _PC}, - "forProvider": {"chart": {"name": "prometheus"}}, - }, -} - - -def _labelled(d: dict, consumer: str) -> dict: - """A copy of d with the usage-consumer label stamped on it.""" - out = {**d, "metadata": {**d.get("metadata", {})}} - out["metadata"]["labels"] = { - **d.get("metadata", {}).get("labels", {}), - "modelplane.ai/usage-consumer": consumer, - } - return out - - -def _usage(api_version: str, kind: str, consumer: str) -> dict: - return { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": _NAMESPACE}, - "spec": { - "of": { - "apiVersion": api_version, - "kind": "ProviderConfig", - "resourceRef": {"name": _PC}, - }, - "by": { - "apiVersion": api_version, - "kind": kind, - "resourceSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/usage-consumer": consumer}, - }, - }, - "replayDeletion": True, - }, - } + +@dataclasses.dataclass +class Case: + """A test case for compose-usages.""" + + name: str + reason: str + req: fnv1.RunFunctionRequest + want: fnv1.RunFunctionResponse -def _composite(namespace: str | None = _NAMESPACE) -> structpb.Struct: +# compose-usages works on any composite and reads only the observed one's +# namespace, so both composites are bare. A ServingStack built from its model +# would need spec.cloud, spec.gateway and spec.secrets, none of which the +# function reads. +def _serving_stack(*, namespace: str | None) -> fnv1.Resource: + """The bare observed ServingStack composite, in namespace unless it's None.""" metadata = {"name": "test"} if namespace is not None: metadata["namespace"] = namespace - return resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": metadata, - } + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": metadata, + } + ) ) -@dataclasses.dataclass -class Case: - """A test case for compose-usages.""" +def _desired_serving_stack(*, namespace: str | None) -> fnv1.Resource: + """The bare desired ServingStack composite, in namespace unless it's None.""" + metadata = {"name": "test"} + if namespace is not None: + metadata["namespace"] = namespace + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": metadata, + } + ) + ) - name: str - req: fnv1.RunFunctionRequest - want: fnv1.RunFunctionResponse + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) COMPOSE_CASES = [ Case( - name="labels each consumer and composes a Usage per ProviderConfig reference", + name="MixedResources", + reason="Each Release or Object that references a ProviderConfig gets a consumer label and a Usage of that ProviderConfig, and other resources are left alone.", req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=_composite())), + observed=fnv1.State( + composite=_serving_stack(namespace="test-ns"), + ), desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), + composite=_desired_serving_stack(namespace="test-ns"), resources={ - "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), - "gateway-namespace": fnv1.Resource(resource=resource.dict_to_struct(_OBJECT)), - "prometheus": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE_WITH_LABEL)), - "config-map": fnv1.Resource(resource=resource.dict_to_struct(_OBJECT_NO_PC)), - "provider-config-helm": fnv1.Resource(resource=resource.dict_to_struct(_PROVIDER_CONFIG)), + "cert-manager": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "test-ns"}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"chart": {"name": "cert-manager"}}, + }, + } + ) + ), + "gateway-namespace": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "test-ns"}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"manifest": {"apiVersion": "v1", "kind": "Namespace"}}, + }, + } + ) + ), + # A Release that already carries a label, which the + # function's label must not replace. + "prometheus": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "test-ns", "labels": {"existing": "keep"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"chart": {"name": "prometheus"}}, + }, + } + ) + ), + # A consumer kind that references no ProviderConfig. + "config-map": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "test-ns"}, + "spec": {"forProvider": {"manifest": {"apiVersion": "v1", "kind": "ConfigMap"}}}, + } + ) + ), + # Not a consumer kind. + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster", "namespace": "test-ns"}, + "spec": {}, + } + ) + ), }, ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), + composite=_desired_serving_stack(namespace="test-ns"), resources={ "cert-manager": fnv1.Resource( - resource=resource.dict_to_struct(_labelled(_RELEASE, "cert-manager")), + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "namespace": "test-ns", + "labels": {"modelplane.ai/usage-consumer": "cert-manager"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"chart": {"name": "cert-manager"}}, + }, + } + ) ), "gateway-namespace": fnv1.Resource( - resource=resource.dict_to_struct(_labelled(_OBJECT, "gateway-namespace")), + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "namespace": "test-ns", + "labels": {"modelplane.ai/usage-consumer": "gateway-namespace"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"manifest": {"apiVersion": "v1", "kind": "Namespace"}}, + }, + } + ) ), - # Existing labels are preserved when the consumer label is stamped. "prometheus": fnv1.Resource( - resource=resource.dict_to_struct(_labelled(_RELEASE_WITH_LABEL, "prometheus")), + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "namespace": "test-ns", + "labels": {"existing": "keep", "modelplane.ai/usage-consumer": "prometheus"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"chart": {"name": "prometheus"}}, + }, + } + ) ), - # An Object with no providerConfigRef is left untouched, no Usage. "config-map": fnv1.Resource( - resource=resource.dict_to_struct(_OBJECT_NO_PC), + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "test-ns"}, + "spec": {"forProvider": {"manifest": {"apiVersion": "v1", "kind": "ConfigMap"}}}, + } + ) ), "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_PROVIDER_CONFIG), + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster", "namespace": "test-ns"}, + "spec": {}, + } + ) ), "usage-pc-cert-manager": fnv1.Resource( resource=resource.dict_to_struct( - _usage("helm.m.crossplane.io/v1beta1", "Release", "cert-manager") + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "test-ns"}, + "spec": { + "of": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "resourceRef": {"name": "test-cluster"}, + }, + "by": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/usage-consumer": "cert-manager"}, + }, + }, + "replayDeletion": True, + }, + } ), ready=fnv1.READY_TRUE, ), "usage-pc-gateway-namespace": fnv1.Resource( resource=resource.dict_to_struct( - _usage("kubernetes.m.crossplane.io/v1alpha1", "Object", "gateway-namespace") + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "test-ns"}, + "spec": { + "of": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ProviderConfig", + "resourceRef": {"name": "test-cluster"}, + }, + "by": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/usage-consumer": "gateway-namespace"}, + }, + }, + "replayDeletion": True, + }, + } ), ready=fnv1.READY_TRUE, ), "usage-pc-prometheus": fnv1.Resource( resource=resource.dict_to_struct( - _usage("helm.m.crossplane.io/v1beta1", "Release", "prometheus") + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "test-ns"}, + "spec": { + "of": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "resourceRef": {"name": "test-cluster"}, + }, + "by": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/usage-consumer": "prometheus"}, + }, + }, + "replayDeletion": True, + }, + } ), ready=fnv1.READY_TRUE, ), @@ -195,45 +312,91 @@ class Case: ), ), Case( - name="no Usages when the composite has no namespace", + name="NoNamespace", + reason="A composite with no namespace gets no Usages, and its consumers get no label.", req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=_composite(namespace=None))), + observed=fnv1.State( + composite=_serving_stack(namespace=None), + ), desired=fnv1.State( - composite=fnv1.Resource(resource=_composite(namespace=None)), + composite=_desired_serving_stack(namespace=None), resources={ - "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), + "cert-manager": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "test-ns"}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"chart": {"name": "cert-manager"}}, + }, + } + ) + ), }, ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=_composite(namespace=None)), + composite=_desired_serving_stack(namespace=None), resources={ - "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), + "cert-manager": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "test-ns"}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"chart": {"name": "cert-manager"}}, + }, + } + ) + ), }, ), context=structpb.Struct(), ), ), Case( - name="no consumers means no Usages", + name="NoConsumers", + reason="A composite whose only desired resource is a ProviderConfig gets no Usages.", req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=_composite())), + observed=fnv1.State( + composite=_serving_stack(namespace="test-ns"), + ), desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), + composite=_desired_serving_stack(namespace="test-ns"), resources={ - "provider-config-helm": fnv1.Resource(resource=resource.dict_to_struct(_PROVIDER_CONFIG)), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster", "namespace": "test-ns"}, + "spec": {}, + } + ) + ), }, ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), + composite=_desired_serving_stack(namespace="test-ns"), resources={ "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_PROVIDER_CONFIG), + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster", "namespace": "test-ns"}, + "spec": {}, + } + ) ), }, ), @@ -243,13 +406,8 @@ class Case: ] -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - @pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction labels consumers and composes their Usages.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) + assert _to_dict(got) == _to_dict(case.want), case.reason diff --git a/functions/compose-vultr-cluster/tests/test_fn.py b/functions/compose-vultr-cluster/tests/test_fn.py index 608ed76f7..c6021b0bf 100644 --- a/functions/compose-vultr-cluster/tests/test_fn.py +++ b/functions/compose-vultr-cluster/tests/test_fn.py @@ -35,494 +35,647 @@ class Case: """A test case for compose-vultr-cluster.""" name: str + reason: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse -# Name of the cluster's connection secret. Derived like the function derives -# it - the hash suffix depends only on the parent and child names. -_KUBECONFIG_SECRET_NAME = resource.child_name("test-cluster", "kubeconfig") - -# The system node pool injected inline into every cluster. -_SYSTEM_POOL = { - "label": "system", - "plan": "vc2-6c-16gb", - "nodeQuantity": 1, - "autoScaler": True, - "minNodes": 1, - "maxNodes": 2, - "labels": [{"key": "modelplane.ai/pool", "value": "system"}], -} - -# The taint every GPU pool carries. -_GPU_TAINTS = [ - {"key": "nvidia.com/gpu", "value": "true", "effect": "NoSchedule"}, -] - - -def _xr(pools: list[v1alpha1.NodePool]) -> dict: - """A VultrCluster XR with the given node pools, as a request dict.""" - return v1alpha1.VultrCluster( +def _xr(*, node_pools: list[v1alpha1.NodePool]) -> fnv1.Resource: + """The observed VultrCluster XR, with the given node pools.""" + xr = v1alpha1.VultrCluster( metadata=metav1.ObjectMeta( name="test-cluster", namespace="modelplane-system", ), spec=v1alpha1.Spec( region="ewr", - nodePools=pools, + nodePools=node_pools, ), - ).model_dump(exclude_none=True, mode="json") - - -def _req( - pools: list[v1alpha1.NodePool], - observed_resources: dict[str, fnv1.Resource] | None = None, -) -> fnv1.RunFunctionRequest: - return fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(pools))), - resources=observed_resources or {}, + ) + return fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json", by_alias=True))) + + +def _desired_xr() -> fnv1.Resource: + """The desired XR, publishing the cluster's kubeconfig Secret.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + ], + }, + } ), ) -def _cluster( - cred_kind: str = "ClusterProviderConfig", - cred_name: str = "default", -) -> dict: - """A Kubernetes cluster golden with only the system pool.""" - return { - "apiVersion": "vke.vultr.m.upbound.io/v1beta1", - "kind": "Kubernetes", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "label": "test-cluster", - "region": "ewr", - "version": "v1.36.2+1", - "haControlplanes": True, - "nodePools": _SYSTEM_POOL, - }, - "writeConnectionSecretToRef": {"name": _KUBECONFIG_SECRET_NAME}, - }, - } - - -def _provider_config() -> dict: - """A provider-kubernetes ProviderConfig golden pointing at the kubeconfig.""" - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ProviderConfig", - "metadata": { - "name": _KUBECONFIG_SECRET_NAME, - "namespace": "modelplane-system", - }, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": _KUBECONFIG_SECRET_NAME, - "key": "kubeconfig", +def _cluster(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed VKE cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vke.vultr.m.upbound.io/v1beta1", + "kind": "Kubernetes", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "label": "test-cluster", + "region": "ewr", + "version": "v1.36.2+1", + "haControlplanes": True, + # The system node pool the function adds inline to every cluster. + "nodePools": { + "label": "system", + "plan": "vc2-6c-16gb", + "nodeQuantity": 1, + "autoScaler": True, + "minNodes": 1, + "maxNodes": 2, + "labels": [{"key": "modelplane.ai/pool", "value": "system"}], + }, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, }, - }, - }, - } + } + ), + ready=ready, + ) -def _gpu_observer() -> dict: - """A provider-kubernetes Object golden that observes the GPU validator DS.""" - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe"], - "providerConfigRef": { - "kind": "ProviderConfig", - "name": _KUBECONFIG_SECRET_NAME, - }, - "readiness": { - "policy": "DeriveFromCelQuery", - "celQuery": ( - "has(object.status.numberReady)" - " && object.status.desiredNumberScheduled >= 1" - " && object.status.numberReady == object.status.desiredNumberScheduled" - ), - }, - "forProvider": { - "manifest": { - "apiVersion": "apps/v1", - "kind": "DaemonSet", - "metadata": { - "name": "nvidia-operator-validator", - "namespace": "gpu-operator", +def _observed_cluster(*, ready: bool) -> fnv1.Resource: + """The VKE cluster as observed, with a Ready condition that's True if ready and False if not.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vke.vultr.m.upbound.io/v1beta1", + "kind": "Kubernetes", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "label": "test-cluster", + "region": "ewr", + "version": "v1.36.2+1", + "haControlplanes": True, + "nodePools": { + "label": "system", + "plan": "vc2-6c-16gb", + "nodeQuantity": 1, + "autoScaler": True, + "minNodes": 1, + "maxNodes": 2, + "labels": [{"key": "modelplane.ai/pool", "value": "system"}], + }, }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, }, - }, - }, - } + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True" if ready else "False", + "reason": "Available" if ready else "Unavailable", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ) -def _node_pool( - label: str, - plan: str, - node_quantity: int, - labels: list, - taints: list | None = None, - *, - auto_scaler: bool = False, - min_nodes: int | None = None, - max_nodes: int | None = None, - cred_kind: str = "ClusterProviderConfig", - cred_name: str = "default", -) -> dict: - """A KubernetesNodePool golden.""" - fp: dict[str, Any] = { - "label": label, - "plan": plan, +def _node_pool_gpu(*, node_quantity: int, autoscaling: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The composed KubernetesNodePool for the gpu-l40s pool, with autoscaling if given.""" + for_provider: dict[str, Any] = { + "label": "gpu-l40s", + "plan": "vcg-l40s-16c-180g-48vram", "nodeQuantity": node_quantity, - "labels": labels, + "labels": [ + {"key": "modelplane.ai/pool", "value": "gpu-l40s"}, + {"key": "modelplane.ai/gpu", "value": "nvidia-l40s"}, + {"key": "nvidia.com/gpu.deploy.device-plugin", "value": "false"}, + ], "clusterIdSelector": {"matchControllerRef": True}, + "taints": [{"key": "nvidia.com/gpu", "value": "true", "effect": "NoSchedule"}], } - if taints: - fp["taints"] = taints - if auto_scaler: - fp["autoScaler"] = True - fp["minNodes"] = min_nodes - fp["maxNodes"] = max_nodes - return { - "apiVersion": "vke.vultr.m.upbound.io/v1beta1", - "kind": "KubernetesNodePool", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": fp, - }, - } - - -def _status() -> dict: - return { - "status": { - "secrets": [ - { - "type": "Kubeconfig", - "name": _KUBECONFIG_SECRET_NAME, - "key": "kubeconfig", + if autoscaling is not None: + for_provider.update(autoscaling) + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vke.vultr.m.upbound.io/v1beta1", + "kind": "KubernetesNodePool", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": for_provider, }, - ], - }, - } + } + ), + ready=ready, + ) -def _observed_ready(desired: dict) -> fnv1.Resource: - """An observed variant of a desired resource with a Ready=True condition.""" - observed = { - **desired, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", +def _provider_config() -> fnv1.Resource: + """The composed provider-kubernetes ProviderConfig for the cluster, which is always ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ProviderConfig", + "metadata": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", }, - ], - }, - } - return fnv1.Resource(resource=resource.dict_to_struct(observed)) + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ) -def _observed_unready(desired: dict) -> fnv1.Resource: - """An observed variant of a desired resource with a Ready=False condition.""" - observed = { - **desired, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "False", - "reason": "Unavailable", - "lastTransitionTime": "2024-01-01T00:00:00Z", +def _gpu_observer(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Object observing the GPU validator DaemonSet.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status.numberReady)" + " && object.status.desiredNumberScheduled >= 1" + " && object.status.numberReady == object.status.desiredNumberScheduled" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "DaemonSet", + "metadata": { + "name": "nvidia-operator-validator", + "namespace": "gpu-operator", + }, + }, + }, }, - ], - }, - } - return fnv1.Resource(resource=resource.dict_to_struct(observed)) - + } + ), + ready=ready, + ) -_GPU_POOL = v1alpha1.NodePool( - name="gpu-l40s", - role="GPU", - plan="vcg-l40s-16c-180g-48vram", - maxNodeCount=4, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), -) -_GPU_POOL_GOLDEN = _node_pool( - label="gpu-l40s", - plan="vcg-l40s-16c-180g-48vram", - node_quantity=1, - labels=[ - {"key": "modelplane.ai/pool", "value": "gpu-l40s"}, - {"key": "modelplane.ai/gpu", "value": "nvidia-l40s"}, - {"key": "nvidia.com/gpu.deploy.device-plugin", "value": "false"}, - ], - taints=_GPU_TAINTS, - auto_scaler=True, - min_nodes=1, - max_nodes=4, -) +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) COMPOSE_CASES = [ Case( - name="cluster composed first; node pools withheld until cluster Ready", - req=_req([_GPU_POOL]), + name="FirstPass", + reason="With nothing observed, a VultrCluster composes only the cluster, holding back its node pools, ProviderConfig and GPU observer until it's Ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + plan="vcg-l40s-16c-180g-48vram", + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], + ), + ), + ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), + "cluster": _cluster(ready=fnv1.READY_UNSPECIFIED), }, ), context=structpb.Struct(), ), ), Case( - name="node pools and GPU observer composed once cluster is Ready; autoscaling from maxNodeCount", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, + name="ClusterReady", + reason="With the cluster observed Ready, a VultrCluster composes the node pool, ProviderConfig and GPU observer, autoscaling the pool from one node to maxNodeCount.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + plan="vcg-l40s-16c-180g-48vram", + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], + ), + resources={ + "cluster": _observed_cluster(ready=True), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), + "cluster": _cluster(ready=fnv1.READY_TRUE), + "node-pool-gpu-l40s": _node_pool_gpu( + node_quantity=1, + autoscaling={"autoScaler": True, "minNodes": 1, "maxNodes": 4}, + ready=fnv1.READY_UNSPECIFIED, ), + "provider-config-kubernetes": _provider_config(), + "gpu-observer": _gpu_observer(ready=fnv1.READY_UNSPECIFIED), }, ), context=structpb.Struct(), ), ), Case( - name="dependents kept when the cluster Ready condition transiently regresses", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_unready(_cluster()), - "provider-config-kubernetes": _observed_ready(_provider_config()), - }, + name="ClusterReadyRegressed", + reason="With the cluster's Ready condition regressed to False but its ProviderConfig observed, a VultrCluster keeps the node pool, ProviderConfig and GPU observer composed.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + plan="vcg-l40s-16c-180g-48vram", + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], + ), + resources={ + "cluster": _observed_cluster(ready=False), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ProviderConfig", + "metadata": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", + }, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), + "cluster": _cluster(ready=fnv1.READY_UNSPECIFIED), + "node-pool-gpu-l40s": _node_pool_gpu( + node_quantity=1, + autoscaling={"autoScaler": True, "minNodes": 1, "maxNodes": 4}, + ready=fnv1.READY_UNSPECIFIED, ), + "provider-config-kubernetes": _provider_config(), + "gpu-observer": _gpu_observer(ready=fnv1.READY_UNSPECIFIED), }, ), context=structpb.Struct(), ), ), Case( - name="observed node pool alone keeps dependents composed", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_unready(_cluster()), - "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN), - }, + name="OnlyNodePoolObserved", + reason="With the cluster not Ready and only the node pool observed, a VultrCluster keeps the node pool, ProviderConfig and GPU observer composed, and marks the Ready node pool ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + plan="vcg-l40s-16c-180g-48vram", + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], + ), + resources={ + "cluster": _observed_cluster(ready=False), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vke.vultr.m.upbound.io/v1beta1", + "kind": "KubernetesNodePool", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "label": "gpu-l40s", + "plan": "vcg-l40s-16c-180g-48vram", + "nodeQuantity": 1, + "labels": [ + {"key": "modelplane.ai/pool", "value": "gpu-l40s"}, + {"key": "modelplane.ai/gpu", "value": "nvidia-l40s"}, + {"key": "nvidia.com/gpu.deploy.device-plugin", "value": "false"}, + ], + "clusterIdSelector": {"matchControllerRef": True}, + "taints": [{"key": "nvidia.com/gpu", "value": "true", "effect": "NoSchedule"}], + "autoScaler": True, + "minNodes": 1, + "maxNodes": 4, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), + "cluster": _cluster(ready=fnv1.READY_UNSPECIFIED), + "node-pool-gpu-l40s": _node_pool_gpu( + node_quantity=1, + autoscaling={"autoScaler": True, "minNodes": 1, "maxNodes": 4}, ready=fnv1.READY_TRUE, ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), - ), + "provider-config-kubernetes": _provider_config(), + "gpu-observer": _gpu_observer(ready=fnv1.READY_UNSPECIFIED), }, ), context=structpb.Struct(), ), ), Case( - name="fixed-size GPU pool", - req=_req( - [ - v1alpha1.NodePool( - name="gpu-l40s", - role="GPU", - plan="vcg-l40s-16c-180g-48vram", - nodeCount=2, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + name="FixedSizePool", + reason="A VultrCluster with a GPU pool that sets nodeCount but no maxNodeCount composes a node pool of that size with no autoscaler.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + plan="vcg-l40s-16c-180g-48vram", + nodeCount=2, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], ), - ], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, + resources={ + "cluster": _observed_cluster(ready=True), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct( - _node_pool( - label="gpu-l40s", - plan="vcg-l40s-16c-180g-48vram", - node_quantity=2, - labels=[ - {"key": "modelplane.ai/pool", "value": "gpu-l40s"}, - {"key": "modelplane.ai/gpu", "value": "nvidia-l40s"}, - {"key": "nvidia.com/gpu.deploy.device-plugin", "value": "false"}, - ], - taints=_GPU_TAINTS, - ), - ), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), + "cluster": _cluster(ready=fnv1.READY_TRUE), + "node-pool-gpu-l40s": _node_pool_gpu( + node_quantity=2, + autoscaling=None, + ready=fnv1.READY_UNSPECIFIED, ), + "provider-config-kubernetes": _provider_config(), + "gpu-observer": _gpu_observer(ready=fnv1.READY_UNSPECIFIED), }, ), context=structpb.Struct(), ), ), + # The pool's nodeCount and minNodeCount are both 2, so this case can't tell + # which of them sets the autoscaler's floor. Case( - name="minNodeCount sets the autoscaler floor; System pool carries no taint", - req=_req( - [ - v1alpha1.NodePool( - name="workers", - role="System", - plan="vc2-6c-16gb", - nodeCount=2, - minNodeCount=2, - maxNodeCount=5, + name="SystemRolePool", + reason="A VultrCluster's System-role pool autoscales from a floor of 2 nodes to maxNodeCount, and gets neither GPU labels nor a taint.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + node_pools=[ + v1alpha1.NodePool( + name="workers", + role="System", + plan="vc2-6c-16gb", + nodeCount=2, + minNodeCount=2, + maxNodeCount=5, + ), + ], ), - ], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, + resources={ + "cluster": _observed_cluster(ready=True), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), + "cluster": _cluster(ready=fnv1.READY_TRUE), "node-pool-workers": fnv1.Resource( resource=resource.dict_to_struct( - _node_pool( - label="workers", - plan="vc2-6c-16gb", - node_quantity=2, - labels=[{"key": "modelplane.ai/pool", "value": "workers"}], - auto_scaler=True, - min_nodes=2, - max_nodes=5, - ), + { + "apiVersion": "vke.vultr.m.upbound.io/v1beta1", + "kind": "KubernetesNodePool", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "label": "workers", + "plan": "vc2-6c-16gb", + "nodeQuantity": 2, + "labels": [{"key": "modelplane.ai/pool", "value": "workers"}], + "clusterIdSelector": {"matchControllerRef": True}, + "autoScaler": True, + "minNodes": 2, + "maxNodes": 5, + }, + }, + } ), ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), - ), + "provider-config-kubernetes": _provider_config(), + "gpu-observer": _gpu_observer(ready=fnv1.READY_UNSPECIFIED), }, ), context=structpb.Struct(), ), ), Case( - name="VultrCluster Ready only once the gpu-observer is Ready", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_ready(_cluster()), - "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN), - "gpu-observer": _observed_ready(_gpu_observer()), - }, + name="GPUObserverReady", + reason="With the GPU observer observed Ready alongside the cluster and node pool, a VultrCluster marks every composed resource ready.", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + plan="vcg-l40s-16c-180g-48vram", + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], + ), + resources={ + "cluster": _observed_cluster(ready=True), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vke.vultr.m.upbound.io/v1beta1", + "kind": "KubernetesNodePool", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "label": "gpu-l40s", + "plan": "vcg-l40s-16c-180g-48vram", + "nodeQuantity": 1, + "labels": [ + {"key": "modelplane.ai/pool", "value": "gpu-l40s"}, + {"key": "modelplane.ai/gpu", "value": "nvidia-l40s"}, + {"key": "nvidia.com/gpu.deploy.device-plugin", "value": "false"}, + ], + "clusterIdSelector": {"matchControllerRef": True}, + "taints": [{"key": "nvidia.com/gpu", "value": "true", "effect": "NoSchedule"}], + "autoScaler": True, + "minNodes": 1, + "maxNodes": 4, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ), + "gpu-observer": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status.numberReady)" + " && object.status.desiredNumberScheduled >= 1" + " && object.status.numberReady == object.status.desiredNumberScheduled" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "DaemonSet", + "metadata": { + "name": "nvidia-operator-validator", + "namespace": "gpu-operator", + }, + }, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ready=fnv1.READY_TRUE, - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), + "cluster": _cluster(ready=fnv1.READY_TRUE), + "node-pool-gpu-l40s": _node_pool_gpu( + node_quantity=1, + autoscaling={"autoScaler": True, "minNodes": 1, "maxNodes": 4}, ready=fnv1.READY_TRUE, ), + "provider-config-kubernetes": _provider_config(), + "gpu-observer": _gpu_observer(ready=fnv1.READY_TRUE), }, ), context=structpb.Struct(), @@ -531,13 +684,8 @@ def _observed_unready(desired: dict) -> fnv1.Resource: ] -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - @pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction composes VKE cluster infrastructure.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) + assert _to_dict(got) == _to_dict(case.want), case.reason From b76a14c6f291eec723d34a25a7f7ec083a21b5b6 Mon Sep 17 00:00:00 2001 From: Nic Cope Date: Tue, 6 Oct 2026 15:38:37 -0700 Subject: [PATCH 5/7] Check the function unit tests against CONTRIBUTING's rules CONTRIBUTING's Tests section sets out how a function's unit tests are written, but only a reviewer checked that a test followed it. This commit adds a checker that turns the rules an AST can decide into a pass or fail: a table and the test that runs it, a case's fields, name and reason, the assertion message, copies, merges and computed child names, module-level objects other than tables, a helper's parameters, and test names. It lives in hack/ and uses only the standard library. A function-test-style flake check type-checks and runs it, and nix run .#fix now lints its directory. A test with a reason to depart from a rule escapes it on the reported line, as in "# noqa: MPT401 # why". That's the form ruff already reads, so ruff now treats MPT codes as external rather than unknown. A case name counts an acronym such as GKE as one word, because many cases name a cloud or protocol that way. The new check fails on five violations in the tests as they stand. Towards #473. Signed-off-by: Nic Cope --- .../compose-model-deployment/tests/test_fn.py | 2 +- .../tests/test_backends.py | 2 +- hack/check_function_tests.py | 319 ++++++++++++++++++ nix/apps.nix | 6 +- nix/checks.nix | 38 ++- pyproject.toml | 5 + 6 files changed, 359 insertions(+), 13 deletions(-) create mode 100644 hack/check_function_tests.py diff --git a/functions/compose-model-deployment/tests/test_fn.py b/functions/compose-model-deployment/tests/test_fn.py index 6a39f27ad..69e79688a 100644 --- a/functions/compose-model-deployment/tests/test_fn.py +++ b/functions/compose-model-deployment/tests/test_fn.py @@ -2285,7 +2285,7 @@ def test_resolve_required(case: ResolveRequiredCase) -> None: @pytest.mark.parametrize("case", INJECT_NAME_CASES, ids=lambda case: case.name) def test_inject_name(case: InjectNameCase) -> None: """_inject_served_model_name puts the served model name first in each container's env.""" - got = case.template.model_copy(deep=True) + got = case.template.model_copy(deep=True) # noqa: MPT401 # _inject_served_model_name edits it in place. fn._inject_served_model_name(got, case.served) assert got.model_dump() == case.want.model_dump(), case.reason diff --git a/functions/compose-model-replica/tests/test_backends.py b/functions/compose-model-replica/tests/test_backends.py index c69872422..b17aa5a78 100644 --- a/functions/compose-model-replica/tests/test_backends.py +++ b/functions/compose-model-replica/tests/test_backends.py @@ -6747,7 +6747,7 @@ def test_cache_env(case: CacheEnvCase) -> None: @pytest.mark.parametrize("case", APPLY_CASES, ids=lambda case: case.name) def test_apply(case: ApplyCase) -> None: """routing.apply fronts a replica's engines with the routing its serving mode selects.""" - composed = {key: k8sobjv1alpha1.Object.model_validate(copy.deepcopy(obj)) for key, obj in case.composed.items()} + composed = {key: k8sobjv1alpha1.Object.model_validate(copy.deepcopy(obj)) for key, obj in case.composed.items()} # noqa: MPT401 # routing.apply edits the manifests in place. got = routing.apply(composed, case.replica, case.provider_config) assert _sorted(objects=_to_dicts(objects=got)) == _sorted(objects=case.want), case.reason diff --git a/hack/check_function_tests.py b/hack/check_function_tests.py new file mode 100644 index 000000000..94f0a15c8 --- /dev/null +++ b/hack/check_function_tests.py @@ -0,0 +1,319 @@ +#!/usr/bin/env python3 +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Check the function unit tests against the rules in CONTRIBUTING.md's Tests section. + +This checks every functions/*/tests/test_*.py, or the files it's given, against +the rules an AST can decide without judgement: how a table and its test are +laid out, a case's fields, name and reason, the assertion message, the calls +that derive or mutate a value, the parameters a helper takes, and the names of +tests. Whether a reason is true is for a reviewer. + +Where a test has a real need to depart from a rule, an escape on the line the +checker reports, with a stated reason, accepts the departure: + + got = case.template.model_copy(deep=True) # noqa: MPT401 # fn edits it in place. +""" + +import ast +import collections.abc +import io +import pathlib +import re +import sys +import tokenize +import typing + +REPO = pathlib.Path(__file__).resolve().parents[1] + +# A table's name. COMPOSE_CASES is run by test_compose. +TABLE = re.compile(r"[A-Z][A-Z0-9_]*_CASES") + +CAMEL_CASE = re.compile(r"[A-Z][A-Za-z0-9]*") +MAX_NAME_WORDS = 4 +MAX_TEST_WORDS = 3 + +# The capital letters that start a word: one that doesn't follow another +# capital, or one that starts a lowercase run. An acronym such as the GKE in +# GKEFirstPass counts as one word, as it does in Go's names. +WORD = re.compile(r"(?[A-Z]+[0-9]+(?:[\s,]+[A-Z]+[0-9]+)*)(?P.*)") + +COPIES = {"copy.copy", "copy.deepcopy"} +DERIVING_METHODS = {"CopyFrom", "MergeFrom", "SetInParent", "model_copy"} + + +class Violation(typing.NamedTuple): + """A rule a test file breaks, at a line.""" + + line: int + code: str + message: str + + +def is_call(node: ast.AST, func: str) -> bool: + """Whether node calls func, written as its dotted name.""" + return isinstance(node, ast.Call) and ast.unparse(node.func) == func + + +def literal(node: ast.expr | None) -> str | None: + """node's value, if it's a string literal.""" + if isinstance(node, ast.Constant) and isinstance(node.value, str): + return node.value + return None + + +def table_name(stmt: ast.stmt) -> str | None: + """The name of the table stmt assigns, if it assigns one.""" + if isinstance(stmt, ast.Assign) and len(stmt.targets) == 1: + target = stmt.targets[0] + elif isinstance(stmt, ast.AnnAssign): + target = stmt.target + else: + return None + if isinstance(target, ast.Name) and TABLE.fullmatch(target.id): + return target.id + return None + + +def runs_cases(fn: ast.FunctionDef) -> bool: + """Whether fn is parametrized over cases.""" + return any( + is_call(d, "pytest.mark.parametrize") and literal(d.args[0]) == "case" + for d in fn.decorator_list + if isinstance(d, ast.Call) and d.args + ) + + +def check_tables(tree: ast.Module) -> list[Violation]: + """Each table is a list, run by the test directly after it, and no other test runs cases.""" + out = [] + for i, stmt in enumerate(tree.body): + before = tree.body[i - 1] if i > 0 else None + if isinstance(stmt, ast.FunctionDef) and runs_cases(stmt) and (before is None or table_name(before) is None): + msg = f"{stmt.name} runs cases but doesn't follow the _CASES table it runs" + out.append(Violation(stmt.lineno, "MPT103", msg)) + + if not isinstance(stmt, ast.Assign | ast.AnnAssign) or (table := table_name(stmt)) is None: + continue + if not isinstance(stmt.value, ast.List): + out.append(Violation(stmt.lineno, "MPT104", f"{table} is built by code, not written as a list of cases")) + test = "test_" + table.removesuffix("_CASES").lower() + after = tree.body[i + 1] if i + 1 < len(tree.body) else None + if not isinstance(after, ast.FunctionDef) or after.name != test: + out.append(Violation(stmt.lineno, "MPT101", f"{table} isn't followed directly by {test}")) + continue + want = f'pytest.mark.parametrize("case", {table}, ids=lambda case: case.name)' + dumped = ast.dump(ast.parse(want, mode="eval").body) + if not any(ast.dump(d) == dumped for d in after.decorator_list): + out.append(Violation(after.lineno, "MPT102", f"{test} isn't decorated with @{want}")) + return out + + +def check_case_classes(tree: ast.Module) -> list[Violation]: + """Each case dataclass's fields start with name and reason, and end with want.""" + out = [] + for stmt in tree.body: + if not isinstance(stmt, ast.ClassDef) or not stmt.name.endswith("Case"): + continue + fields = [ + f"{ast.unparse(s.target)}: {ast.unparse(s.annotation)}" for s in stmt.body if isinstance(s, ast.AnnAssign) + ] + if fields[:2] != ["name: str", "reason: str"] or not fields[2:] or not fields[-1].startswith("want:"): + msg = f"{stmt.name}'s fields don't start name: str, reason: str and end want" + out.append(Violation(stmt.lineno, "MPT201", msg)) + return out + + +def check_cases(tree: ast.Module) -> list[Violation]: + """Each case passes a short, unique CamelCase name and a one-sentence reason, as literals.""" + out = [] + for stmt in tree.body: + if not isinstance(stmt, ast.Assign | ast.AnnAssign) or (table := table_name(stmt)) is None: + continue + if not isinstance(stmt.value, ast.List): + continue + seen = set() + for entry in stmt.value.elts: + kwargs = {k.arg: k.value for k in entry.keywords} if isinstance(entry, ast.Call) else {} + name, reason = literal(kwargs.get("name")), literal(kwargs.get("reason")) + if name is None or reason is None: + msg = "a case doesn't pass name= and reason= as string literals" + out.append(Violation(entry.lineno, "MPT202", msg)) + continue + line = kwargs["name"].lineno + if not CAMEL_CASE.fullmatch(name) or len(WORD.findall(name)) > MAX_NAME_WORDS: + out.append(Violation(line, "MPT203", f"{name!r} isn't CamelCase of at most four words")) + if name in seen: + out.append(Violation(line, "MPT204", f"{name!r} names another case in {table}")) + seen.add(name) + if not reason.endswith(".") or SENTENCE_BREAK.search(reason): + msg = "reason isn't one sentence ending in a full stop" + out.append(Violation(kwargs["reason"].lineno, "MPT205", msg)) + return out + + +def asserts(node: ast.AST) -> collections.abc.Iterator[ast.Assert]: + """The asserts under node, other than those in a with pytest.raises block.""" + for child in ast.iter_child_nodes(node): + if isinstance(child, ast.With) and any(is_call(i.context_expr, "pytest.raises") for i in child.items): + continue + if isinstance(child, ast.Assert): + yield child + yield from asserts(child) + + +def check_assertions(tree: ast.Module) -> list[Violation]: + """A case test's asserts pass case.reason as their message.""" + return [ + Violation(a.lineno, "MPT301", f"{stmt.name}'s assert doesn't pass case.reason as its message") + for stmt in tree.body + if isinstance(stmt, ast.FunctionDef) and runs_cases(stmt) + for a in asserts(stmt) + if a.msg is None or ast.unparse(a.msg) != "case.reason" + ] + + +def check_derivations(tree: ast.Module) -> list[Violation]: + """Nothing copies, merges or computes a value a case should write out.""" + out = [] + for node in ast.walk(tree): + if isinstance(node, ast.Attribute) and (node.attr in DERIVING_METHODS or ast.unparse(node) in COPIES): + out.append(Violation(node.lineno, "MPT401", f"{ast.unparse(node)} derives or mutates a value")) + if isinstance(node, ast.ImportFrom) and node.module == "copy": + out.append(Violation(node.lineno, "MPT401", "the copy module derives a value")) + if (isinstance(node, ast.Attribute) and node.attr == "child_name") or ( + isinstance(node, ast.ImportFrom) and any(a.name == "child_name" for a in node.names) + ): + out.append(Violation(node.lineno, "MPT402", "child_name computes a name a case should write as a literal")) + return out + + +def immutable(node: ast.expr) -> bool: + """Whether node is a constant, a tuple of constants, an alias, or a type such as Literal[...].""" + if isinstance(node, ast.Tuple): + return all(immutable(e) for e in node.elts) + if isinstance(node, ast.BinOp): + return immutable(node.left) and immutable(node.right) + return isinstance(node, ast.Constant | ast.JoinedStr | ast.Subscript | ast.Name | ast.Attribute) + + +def check_globals(tree: ast.Module) -> list[Violation]: + """The tables are the only module-level objects that aren't constants.""" + out = [] + for stmt in tree.body: + if not isinstance(stmt, ast.Assign | ast.AnnAssign) or stmt.value is None or table_name(stmt) is not None: + continue + if not immutable(stmt.value): + target = ast.unparse(stmt.targets[0] if isinstance(stmt, ast.Assign) else stmt.target) + out.append(Violation(stmt.lineno, "MPT403", f"module-level {target} isn't a table or a constant")) + return out + + +def check_helpers(tree: ast.Module) -> list[Violation]: + """A helper takes only keyword-only parameters, with no defaults.""" + out = [] + for stmt in tree.body: + if not isinstance(stmt, ast.FunctionDef | ast.AsyncFunctionDef) or not stmt.name.startswith("_"): + continue + if stmt.name == "_to_dict": + continue + args = stmt.args + params = [a.arg for a in [*args.posonlyargs, *args.args]] + params += [f"*{args.vararg.arg}"] if args.vararg else [] + params += [f"**{args.kwarg.arg}"] if args.kwarg else [] + params += [f"{a.arg}=" for a, d in zip(args.kwonlyargs, args.kw_defaults, strict=True) if d is not None] + if params: + msg = f"{stmt.name} takes {', '.join(params)}; a helper's parameters are keyword-only with no defaults" + out.append(Violation(stmt.lineno, "MPT501", msg)) + return out + + +def check_tests(tree: ast.Module) -> list[Violation]: + """Tests are functions named in at most three words, with no unittest.""" + out = [] + for node in ast.walk(tree): + if isinstance(node, ast.Import) and any(a.name.split(".")[0] == "unittest" for a in node.names): + out.append(Violation(node.lineno, "MPT601", "unittest is imported")) + if isinstance(node, ast.ImportFrom) and (node.module or "").split(".")[0] == "unittest": + out.append(Violation(node.lineno, "MPT601", "unittest is imported")) + if isinstance(node, ast.ClassDef) and any(ast.unparse(b).split(".")[-1] == "TestCase" for b in node.bases): + out.append(Violation(node.lineno, "MPT602", f"{node.name} derives from TestCase")) + for stmt in tree.body: + if not isinstance(stmt, ast.FunctionDef | ast.AsyncFunctionDef) or not stmt.name.startswith("test_"): + continue + if len(stmt.name.removeprefix("test_").split("_")) > MAX_TEST_WORDS: + out.append(Violation(stmt.lineno, "MPT603", f"{stmt.name} has more than three words after test_")) + return out + + +def escapes(source: str) -> tuple[dict[int, set[str]], list[Violation]]: + """The codes each line escapes, and the escapes that state no reason.""" + escaped: dict[int, set[str]] = {} + unexplained = [] + for tok in tokenize.generate_tokens(io.StringIO(source).readline): + if tok.type != tokenize.COMMENT or (match := NOQA.search(tok.string)) is None: + continue + codes = {c for c in re.split(r"[\s,]+", match["codes"]) if c.startswith("MPT")} + if not codes: + continue + escaped[tok.start[0]] = codes + if not match["reason"].strip(" #-:"): + msg = "escape states no reason; write # noqa: MPTnnn # why" + unexplained.append(Violation(tok.start[0], "MPT001", msg)) + return escaped, unexplained + + +def check(source: str) -> list[Violation]: + """The rules source breaks, other than those it escapes.""" + tree = ast.parse(source) + found = [ + *check_tables(tree), + *check_case_classes(tree), + *check_cases(tree), + *check_assertions(tree), + *check_derivations(tree), + *check_globals(tree), + *check_helpers(tree), + *check_tests(tree), + ] + escaped, unexplained = escapes(source) + return sorted([v for v in found if v.code not in escaped.get(v.line, set())] + unexplained) + + +def main() -> int: + """Check the files named on the command line, or every function's tests.""" + paths = [pathlib.Path(a).resolve() for a in sys.argv[1:]] or sorted(REPO.glob("functions/*/tests/test_*.py")) + count = 0 + for path in paths: + shown = path.relative_to(REPO) if path.is_relative_to(REPO) else path + for v in check(path.read_text()): + print(f"{shown}:{v.line}: {v.code} {v.message}", file=sys.stderr) + count += 1 + print(f"\nchecked {len(paths)} test file(s)") + if count: + print(f"{count} violation(s) of CONTRIBUTING.md's Tests section", file=sys.stderr) + return 1 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/nix/apps.nix b/nix/apps.nix index 8dc475f4c..5eb57c432 100644 --- a/nix/apps.nix +++ b/nix/apps.nix @@ -31,7 +31,7 @@ -ignore '**/*.toml' \ -ignore '**/*.yaml' \ -ignore '**/*.yml' \ - functions/ docs/utils/validate/ nix.sh + functions/ docs/utils/validate/ hack/ nix.sh echo "Formatting and linting Nix..." statix fix . @@ -46,8 +46,8 @@ find . -name '*.sh' -type f -exec shellcheck {} + echo "Formatting and linting Python..." - ruff format functions/ - ruff check --fix functions/ + ruff format functions/ docs/utils/validate/ hack/ + ruff check --fix functions/ docs/utils/validate/ hack/ echo "Refreshing uv.lock..." uv lock diff --git a/nix/checks.nix b/nix/checks.nix index 04e7d4e8f..61cb6e2b7 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -93,6 +93,27 @@ in touch $out/.docs-manifests-validated ''; + # Check the function unit tests against the rules in CONTRIBUTING.md's Tests + # section that an AST can decide, such as how a table and its test are laid + # out and what a case's name and reason look like. The checker uses only the + # standard library, so it runs on the plain interpreter and ty needs no venv + # to type-check it. + function-test-style = + pkgs.runCommand "modelplane-function-test-style" + { + nativeBuildInputs = [ + pkgs.python3 + pkgs.unstable.ty + ]; + } + '' + cd ${self} + ty check hack/check_function_tests.py + python3 hack/check_function_tests.py + mkdir -p $out + touch $out/.function-test-style-checked + ''; + python = pkgs.runCommand "modelplane-python-checks" { @@ -102,8 +123,8 @@ in cp -r ${self} src chmod -R u+w src cd src - ruff format --check functions/ docs/utils/validate/ - ruff check functions/ docs/utils/validate/ + ruff format --check functions/ docs/utils/validate/ hack/ + ruff check functions/ docs/utils/validate/ hack/ mkdir -p $out touch $out/.python-checks-passed ''; @@ -137,11 +158,12 @@ in ''; # Fail if any hand-written source file is missing its Apache 2.0 license - # header. Scoped to the files we author: the composition functions and the - # docs manifest validator. Generated models under schemas/python carry their - # own codegen banner, and config (*.toml) and vendored upstream CRDs (*.yaml) - # are excluded. addlicense -check only reads, so it runs against the store - # path directly. Run 'nix run .#fix' to add any missing headers. + # header. Scoped to the files we author: the composition functions, the docs + # manifest validator, and the scripts in hack/. Generated models under + # schemas/python carry their own codegen banner, and config (*.toml) and + # vendored upstream CRDs (*.yaml) are excluded. addlicense -check only reads, + # so it runs against the store path directly. Run 'nix run .#fix' to add any + # missing headers. license = pkgs.runCommand "modelplane-license-check" { @@ -153,7 +175,7 @@ in -ignore '**/*.toml' \ -ignore '**/*.yaml' \ -ignore '**/*.yml' \ - functions/ docs/utils/validate/ nix.sh + functions/ docs/utils/validate/ hack/ nix.sh mkdir -p $out touch $out/.license-check-passed ''; diff --git a/pyproject.toml b/pyproject.toml index 7ae62a000..b1bffcf3a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -60,6 +60,9 @@ select = [ "PT", # flake8-pytest-style (pytest idioms, no unittest assertions) "RUF", # ruff-specific rules (incl. unused/malformed noqa) ] +# The function unit test checker (hack/check_function_tests.py) takes its +# escapes as noqa codes, which ruff would otherwise reject as unknown. +external = ["MPT"] [tool.ruff.lint.pylint] # Builder functions like helm_release use named parameters for clarity. @@ -82,6 +85,8 @@ allow-star-arg-any = true "functions/*/function/fn.py" = ["N802"] # The docs manifest validator is a CLI script; print is its output. "docs/utils/validate/**" = ["T201"] +# The function unit test checker is a CLI script; print is its output. +"hack/check_function_tests.py" = ["T201"] # Generated serving stack lists carry upstream chart values verbatim; long # string literals and unicode arrive from the recipes. "functions/*/function/stacks/clouds/generated/**" = ["E501", "RUF001", "RUF002", "RUF003"] From e1da1d8a0ba24af05756a4c995456495ef4e60cf Mon Sep 17 00:00:00 2001 From: Nic Cope Date: Mon, 5 Oct 2026 16:45:26 -0700 Subject: [PATCH 6/7] Write the e2e tests as a pytest suite The local e2e was a shell script, e2e/run.sh. Most of it checked the running environment, with polling loops written out by hand, JSON and logs matched as text, and an exit at the first failed check, so one fault hid every check after it. This commit replaces run.sh with a pytest suite that makes the checks run.sh's --verify made, one test each, and brings the clusters up the way run.sh did. Each test waits only for what it needs, so a gateway that refuses every caller fails only the tests that need it to serve. nix run .#e2e keeps its flags, and gains --test, to rerun the tests without bringing the clusters up again. Four checks are stricter. The response must name the served model in its top-level model field, where a model key at any depth passed. An unclaimed model must get a 404, where any status but 200 passed. /v1/models must list the model by its exact name, where a substring of another name passed. And the usage record must carry the full endpoint name, where a prefix of it passed. Fixes #473. Signed-off-by: Nic Cope --- .github/workflows/e2e.yml | 4 +- CONTRIBUTING.md | 5 +- e2e/README.md | 65 ++- e2e/__init__.py | 15 + e2e/dra-example-driver.yaml | 4 +- e2e/environment.py | 291 ++++++++++++ e2e/gateway.py | 110 +++++ e2e/kube.py | 120 +++++ e2e/lean-control-plane.yaml | 6 +- e2e/manifests/00-namespaces.yaml | 5 +- e2e/manifests/10-inference-gateway.yaml | 4 +- e2e/manifests/20-inference-class.yaml | 2 +- e2e/manifests/30-inference-cluster.yaml | 2 +- e2e/run.sh | 555 ----------------------- e2e/tests/conftest.py | 238 ++++++++++ e2e/tests/test_auth.py | 60 +++ e2e/tests/test_cluster_gateway.py | 70 +++ e2e/tests/test_metering.py | 54 +++ e2e/tests/test_routing.py | 86 ++++ e2e/tests/test_telemetry.py | 171 +++++++ e2e/wait.py | 50 +++ flake.nix | 2 +- nix/apps.nix | 75 +++- nix/checks.nix | 39 +- pyproject.toml | 12 +- uv.lock | 562 ++++++++++++++++++++++++ 26 files changed, 1982 insertions(+), 625 deletions(-) create mode 100644 e2e/__init__.py create mode 100644 e2e/environment.py create mode 100644 e2e/gateway.py create mode 100644 e2e/kube.py delete mode 100644 e2e/run.sh create mode 100644 e2e/tests/conftest.py create mode 100644 e2e/tests/test_auth.py create mode 100644 e2e/tests/test_cluster_gateway.py create mode 100644 e2e/tests/test_metering.py create mode 100644 e2e/tests/test_routing.py create mode 100644 e2e/tests/test_telemetry.py create mode 100644 e2e/wait.py diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml index dae613934..15f5c1437 100644 --- a/.github/workflows/e2e.yml +++ b/.github/workflows/e2e.yml @@ -66,8 +66,8 @@ jobs: # The exact command a developer runs locally (--print-build-logs is only CI # verbosity), so a green run here and a green run on a laptop mean the same - # thing. It brings up both clusters, deploys the mock model, waits for the - # ModelService, and asserts a live 200 — exiting non-zero on any failure. + # thing. It brings up both clusters, deploys the mock model, and runs the + # tests in e2e/tests/, exiting non-zero if any fails. - name: Run e2e run: nix run .#e2e --print-build-logs -- --verify diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index fa3d3a7a8..25be6e88b 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -105,7 +105,8 @@ curl -fsSL https://install.determinate.systems/nix | sh -s -- install `nix flake check` runs all of the project's checks inside the Nix sandbox: Python, shell, and Nix linters and formatters, the [ty](https://docs.astral.sh/ty) -type checker on every composition function, plus unit tests for every function. +type checker on every composition function and the end-to-end tests, plus unit +tests for every function. Run `nix flake show` to see what else is available. ```bash @@ -120,7 +121,7 @@ composition function renders the right resources. The integration layer is `nix run .#e2e`, which brings up two local `kind` clusters and runs the whole path — scheduling, the serving-stack install on a registered cluster, gateway routing, a live request — with no cloud credentials. Add `-- --verify` -and it waits for readiness, asserts a 200, and exits non-zero on failure. That +and it runs the pytest suite in `e2e/tests/`, exiting non-zero if a test fails. That verify command is what the label-gated `E2E` workflow runs on CI (add the `test-e2e` label to a PR), so a green local `--verify` and a green CI run mean the same thing. See `e2e/README.md`. diff --git a/e2e/README.md b/e2e/README.md index ff7aa6751..f5978ded4 100644 --- a/e2e/README.md +++ b/e2e/README.md @@ -97,10 +97,10 @@ runs, so prefer a separate control plane for cloud work. (kube-prometheus-stack et al.) add up fast; a full Docker disk surfaces as `no space left on device`. Reclaim between runs with `docker builder prune -af` and `docker image prune -af`. -- Everything else — `kind`, `kubectl`, `curl`, `git`, and the `crossplane` CLI — - is provided by the flake via `nix run .#e2e`. +- Everything else — `kind`, `kubectl`, the `docker` CLI, Python and pytest, and + the `crossplane` CLI — is provided by the flake via `nix run .#e2e`. -The workload cluster is pinned to **k8s v1.34** (in `run.sh`) for the +The workload cluster is pinned to **k8s v1.34** (in `environment.py`) for the `resource.k8s.io` (DRA) APIs, on-by-default in 1.34: both the serving stack's NVIDIA DRA driver and the dra-example-driver register DeviceClasses, and the example driver publishes the `ResourceSlice`s the engine's `ResourceClaim` binds @@ -110,10 +110,20 @@ against. The control-plane cluster needs no DRA. ```bash nix run .#e2e # bring up both clusters + deploy the mock model -nix run .#e2e -- --verify # same, then wait for readiness and assert a live 200 +nix run .#e2e -- --verify # same, then run the tests +nix run .#e2e -- --test # run the tests against clusters already up nix run .#e2e -- --clean # tear both clusters down ``` +Arguments after `--verify` go to pytest, so `nix run .#e2e -- --verify -k usage` +runs only the tests whose names match. Bring-up reuses clusters that are already +up. To rerun the tests against an environment that's up, without bringing it up +again, use `--test`: + +```bash +nix run .#e2e -- --test -k usage +``` + `crossplane project run` installs the config and applies the resources, then returns; the serving-stack install and model rollout reconcile in the background. So wait for the `ModelService` to become ready before curling. The gateway's @@ -139,16 +149,16 @@ kubectl run curl -n ml-team --rm -it --image=curlimages/curl@sha256:7c12af72ceb3 -d '{"model":"ml-team/mock","max_tokens":16,"messages":[{"role":"user","content":"hi"}]}' ``` -`--verify` runs both of those, among other checks, and exits non-zero on -failure. It's the exact command the `E2E` CI workflow runs, so a green `--verify` -locally and a green CI run mean the same thing; use the manual curls above to -poke the endpoints interactively. +`--verify` runs the tests in `e2e/tests/`, which send both of those requests among +others, and exits non-zero if any fails. It's the exact command the `E2E` CI +workflow runs, so a green `--verify` locally and a green CI run mean the same +thing; use the manual curls above to poke the endpoints interactively. ## How it's structured `nix run .#e2e` materialises the Nix-built function images and hands off to -`run.sh`, which does the cross-cluster orchestration that `crossplane project -run` flags can't express: +`environment.py`, which does the cross-cluster orchestration that `crossplane +project run` flags can't express: 1. Create the **workload** kind cluster (pinned v1.34). 2. Install MetalLB on it (the serving stack doesn't) with a pool inside the @@ -164,13 +174,26 @@ run` flags can't express: control-plane pods over the shared kind network), then apply the Modelplane manifests. -Everything the control plane needs is a declarative manifest; the shell in -`run.sh` is only the irreducible cross-cluster setup (a second cluster, its -MetalLB and DRA driver, and the cross-cluster kubeconfig). +Everything the control plane needs is a declarative manifest; `environment.py` +is only the irreducible cross-cluster setup (a second cluster, its MetalLB and +DRA driver, and the cross-cluster kubeconfig). With `--verify`, pytest then runs +the tests in `tests/`, one file per area. Fixtures in `conftest.py` start a curl pod on each +cluster to send requests from, and wait for the model to route and to serve. The +tests read the clusters through the Kubernetes API, with the official Python +client, while bring-up drives the kind, crossplane, docker and kubectl CLIs. ``` e2e/ - run.sh # two-cluster orchestration + environment.py # two-cluster bring-up and teardown + kube.py, gateway.py # Kubernetes API and curl helpers + wait.py # polling until a condition holds + tests/ + conftest.py # fixtures: the clusters, curl pods, readiness + test_auth.py # the InferenceGateway refuses unknown callers + test_routing.py # it routes OpenAI and Anthropic requests + test_metering.py # it logs a usage record per request + test_cluster_gateway.py # the cluster gateway requires a client certificate + test_telemetry.py # the engine's series reach the collector's sink dra-example-driver.yaml # vendored fake DRA GPU driver (applied to workload) manifests/ # applied to the control plane after setup 00-namespaces.yaml @@ -179,6 +202,7 @@ e2e/ 30-inference-cluster.yaml # source: Existing -> the workload cluster 40-model-deployment.yaml 50-model-service.yaml + 60-telemetry.yaml # a TelemetryDestination with a debug sink ``` ## Why the extra moving parts @@ -186,11 +210,11 @@ e2e/ - **MetalLB on the workload cluster.** Both gateways run there, and both need `LoadBalancer` addresses kind can't provide: the serving stack gates the cluster gateway's readiness on having one (`READY_CEL` in its `gateway.py`). - Nothing Modelplane composes installs MetalLB, so `run.sh` does, with a pool + Nothing Modelplane composes installs MetalLB, so bring-up does, with a pool inside the detected kind Docker subnet (see caveat) so the control plane can route to the addresses it hands out. - **Fake DRA driver.** A `claim: DRA` engine emits a `ResourceClaim`; with no DRA - driver it stays Pending and the pod never schedules. `run.sh` applies the + driver it stays Pending and the pod never schedules. Bring-up applies the vendored **dra-example-driver**, which publishes fake `gpu.example.com` devices so the claim binds on a GPU-less node. - **Cross-cluster kubeconfig.** `source: Existing` needs a kubeconfig the @@ -198,12 +222,12 @@ e2e/ `kind get kubeconfig --internal` gives an address routable across the shared kind network; a host kubeconfig (`127.0.0.1:`) wouldn't be. - **Node label.** On a BYO cluster Modelplane doesn't provision/label pools, so - `run.sh` labels the workload node `modelplane.ai/pool=gpu-synthetic` (matching + bring-up labels the workload node `modelplane.ai/pool=gpu-synthetic` (matching `nodePools[].name`); without it worker pods stay Pending. ## Caveats / open questions -- **Cross-cluster networking uses the detected kind subnet.** `run.sh` reads the +- **Cross-cluster networking uses the detected kind subnet.** Bring-up reads the `kind` Docker network's subnet (usually 172.18.0.0/16, but kind bumps to 172.19/... when earlier networks already hold 172.18) and derives the workload cluster's MetalLB pool from it. A hardcoded 172.18 would leave the LB @@ -212,10 +236,11 @@ e2e/ for the config to install, then applies the resources and exits — it doesn't block on XR readiness. The serving-stack install (the long pole) and the model rollout happen after, so watch the `ModelService`'s `RoutingReady` rather than - the command's exit. `--timeout` in `run.sh` bounds the build and config install. + the command's exit. `--timeout` in `environment.py` bounds the build and config + install. - **Two DRA drivers on a GPU-less node.** The serving stack's **NVIDIA** DRA driver targets NFD-GPU-labelled nodes, so it sits at 0/0 (inert) yet its Helm - release still reports Ready. The **dra-example-driver** `run.sh` installs is the + release still reports Ready. The **dra-example-driver** bring-up installs is the active one — it publishes the fake `gpu.example.com` devices the engine binds. - **Serving-stack weight.** cert-manager, Envoy Gateway, Envoy AI Gateway, GAIE CRDs, kube-prometheus-stack, LeaderWorkerSet, NFD, DRA driver — all on the diff --git a/e2e/__init__.py b/e2e/__init__.py new file mode 100644 index 000000000..ac2a31de4 --- /dev/null +++ b/e2e/__init__.py @@ -0,0 +1,15 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""End-to-end tests that bring Modelplane up on kind and send it traffic. See README.md.""" diff --git a/e2e/dra-example-driver.yaml b/e2e/dra-example-driver.yaml index f81566700..793786c1d 100644 --- a/e2e/dra-example-driver.yaml +++ b/e2e/dra-example-driver.yaml @@ -11,8 +11,8 @@ # # then prepend the Namespace below. The template output already starts with # # a --- separator, so don't add another after the Namespace.) # -# run.sh applies this to the workload cluster. The kubeletplugin publishes 8 fake -# gpu.example.com devices via a ResourceSlice, so a `claim: DRA` engine's +# Bring-up applies this to the workload cluster. The kubeletplugin publishes 8 +# fake gpu.example.com devices via a ResourceSlice, so a `claim: DRA` engine's # ResourceClaim binds on a GPU-less node — exercising the real DRA allocation # path the fleet scheduler and composition emit, without a GPU. apiVersion: v1 diff --git a/e2e/environment.py b/e2e/environment.py new file mode 100644 index 000000000..7c514f5ff --- /dev/null +++ b/e2e/environment.py @@ -0,0 +1,291 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Bring up, and tear down, the two kind clusters the end-to-end tests run on. + +The workload cluster runs the serving stack, both gateways and the model. The +control plane runs Crossplane and the Configuration, managed by `crossplane +project run`, and registers the workload cluster as InferenceCluster local with +source: Existing. See README.md. + + python -m e2e.environment up [--no-apply] + python -m e2e.environment down +""" + +import argparse +import ipaddress +import json +import logging +import os +import pathlib +import shlex +import subprocess +import tempfile + +log = logging.getLogger(__name__) + +ROOT = pathlib.Path(__file__).resolve().parent.parent + +CONTROL_PLANE = "modelplane-e2e-local" +WORKLOAD = "modelplane-e2e-workload" +CONTROL_PLANE_CONTEXT = f"kind-{CONTROL_PLANE}" +WORKLOAD_CONTEXT = f"kind-{WORKLOAD}" + +# Pinned so the workload cluster has the DRA APIs the serving stack's NVIDIA DRA +# driver needs (resource.k8s.io, GA in k8s 1.34). The control plane needs no DRA, +# so its image doesn't matter. v1.34.2 or newer: older kubelets deadlock on an +# idle DRA connection (k/k#133934). +WORKLOAD_NODE_IMAGE = "kindest/node:v1.34.8@sha256:02722c2dedddcfc00febf5d27fbeb9b7b2c14294c82109ff4a85d89ac9ba3256" +DEADLOCKING_KUBELETS = ("v1.34.0", "v1.34.1") + +METALLB_URL = "https://raw.githubusercontent.com/metallb/metallb/v0.14.8/config/manifests/metallb-native.yaml" +CROSSPLANE_VERSION = "2.4.0" + +MANIFESTS = ROOT / "e2e" / "manifests" + + +class BringUpError(Exception): + """The environment can't be brought up as asked.""" + + +def up(*, apply_manifests: bool) -> None: + """Bring up both clusters and install Modelplane, then apply the manifests the tests use.""" + up_workload() + up_control_plane() + if not apply_manifests: + log.info("Control plane ready. Skipped applying %s", MANIFESTS) + return + log.info("Applying %s", MANIFESTS) + kubectl(CONTROL_PLANE_CONTEXT, "apply", f"--filename={MANIFESTS}") + + +def up_workload() -> None: + """Create the workload cluster, and install what Modelplane expects a cluster to have already.""" + create_workload_cluster() + + # Both kind clusters share one Docker network. MetalLB hands out + # LoadBalancer addresses from it, and the control plane must route to them, + # so the pool has to sit inside the network's actual subnet. That's usually + # 172.18.0.0/16, but kind moves to 172.19 and beyond when another Docker + # network holds 172.18. The serving stack doesn't install MetalLB, and the + # pool needs room for two Services, one per gateway. + prefix = kind_subnet_prefix() + log.info("Installing MetalLB on the workload cluster, with pool %s.255.100-149", prefix) + kubectl(WORKLOAD_CONTEXT, "apply", f"--filename={METALLB_URL}") + kubectl(WORKLOAD_CONTEXT, "rollout", "status", "--namespace=metallb-system", "deploy/controller", "--timeout=180s") + pool = { + "apiVersion": "v1", + "kind": "List", + "items": [ + { + "apiVersion": "metallb.io/v1beta1", + "kind": "IPAddressPool", + "metadata": {"name": "kind-pool", "namespace": "metallb-system"}, + "spec": {"addresses": [f"{prefix}.255.100-{prefix}.255.149"]}, + }, + { + "apiVersion": "metallb.io/v1beta1", + "kind": "L2Advertisement", + "metadata": {"name": "kind-l2", "namespace": "metallb-system"}, + "spec": {"ipAddressPools": ["kind-pool"]}, + }, + ], + } + kubectl(WORKLOAD_CONTEXT, "apply", "--filename=-", stdin=json.dumps(pool)) + + # Fake DRA GPUs, so a `claim: DRA` engine's ResourceClaim binds on this + # GPU-less node. Without a DRA driver the claim stays Pending and the engine + # never schedules, and the fleet scheduler rejects an engine whose only + # device is Synthetic. + log.info("Installing dra-example-driver (fake GPUs) on the workload cluster") + kubectl(WORKLOAD_CONTEXT, "apply", f"--filename={ROOT / 'e2e' / 'dra-example-driver.yaml'}") + kubectl( + WORKLOAD_CONTEXT, + "rollout", "status", "--namespace=dra-example-driver", "ds/dra-example-driver-kubeletplugin", "--timeout=120s", + ) # fmt: skip + + # Modelplane doesn't label a BYO cluster's nodes. The gpu-synthetic pool + # selects on this. + log.info("Labelling the workload node for pool gpu-synthetic") + kubectl( + WORKLOAD_CONTEXT, + "label", + "node", + f"{WORKLOAD}-control-plane", + "modelplane.ai/pool=gpu-synthetic", + "--overwrite", + ) + + +def create_workload_cluster() -> None: + """Create the workload cluster, or reuse one running the pinned Kubernetes minor version.""" + if WORKLOAD not in output("kind", "get", "clusters").split(): + log.info("Creating workload cluster %s (k8s v1.34, for DRA)", WORKLOAD) + run("kind", "create", "cluster", f"--name={WORKLOAD}", f"--image={WORKLOAD_NODE_IMAGE}") + return + + # An older cluster lacks the DRA APIs, and would fail the run confusingly + # later. + try: + version = output( + "kubectl", f"--context={WORKLOAD_CONTEXT}", + "get", "nodes", "--output=jsonpath={.items[0].status.nodeInfo.kubeletVersion}", + ) # fmt: skip + except subprocess.CalledProcessError: + version = "unreachable" + if version in DEADLOCKING_KUBELETS: + msg = ( + f"workload cluster {WORKLOAD} is {version}, whose kubelet deadlocks on an idle DRA connection " + "(fixed in v1.34.2); recreate it with: nix run .#e2e -- --clean" + ) + raise BringUpError(msg) + if not version.startswith("v1.34."): + msg = ( + f"workload cluster {WORKLOAD} is {version}, but v1.34 is required for the DRA APIs; " + "recreate it with: nix run .#e2e -- --clean" + ) + raise BringUpError(msg) + log.info("Reusing workload cluster %s (%s)", WORKLOAD, version) + + +def kind_subnet_prefix() -> str: + """Return the first two octets of the kind Docker network's IPv4 subnet.""" + subnets = output( + "docker", "network", "inspect", "kind", "--format={{range .IPAM.Config}}{{println .Subnet}}{{end}}" + ) + for subnet in subnets.split(): + network = ipaddress.ip_network(subnet) + if isinstance(network, ipaddress.IPv4Network): + log.info("kind Docker subnet is %s", network) + return ".".join(str(network.network_address).split(".")[:2]) + msg = f"could not find an IPv4 subnet on the kind Docker network: {subnets!r}" + raise BringUpError(msg) + + +def up_control_plane() -> None: + """Build Modelplane and install it on a kind control plane, then register the workload cluster with it.""" + log.info("Building and running the control plane %s", CONTROL_PLANE) + # The nix app runs with no system PATH, so a Docker config whose credsStore + # is "desktop" would break package resolution. The provider packages are + # public, so an empty config is enough. + with tempfile.TemporaryDirectory() as docker_config: + pathlib.Path(docker_config, "config.json").write_text("{}") + # The lean control plane's narrowed MRAP goes in with --init-resources, + # ahead of the providers, so the cloud providers stay dormant (safe-start + # scales them to zero). prerequisites.yaml can't go the same way: it + # opens with a comment-only YAML document, which `crossplane project run` + # rejects and kubectl skips. + run( + "crossplane", "project", "run", + f"--control-plane-name={CONTROL_PLANE}", "--cluster-admin", "--timeout=25m", + f"--init-resources={ROOT / 'e2e' / 'lean-control-plane.yaml'}", + f"--crossplane-version={CROSSPLANE_VERSION}", + env={"DOCKER_CONFIG": docker_config}, + ) # fmt: skip + + # Finish the setup the install guide does by hand: apply the RBAC + # prerequisites, then point the two providers at the DeploymentRuntimeConfigs + # they define. The providers install before prerequisites.yaml, and an + # ImageConfig binds only when a ProviderRevision is created, so provider-helm + # would otherwise come up without the RBAC it grants, and + # provider-kubernetes without --sanitize-secrets. + log.info("Applying the prerequisites and the provider runtime configs") + prerequisites = ROOT / "docs" / "manifests" / "install" / "prerequisites.yaml" + kubectl(CONTROL_PLANE_CONTEXT, "apply", f"--filename={prerequisites}") + for provider, runtime_config in ( + ("upbound-provider-helm", "provider-helm-modelplane"), + ("upbound-provider-kubernetes", "provider-kubernetes-modelplane"), + ): + patch = { + "spec": { + "runtimeConfigRef": { + "apiVersion": "pkg.crossplane.io/v1beta1", + "kind": "DeploymentRuntimeConfig", + "name": runtime_config, + } + } + } + kubectl( + CONTROL_PLANE_CONTEXT, + "patch", f"provider.pkg.crossplane.io/{provider}", "--type=merge", f"--patch={json.dumps(patch)}", + ) # fmt: skip + + # InferenceCluster local reads this kubeconfig to reach the workload + # cluster. --internal gives the address on the kind network, which the + # control plane's provider pods can reach and 127.0.0.1 isn't. + # prerequisites.yaml creates its namespace. + log.info("Registering the workload cluster's kubeconfig with the control plane") + secret = { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "local-cluster-kubeconfig", "namespace": "modelplane-system"}, + "stringData": {"kubeconfig": output("kind", "get", "kubeconfig", "--internal", f"--name={WORKLOAD}")}, + } + kubectl(CONTROL_PLANE_CONTEXT, "apply", "--filename=-", stdin=json.dumps(secret)) + + +def down() -> None: + """Delete both clusters, and the control plane's local registry.""" + # Delete both clusters whatever project stop returns: it can exit 0 without + # removing the cluster. It's here for the local registry it also manages. + for cmd in ( + ("crossplane", "project", "stop", f"--control-plane-name={CONTROL_PLANE}"), + ("kind", "delete", "cluster", f"--name={CONTROL_PLANE}"), + ("kind", "delete", "cluster", f"--name={WORKLOAD}"), + ("docker", "rm", "--force", f"{CONTROL_PLANE}-registry"), + ): + try: + run(*cmd) + except subprocess.CalledProcessError as e: + log.warning("%s", e) + + +def kubectl(context: str, *args: str, stdin: str | None = None) -> None: + """Run kubectl against a cluster, logging what it prints.""" + run("kubectl", f"--context={context}", *args, stdin=stdin) + + +def run(*cmd: str, stdin: str | None = None, env: dict[str, str] | None = None) -> None: + """Run a command from the repository root, letting it print as it runs. + + Bring-up takes most of a CI run, so its progress shows live rather than + being captured and shown only if it fails. + """ + log.info("$ %s", shlex.join(cmd)) + subprocess.run(cmd, cwd=ROOT, env={**os.environ, **(env or {})}, input=stdin, text=True, check=True) + + +def output(*cmd: str) -> str: + """Run a command from the repository root, and return what it wrote to stdout.""" + return subprocess.run(cmd, cwd=ROOT, stdout=subprocess.PIPE, text=True, check=True).stdout.strip() + + +def main() -> None: + """Bring the environment up or down.""" + parser = argparse.ArgumentParser(prog="python -m e2e.environment", description=__doc__.split("\n\n")[0]) + commands = parser.add_subparsers(dest="command", required=True) + up_command = commands.add_parser("up", help="bring up both clusters, install Modelplane and apply the manifests") + up_command.add_argument("--no-apply", action="store_true", help="skip the manifests, to apply them by hand") + commands.add_parser("down", help="delete both clusters") + args = parser.parse_args() + + logging.basicConfig(level=logging.INFO, format="%(message)s") + if args.command == "down": + down() + return + up(apply_manifests=not args.no_apply) + + +if __name__ == "__main__": + main() diff --git a/e2e/gateway.py b/e2e/gateway.py new file mode 100644 index 000000000..dce5fe2d1 --- /dev/null +++ b/e2e/gateway.py @@ -0,0 +1,110 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Send requests through Modelplane's gateways, and read what they log. + +The gateways' addresses are on the kind Docker network, which a macOS host +can't route to. So requests go from a pod on one of the clusters, which can. +""" + +import dataclasses +import json +import typing + +from e2e import kube + +# The InferenceGateway's Envoy proxy pods, on the workload cluster. +PROXY_NAMESPACE = "envoy-gateway-system" +PROXY_SELECTOR = "gateway.envoyproxy.io/owning-gateway-name=inference-gateway" + +# The key manifests/10-inference-gateway.yaml gives caller e2e. +CALLER_KEY = "sk-e2e-caller" + +# How an OpenAI client sends the caller's key. +BEARER = {"authorization": f"Bearer {CALLER_KEY}"} + + +@dataclasses.dataclass(frozen=True) +class Serving: + """A model the InferenceGateway routes, and where to reach it.""" + + # The name a caller sends as the request's model: the ModelService's + # status.model. + model: str + openai: str + anthropic: str + + +@dataclasses.dataclass(frozen=True) +class Response: + """What came back from a request.""" + + # The HTTP status, or 0 if no HTTP response came back. + status: int + # The body, or why no HTTP response came back. + body: str + + def json(self) -> typing.Any: # noqa: ANN401 - a JSON body can decode to any type. + """Decode the body as JSON.""" + return json.loads(self.body) + + +@dataclasses.dataclass(frozen=True) +class Client: + """Runs curl in a pod on a cluster that can reach the gateways.""" + + cluster: kube.Cluster + namespace: str + pod: str + + def request(self, url: str, headers: dict[str, str], body: object | None = None) -> Response: + """Send a request, a POST of body as JSON if there is one and otherwise a GET.""" + curl = ["curl", "--silent", "--show-error", "--max-time", "15", "--write-out", "\n%{http_code}", url] + for name, value in headers.items(): + curl += ["--header", f"{name}: {value}"] + if body is not None: + curl += ["--header", "content-type: application/json", "--data", json.dumps(body)] + result = self.cluster.exec(self.pod, self.namespace, curl) + # curl exits non-zero only when no HTTP response came back. + if result.code != 0: + return Response(status=0, body=result.stderr.strip()) + # --write-out puts the status on a line of its own, after the body. + text, _, status = result.stdout.rpartition("\n") + return Response(status=int(status), body=text) + + def connect(self, url: str) -> int: + """GET a URL without verifying the server's certificate, and return curl's exit code.""" + curl = ["curl", "--silent", "--show-error", "--insecure", "--max-time", "15", "--output", "/dev/null", url] + return self.cluster.exec(self.pod, self.namespace, curl).code + + +def usage_records(cluster: kube.Cluster) -> list[dict[str, typing.Any]]: + """Return every usage record the InferenceGateway's proxy pods have logged. + + A request lands on any one of the proxy pods, so this reads them all. + """ + pods = cluster.core.list_namespaced_pod( + PROXY_NAMESPACE, label_selector=PROXY_SELECTOR, _request_timeout=kube.TIMEOUT_SECONDS + ) + records = [] + for pod in pods.items: + for line in cluster.logs(pod.metadata.name, PROXY_NAMESPACE, "envoy", tail_lines=None).splitlines(): + # Envoy logs other things too. The access log is the JSON objects. + try: + record = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(record, dict) and "caller" in record: + records.append(record) + return records diff --git a/e2e/kube.py b/e2e/kube.py new file mode 100644 index 000000000..1fe333a51 --- /dev/null +++ b/e2e/kube.py @@ -0,0 +1,120 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Read a cluster's resources through the Kubernetes API. + +The official Kubernetes client returns typed objects and typed errors for the +built-in kinds. Cluster wraps the parts of it that need care: Modelplane's own +resources, which come back as dicts for the tests to validate into the +generated models, a command's exit code, and container logs. +""" + +import dataclasses +import shlex +import typing + +import urllib3 +from kubernetes import client, config, stream + +# A bound on each API call, so a hung API server fails the test that hit it +# rather than the whole run. Pass it as _request_timeout: the client has no +# default. +TIMEOUT_SECONDS = 60 + +# What a wait retries: the condition not holding yet, or the API server +# failing a request or a command's exec stalling while the cluster converges. +RETRY = (AssertionError, client.ApiException, urllib3.exceptions.HTTPError, TimeoutError) + + +@dataclasses.dataclass(frozen=True) +class Exec: + """How a command run in a container exited, and what it printed.""" + + code: int + stdout: str + stderr: str + + +class Cluster: + """A cluster, addressed by its kubeconfig context.""" + + def __init__(self, context: str) -> None: + """Connect to the cluster a kubeconfig context names.""" + api = config.new_client_from_config(context=context) + self.core = client.CoreV1Api(api) + self.apps = client.AppsV1Api(api) + self.custom = client.CustomObjectsApi(api) + + def modelplane(self, plural: str, name: str, namespace: str | None) -> dict[str, typing.Any] | None: + """Return a Modelplane resource, or None if it doesn't exist.""" + try: + if namespace is None: + return self.custom.get_cluster_custom_object( + "modelplane.ai", "v1alpha1", plural, name, _request_timeout=TIMEOUT_SECONDS + ) + return self.custom.get_namespaced_custom_object( + "modelplane.ai", "v1alpha1", namespace, plural, name, _request_timeout=TIMEOUT_SECONDS + ) + except client.ApiException as e: + if e.status == 404: + return None + raise + + def exec(self, pod: str, namespace: str, command: list[str]) -> Exec: + """Run a command in a pod's only container, and return how it exited.""" + # The exec API streams over a websocket. Without _preload_content the + # stream stays open until the command exits, which is what yields its + # exit code. + resp = stream.stream( + self.core.connect_get_namespaced_pod_exec, + pod, + namespace, + command=command, + stdout=True, + stderr=True, + stdin=False, + tty=False, + _preload_content=False, + ) + resp.run_forever(timeout=TIMEOUT_SECONDS) + if resp.returncode is None: + msg = f"{shlex.join(command)} in {namespace}/{pod} didn't exit within {TIMEOUT_SECONDS}s" + raise TimeoutError(msg) + return Exec(code=resp.returncode, stdout=resp.read_stdout(), stderr=resp.read_stderr()) + + def logs(self, pod: str, namespace: str, container: str, *, tail_lines: int | None) -> str: + """Return a container's logs: its last tail_lines lines, or all of them if tail_lines is None.""" + # With its content preloaded, the client tries to deserialize the logs, + # and returns them as the repr of a bytes object. + resp = self.core.read_namespaced_pod_log( + pod, + namespace, + container=container, + tail_lines=tail_lines, + _preload_content=False, + _request_timeout=TIMEOUT_SECONDS, + ) + return resp.data.decode() + + +def rolled_out(d: client.V1Deployment) -> None: + """Fail unless a Deployment has finished rolling out, by the test kubectl rollout status makes.""" + name = d.metadata.name + assert (d.status.observed_generation or 0) >= d.metadata.generation, ( + f"Deployment {name}'s controller hasn't seen its latest spec" + ) + want = d.spec.replicas + assert (d.status.updated_replicas or 0) == want, f"Deployment {name} is still updating pods" + assert (d.status.replicas or 0) == want, f"Deployment {name} still has old pods" + assert (d.status.available_replicas or 0) == want, f"Deployment {name} has unavailable pods" diff --git a/e2e/lean-control-plane.yaml b/e2e/lean-control-plane.yaml index d02bea4a5..a8843bdee 100644 --- a/e2e/lean-control-plane.yaml +++ b/e2e/lean-control-plane.yaml @@ -3,9 +3,9 @@ # that only exist to *provision* clusters; here the workload cluster already # exists, so the compositions use only provider-kubernetes and provider-helm. # -# run.sh feeds this via `crossplane project run --init-resources`, which applies -# it BEFORE the Configuration's dependencies install, so the cloud providers -# never activate their managed resources or run their controllers. +# Bring-up feeds this via `crossplane project run --init-resources`, which +# applies it BEFORE the Configuration's dependencies install, so the cloud +# providers never activate their managed resources or run their controllers. # # Replace the Helm chart's default catch-all MRAP (activate: ["*"]) with one # that activates only the managed resources the BYO compositions use. The cloud diff --git a/e2e/manifests/00-namespaces.yaml b/e2e/manifests/00-namespaces.yaml index 96630af9d..a71a90288 100644 --- a/e2e/manifests/00-namespaces.yaml +++ b/e2e/manifests/00-namespaces.yaml @@ -1,6 +1,7 @@ -# The model resources live in ml-team; run.sh applies these manifests with +# The model resources live in ml-team; bring-up applies these manifests with # kubectl once the control plane is ready. modelplane-system is created earlier -# by prerequisites.yaml, before run.sh adds the workload kubeconfig Secret to it. +# by prerequisites.yaml, before bring-up adds the workload kubeconfig Secret to +# it. apiVersion: v1 kind: Namespace metadata: diff --git a/e2e/manifests/10-inference-gateway.yaml b/e2e/manifests/10-inference-gateway.yaml index b6d524121..29bcede14 100644 --- a/e2e/manifests/10-inference-gateway.yaml +++ b/e2e/manifests/10-inference-gateway.yaml @@ -3,7 +3,7 @@ # # Co-locating the InferenceGateway with the models it serves is a supported shape, # and the one this test uses: the workload cluster hosts both gateways, each with -# its own LoadBalancer address from the MetalLB run.sh installs there. +# its own LoadBalancer address from the MetalLB bring-up installs there. # # No TLS, so the gateway answers on its address over plain HTTP. It # authenticates callers by API key, against the Secret below. @@ -20,7 +20,7 @@ spec: matchLabels: modelplane.ai/inference-keys: "true" --- -# One caller, identified as e2e. run.sh sends its key in Authorization for +# One caller, identified as e2e. The tests send its key in Authorization for # OpenAI requests and in x-api-key for Anthropic ones. apiVersion: v1 kind: Secret diff --git a/e2e/manifests/20-inference-class.yaml b/e2e/manifests/20-inference-class.yaml index 146266dc9..d86864804 100644 --- a/e2e/manifests/20-inference-class.yaml +++ b/e2e/manifests/20-inference-class.yaml @@ -1,5 +1,5 @@ # A fake-GPU class backed by kubernetes-sigs/dra-example-driver (installed on the -# workload cluster by run.sh). claim: DRA is required — the fleet scheduler +# workload cluster by bring-up). claim: DRA is required — the fleet scheduler # rejects an engine whose only device is Synthetic, because it needs a claimable # device to bind a ResourceClaim through (see scheduling.py). The example driver # publishes fake gpu.example.com devices (memory 80Gi) with no real hardware, so diff --git a/e2e/manifests/30-inference-cluster.yaml b/e2e/manifests/30-inference-cluster.yaml index d44d9f1f0..1c9c16e49 100644 --- a/e2e/manifests/30-inference-cluster.yaml +++ b/e2e/manifests/30-inference-cluster.yaml @@ -1,4 +1,4 @@ -# Registers the separate workload kind cluster via source: Existing. run.sh +# Registers the separate workload kind cluster via source: Existing. Bring-up # creates that cluster, builds the referenced kubeconfig Secret in # modelplane-system (--internal address, reachable from the control plane's # provider pods over the shared kind network), and labels its node diff --git a/e2e/run.sh b/e2e/run.sh deleted file mode 100644 index 04a050f65..000000000 --- a/e2e/run.sh +++ /dev/null @@ -1,555 +0,0 @@ -#!/usr/bin/env bash -# Two-cluster local e2e (no cloud, no GPU). Usually invoked via -# `nix run .#e2e` (which provides the tooling and the Nix-built function -# images). See README.md. -# -# Two clusters: -# - a workload kind cluster (this script creates it), registered via -# source: Existing, where the serving stack, the InferenceGateway and the -# model run; -# - a control-plane cluster (crossplane project run manages it) with crossplane -# + the config. -set -euo pipefail - -CP=modelplane-e2e-local -WL=modelplane-e2e-workload -# Pinned so the workload cluster has the DRA APIs the serving stack's NVIDIA DRA -# driver needs (resource.k8s.io, GA in k8s 1.34). The control-plane cluster that -# project run creates needs no DRA, so its image doesn't matter here. -# v1.34.2 or newer: older kubelets deadlock on an idle DRA connection (k/k#133934). -WL_NODE_IMAGE=kindest/node:v1.34.8@sha256:02722c2dedddcfc00febf5d27fbeb9b7b2c14294c82109ff4a85d89ac9ba3256 -METALLB_URL=https://raw.githubusercontent.com/metallb/metallb/v0.14.8/config/manifests/metallb-native.yaml -# Pinned by digest (a multi-arch manifest list) so a moving :latest can't flake -# the verify curl pod. -CURL_IMAGE=curlimages/curl@sha256:7c12af72ceb38b7432ab85e1a265cff6ae58e06f95539d539b654f2cfa64bb13 -ROOT="$(git rev-parse --show-toplevel)" - -log() { printf '\n\033[1;34m==> %s\033[0m\n' "$*"; } - -if [ "${1:-}" = "--clean" ]; then - # Always delete both clusters — don't gate kind delete on project stop's exit - # code (it can exit 0 without removing the cluster). project stop is - # best-effort for the local registry it also manages. - crossplane project stop --control-plane-name "$CP" 2>/dev/null || true - kind delete cluster --name "$CP" || true - kind delete cluster --name "$WL" || true - docker rm -f "${CP}-registry" >/dev/null 2>&1 || true - exit 0 -fi - -# One temp dir for everything this run creates (the isolated Docker config), -# removed on exit. mktemp -d gives a fresh unique path and the trap captures it -# on the next line, so the rm -rf can never reach a real directory. -work="$(mktemp -d)" -trap 'rm -rf "$work"' EXIT - -WLCTX="kind-$WL" -if kind get clusters 2>/dev/null | grep -qx "$WL"; then - # Reuse an existing workload cluster only if it's the pinned version. An - # older one lacks the DRA APIs and would fail the run confusingly later. - ver="$(kubectl --context "$WLCTX" get nodes -o jsonpath='{.items[0].status.nodeInfo.kubeletVersion}' 2>/dev/null || true)" - case "$ver" in - v1.34.0 | v1.34.1) - echo "workload cluster $WL is $ver, whose kubelet deadlocks on an idle DRA connection (fixed in v1.34.2); recreate it with: nix run .#e2e -- --clean" >&2 - exit 1 - ;; - v1.34.*) log "Reusing workload cluster $WL ($ver)" ;; - *) - echo "workload cluster $WL is ${ver:-unreachable}, but v1.34 is required for the DRA APIs; recreate it with: nix run .#e2e -- --clean" >&2 - exit 1 - ;; - esac -else - log "Creating workload cluster $WL (k8s v1.34, for DRA)" - kind create cluster --name "$WL" --image "$WL_NODE_IMAGE" -fi - -# Both kind clusters share one Docker network; MetalLB hands out LoadBalancer IPs -# from it, and the control plane must ROUTE to the workload gateways' IPs across -# it. So the pool must sit inside the *actual* kind subnet — normally -# 172.18.0.0/16, but kind bumps to 172.19/172.20/... when earlier Docker networks -# already hold 172.18. Detect it and derive the pool from its prefix; a hardcoded -# 172.18 leaves the LB IPs off-subnet and silently breaks cross-cluster routing -# (curl times out). -# `|| true` so a detection miss (grep finds nothing) doesn't trip set -e here — -# the explicit check below then reports it instead of an opaque abort. -SUBNET="$(docker network inspect kind -f '{{range .IPAM.Config}}{{println .Subnet}}{{end}}' | grep -E '^[0-9]+\.' | head -1 || true)" -PREFIX="$(printf '%s' "$SUBNET" | cut -d. -f1-2)" -[ -n "$PREFIX" ] || { - echo "could not detect the kind Docker subnet" >&2 - exit 1 -} -log "kind Docker subnet ${SUBNET} -> MetalLB pool ${PREFIX}.255.x" - -# The serving stack doesn't install MetalLB, so the workload cluster needs it -# here. Both gateways live on this cluster: the cluster gateway fronting the -# engine pods, and the InferenceGateway callers reach. The pool has to be big -# enough for two LoadBalancer Services. -log "Installing MetalLB on the workload cluster (pool ${PREFIX}.255.100-.149)" -kubectl --context "$WLCTX" apply -f "$METALLB_URL" -kubectl --context "$WLCTX" -n metallb-system rollout status deploy/controller --timeout=180s -kubectl --context "$WLCTX" apply -f - <"$docker_config/config.json" -export DOCKER_CONFIG="$docker_config" - -# Flags pick the mode: -# --no-apply install and finish the control plane but skip the model -# manifests — for gradual, manual apply/debugging. -# --verify after apply, wait for the ModelService and assert a live 200, -# exiting non-zero on failure. This is exactly what CI runs, so -# running it locally gives the same pass/fail signal (dev/CI parity). -manifests="$ROOT/e2e/manifests" -cpctx="kind-$CP" -apply_manifests=1 -verify=0 -case "${1:-}" in ---no-apply) apply_manifests=0 ;; ---verify) verify=1 ;; -esac - -log "Building + running the control plane" -cd "$ROOT" -# Install the config with the lean control-plane's narrowed MRAP applied before -# the providers, so the cloud providers stay dormant (safe-start scales them to -# zero). prerequisites.yaml is applied afterwards with kubectl, not through -# --init-resources: it opens with a comment-only YAML document that `crossplane -# project run` rejects but kubectl skips. -crossplane project run \ - --control-plane-name "$CP" --cluster-admin --timeout 25m \ - --init-resources "$ROOT/e2e/lean-control-plane.yaml" \ - --crossplane-version=2.4.0 - -# Config healthy. Finish the setup the install guide does by hand (as the -# nix run app now does too, PR #375): apply the RBAC prerequisites, then point -# the two providers at the DeploymentRuntimeConfigs they define. Providers -# install before prerequisites.yaml, and an ImageConfig binds only at -# ProviderRevision creation, so provider-helm otherwise comes up without the -# granted RBAC and provider-kubernetes without --sanitize-secrets. -log "Finishing control-plane setup: prerequisites + provider runtime configs" -kubectl --context "$cpctx" apply -f "$ROOT/docs/manifests/install/prerequisites.yaml" -kubectl --context "$cpctx" patch provider.pkg.crossplane.io upbound-provider-helm --type merge \ - -p '{"spec":{"runtimeConfigRef":{"apiVersion":"pkg.crossplane.io/v1beta1","kind":"DeploymentRuntimeConfig","name":"provider-helm-modelplane"}}}' -kubectl --context "$cpctx" patch provider.pkg.crossplane.io upbound-provider-kubernetes --type merge \ - -p '{"spec":{"runtimeConfigRef":{"apiVersion":"pkg.crossplane.io/v1beta1","kind":"DeploymentRuntimeConfig","name":"provider-kubernetes-modelplane"}}}' - -# The InferenceCluster (source: Existing) reads this kubeconfig to reach the -# workload cluster; --internal gives an address routable from the control plane's -# provider pods. It lives in modelplane-system, created by prerequisites.yaml above. -{ - printf 'apiVersion: v1\nkind: Secret\nmetadata: {name: local-cluster-kubeconfig, namespace: modelplane-system}\nstringData:\n kubeconfig: |\n' - kind get kubeconfig --internal --name "$WL" | sed 's/^/ /' -} | kubectl --context "$cpctx" apply -f - - -if [ "$apply_manifests" = 0 ]; then - log "--no-apply: control plane ready; apply manifests from $manifests" - exit 0 -fi - -# RBAC is in place, so the compositions can reach the workload cluster. Apply the -# model manifests. -kubectl --context "$cpctx" apply -f "$manifests/" - -if [ "$verify" = 0 ]; then - log "Done. Curl the ModelService per the README; clean up with: nix run .#e2e -- --clean" - exit 0 -fi - -# --verify: project run returns once the config is healthy and the resources are -# applied, so the serving-stack install and model rollout are still reconciling. -# Wait for the ModelService to report RoutingReady, then route a real request to -# the engine and assert a 200. Any failure exits non-zero — that is what makes -# this usable as a CI gate. -log "Verifying the model serves end to end" -ns=ml-team -svc=mock - -# Wait for the ModelService to report RoutingReady, which means its route is -# composed and applied on every gateway serving it. status.model and the -# gateway's endpoints both publish long before that - neither depends on a -# replica existing - so gating on either would start curling while the engine is -# still rolling out. -ready="" -for _ in $(seq 1 80); do - ready="$(kubectl --context "$cpctx" -n "$ns" get modelservice "$svc" \ - -o jsonpath='{.status.conditions[?(@.type=="RoutingReady")].status}' 2>/dev/null || true)" - [ "$ready" = "True" ] && break - sleep 15 -done -[ "$ready" = "True" ] || { - echo "verify: ModelService $ns/$svc never became RoutingReady" >&2 - kubectl --context "$cpctx" -n "$ns" get modelservice "$svc" -o jsonpath='{range .status.conditions[*]}{.type}={.status} {.reason}: {.message}{"\n"}{end}' >&2 || true - kubectl --context "$cpctx" -n "$ns" get modelendpoint -o wide >&2 || true - kubectl --context "$cpctx" -n "$ns" get modelreplica -o wide >&2 || true - exit 1 -} - -# AI Gateway rolls the gateway's proxy pods once the first route reaches it, to -# stamp them with the hash of its sidecar's config, so a fresh gateway is still -# replacing its pods when the route goes ready. Requests during that rollout -# can fail, so wait for it to finish before asserting anything. The rollout -# starts only once AI Gateway has seen the route, so first wait for the stamp. -proxy=gateway.envoyproxy.io/owning-gateway-name=inference-gateway -stamp="" -for _ in $(seq 1 40); do - stamp="$(kubectl --context "$WLCTX" -n envoy-gateway-system get deploy -l "$proxy" \ - -o jsonpath='{.items[0].spec.template.metadata.annotations.aigateway\.envoyproxy\.io/extproc-config-hash}' 2>/dev/null || true)" - [ -n "$stamp" ] && break - sleep 3 -done -[ -n "$stamp" ] || { - echo "verify: AI Gateway never stamped the InferenceGateway's proxy pods" >&2 - exit 1 -} -kubectl --context "$WLCTX" -n envoy-gateway-system rollout status deploy -l "$proxy" --timeout=5m || { - echo "verify: the InferenceGateway's proxy pods never finished rolling out" >&2 - kubectl --context "$WLCTX" -n envoy-gateway-system get pods -l "$proxy" -o wide >&2 || true - exit 1 -} - -# A caller names a ModelService as the request's model, so read it from status. -model="$(kubectl --context "$cpctx" -n "$ns" get modelservice "$svc" -o jsonpath='{.status.model}')" -[ -n "$model" ] || { - echo "verify: ModelService $ns/$svc published no model name" >&2 - exit 1 -} - -base="" -for _ in $(seq 1 80); do - base="$(kubectl --context "$cpctx" get inferencegateway local -o jsonpath='{.status.endpoints.openAI}' 2>/dev/null || true)" - [ -n "$base" ] && break - sleep 15 -done -[ -n "$base" ] || { - echo "verify: InferenceGateway local never published an OpenAI endpoint" >&2 - exit 1 -} -log "Gateway ${base}, model ${model}" - -# GET a URL from the workload cluster, reporting curl's own exit code rather -# than an HTTP status. Used to assert a request is refused before there is any -# HTTP response to report. -k skips server verification, so a non-zero exit is -# the server rejecting us rather than us rejecting its certificate. -wl_curl_exit() { - local pod="$1" url="$2" - kubectl --context "$WLCTX" -n default run "$pod" --restart=Never \ - --labels=app.kubernetes.io/name=e2e-verify --image="$CURL_IMAGE" \ - --command -- sh -c "curl -sS -k --max-time 15 -o /dev/null \"$url\"; echo EXIT=\$?" \ - >/dev/null 2>&1 || true - local c="" - for _ in $(seq 1 30); do - c="$(kubectl --context "$WLCTX" -n default logs "$pod" 2>/dev/null | sed -n 's/.*EXIT=\([0-9]*\).*/\1/p' || true)" - [ -n "$c" ] && break - sleep 2 - done - kubectl --context "$WLCTX" -n default delete pod "$pod" --now >/dev/null 2>&1 || true - printf '%s' "$c" -} - -# The address is on the kind Docker subnet the host can't route to on macOS, so -# curl from a pod on the control plane, reading the status from the pod's logs -# (not `run -i`, whose attach drops output on a headless runner). curl_status -# runs one throwaway pod per call and echoes the HTTP code; it polls the logs -# (curl writes the code once, then exits) so a failed attempt costs seconds, and -# a unique pod name per call keeps retries from reading a prior pod's output. -curl_status() { - local pod="$1" url="$2" - shift 2 - kubectl --context "$cpctx" -n "$ns" run "$pod" --restart=Never \ - --labels=app.kubernetes.io/name=e2e-verify --image="$CURL_IMAGE" \ - --command -- curl -sS --max-time 15 -o /dev/null -w '%{http_code}' "$url" "$@" \ - >/dev/null 2>&1 || true - local c="" - for _ in $(seq 1 30); do - c="$(kubectl --context "$cpctx" -n "$ns" logs "$pod" 2>/dev/null | tr -dc '0-9' || true)" - [ -n "$c" ] && break - sleep 2 - done - printf '%s' "$c" -} - -# curl_body is the same, but returns the response body. Used where the assertion -# is about what came back rather than only that something did. -curl_body() { - local pod="$1" url="$2" - shift 2 - kubectl --context "$cpctx" -n "$ns" run "$pod" --restart=Never \ - --labels=app.kubernetes.io/name=e2e-verify --image="$CURL_IMAGE" \ - --command -- curl -sS --max-time 15 "$url" "$@" >/dev/null 2>&1 || true - local b="" - for _ in $(seq 1 30); do - b="$(kubectl --context "$cpctx" -n "$ns" logs "$pod" 2>/dev/null || true)" - [ -n "$b" ] && break - sleep 2 - done - printf '%s' "$b" -} - -cleanup_verify_pods() { - kubectl --context "$cpctx" -n "$ns" delete pod -l app.kubernetes.io/name=e2e-verify --now >/dev/null 2>&1 || true -} - -# The gateway authenticates callers against the key in -# e2e/manifests/10-inference-gateway.yaml. OpenAI requests send it as a bearer -# token, Anthropic ones in x-api-key. -caller_key=sk-e2e-caller - -# OpenAI /v1/chat/completions, retried: the gateway can publish an endpoint a -# moment before the route is serving, and a slower CI runner widens that gap. -oai='{"model":"'"$model"'","messages":[{"role":"user","content":"ping"}]}' -code="" -for attempt in $(seq 1 10); do - code="$(curl_status "e2e-verify-oai-$attempt" "$base/chat/completions" -H "authorization: Bearer $caller_key" -H 'content-type: application/json' -d "$oai")" - log "verify attempt $attempt (OpenAI): HTTP ${code:-none}" - [ "$code" = "200" ] && break - sleep 10 -done -[ "$code" = "200" ] || { - echo "verify: $base/chat/completions did not return 200 within retries (last: ${code:-none})" >&2 - cleanup_verify_pods - exit 1 -} - -# The same request with no key, and with a key no Secret holds, is refused. -for nokey in none wrong; do - if [ "$nokey" = none ]; then - kcode="$(curl_status "e2e-verify-nokey-$nokey" "$base/chat/completions" -H 'content-type: application/json' -d "$oai")" - else - kcode="$(curl_status "e2e-verify-nokey-$nokey" "$base/chat/completions" -H 'authorization: Bearer sk-wrong' -H 'content-type: application/json' -d "$oai")" - fi - log "verify (key: ${nokey}): HTTP ${kcode:-none}" - [ "$kcode" = "401" ] || { - echo "verify: expected 401 for a request with key ${nokey}, got ${kcode:-none}" >&2 - cleanup_verify_pods - exit 1 - } -done - -# The engine only answers to the name Modelplane started it under, and rejects -# anything else with a 404. So a 200 above already proves the gateway rewrote the -# caller's ModelService name to the deployment's. Assert the response reports the -# served model rather than what the caller asked for, which is the visible half -# of the same mechanism. -body="$(curl_body e2e-verify-served "$base/chat/completions" -H "authorization: Bearer $caller_key" -H 'content-type: application/json' -d "$oai")" -case "$body" in -*'"model": "ml-team/mock-demo"'* | *'"model":"ml-team/mock-demo"'*) - log "verify (model rewriting): caller asked for ${model}, engine served ml-team/mock-demo" - ;; -*) - echo "verify: response did not report the served model; got: $body" >&2 - cleanup_verify_pods - exit 1 - ;; -esac - -# A model no ModelService claims must not route anywhere. Catches a route -# matching too broadly, which would send a caller to an arbitrary backend. -ncode="$(curl_status e2e-verify-unknown "$base/chat/completions" -H "authorization: Bearer $caller_key" \ - -H 'content-type: application/json' -d '{"model":"ml-team/nope","messages":[{"role":"user","content":"ping"}]}')" -log "verify (unknown model): HTTP ${ncode:-none}" -[ "$ncode" = "200" ] && { - echo "verify: an unclaimed model name was routed and served" >&2 - cleanup_verify_pods - exit 1 -} - -# The cluster gateway must refuse a caller that presents no client certificate. -# Every check above goes through the InferenceGateway, which holds a -# certificate, so none of them would notice this lapsing. A ClientTrafficPolicy -# that stopped applying, or an HTTP listener beside the HTTPS one, would leave -# the engines open to anything that can reach the load balancer. -# -# Run from the workload cluster, where the Service compose-inference-gateway -# composed resolves the gateway's name. A plain GET is enough: the handshake -# fails before any request is sent. The trailing dot skips the pod's search -# domains, which ndots:5 would otherwise try ahead of the name itself. A resolve -# failure returns curl 6, which the checks below reject rather than pass. -cluster_gw_name="$(kubectl --context "$cpctx" get inferencecluster local -o jsonpath='{.status.gateway.hostname}')" -cluster_gw="https://${cluster_gw_name}./v1/models" -ecode="$(wl_curl_exit e2e-verify-nocert "$cluster_gw")" -log "verify (cluster gateway, no client certificate): curl exit ${ecode:-none}" -case "$ecode" in -0) - echo "verify: the cluster gateway served a caller presenting no client certificate" >&2 - cleanup_verify_pods - exit 1 - ;; -35 | 52 | 55 | 56) ;; -*) - echo "verify: expected the cluster gateway to refuse an uncertified caller mid-handshake," >&2 - echo "verify: but curl failed with ${ecode:-no exit code}, which is a different failure" >&2 - cleanup_verify_pods - exit 1 - ;; -esac - -# And nothing on port 80. The serving HTTPRoutes carry no sectionName, so they -# attach to every listener there is, and an HTTP listener would serve the -# engines without a certificate. The gateway's only listener is HTTPS, and the -# load balancer publishes a port per listener, so the connection is refused. -hcode="$(wl_curl_exit e2e-verify-plaintext "http://${cluster_gw_name}./v1/models")" -log "verify (cluster gateway, plaintext): curl exit ${hcode:-none}" -case "$hcode" in -0) - echo "verify: the cluster gateway served plaintext HTTP on port 80" >&2 - cleanup_verify_pods - exit 1 - ;; -7 | 28 | 35 | 52 | 56) ;; -*) - echo "verify: expected no listener on port 80, but curl failed with ${hcode:-no exit code}," >&2 - echo "verify: which is a different failure" >&2 - cleanup_verify_pods - exit 1 - ;; -esac - -# /v1/models lists what this gateway serves. Only exact model matches appear, so -# this also proves the route matches exactly rather than by pattern. -models="$(curl_body e2e-verify-models "$base/models" -H "authorization: Bearer $caller_key")" -case "$models" in -*"$model"*) log "verify (/v1/models): lists ${model}" ;; -*) - echo "verify: /v1/models did not list $model; got: $models" >&2 - cleanup_verify_pods - exit 1 - ;; -esac - -# Anthropic's Messages API on the same gateway. The endpoint's API is OpenAI, so -# the gateway translates the request. The mock serves /v1/messages too, the way -# vLLM does, so a 200 alone doesn't tell translation from passthrough. The key -# goes in x-api-key, as Anthropic clients send it. -anthropic_base="${base%/v1}/anthropic/v1" -ant='{"model":"'"$model"'","max_tokens":16,"messages":[{"role":"user","content":"ping"}]}' -mcode="$(curl_status e2e-verify-anthropic "$anthropic_base/messages" -H "x-api-key: $caller_key" -H 'content-type: application/json' -H 'anthropic-version: 2023-06-01' -d "$ant")" -log "verify (Anthropic /v1/messages): HTTP ${mcode:-none}" -[ "$mcode" = "200" ] || { - echo "verify: $anthropic_base/messages did not return 200 (got: ${mcode:-none})" >&2 - cleanup_verify_pods - exit 1 -} - -# Assert the gateway emits a usage record attributing the request's tokens to -# its caller. Read it off the InferenceGateway's Envoy. It runs two proxy pods -# and each request lands on either, so read both. --tail has to be explicit, -# because with a selector kubectl logs keeps only the last 10 lines per pod. -usage="$(kubectl --context "$WLCTX" -n envoy-gateway-system logs \ - -l "$proxy" -c envoy --tail=200 2>/dev/null | - grep '"input_tokens":12' | tail -1 || true)" -# Check each field on its own. The access log serialises its keys -# alphabetically, so a single glob spanning two of them depends on that order. -# -# The endpoint is the ModelRoute's backend for it, in ml-team's mirrored -# namespace (child_name("mp", "ml-team")) and named after the route -# (child_name("mock", "local")) and the endpoint. -missing="" -for want in \ - '"caller":"e2e"' \ - '"service":"'"$model"'"' \ - '"endpoint":"mp-ml-team-51733/mock-local-' \ - '"served_model":"ml-team/mock-demo"' \ - '"input_tokens":12' \ - '"output_tokens":9' \ - '"total_tokens":21' \ - '"status":200'; do - case "$usage" in - *"$want"*) ;; - *) missing="$missing $want" ;; - esac -done -[ -z "$missing" ] || { - echo "verify: usage record missing:$missing" >&2 - echo "verify: record was: ${usage:-none}" >&2 - cleanup_verify_pods - exit 1 -} -log "verify (usage record): ${usage}" - -cleanup_verify_pods -log "End to end OK: ${base} authenticates callers, serves ${model} over OpenAI and Anthropic, rewrites the model, and meters it" - -# Telemetry. The TelemetryDestination went in with the rest of the manifests, so -# the collector composed while the model rolled out and there's nothing to wait -# for beyond the first scrape. Its debug sink prints what reached it to its own -# log, which is the whole path in one assertion: discovery found the engine by -# the labels compose-model-replica stamps, the built-in mappings renamed its -# series, the unit conversion ran, and the identity came off the pod. -log "Verifying the fleet's telemetry" -kubectl --context "$WLCTX" -n modelplane-system rollout status deploy/modelplane-collector --timeout=180s || { - echo "verify: the collector never rolled out on the workload cluster" >&2 - kubectl --context "$WLCTX" -n modelplane-system describe deploy/modelplane-collector >&2 || true - exit 1 -} - -# One log read per attempt, not one per assertion. The window is generous -# because the collector's config arrives by reconcile: on a fresh install it can -# roll out once against the destination and again once the MetricMappings land, -# and the restart the config change triggers starts its log over. In steady -# state the first read already has everything. -telemetry="" -for _ in $(seq 1 30); do - telemetry="$(kubectl --context "$WLCTX" -n modelplane-system logs deploy/modelplane-collector --tail=4000 2>/dev/null || true)" - case "$telemetry" in *modelplane_gpu_memory_used_bytes*) break ;; esac - sleep 10 -done - -missing="" -for want in \ - 'Name: modelplane_requests_waiting' \ - 'Name: modelplane_gpu_memory_used_bytes' \ - 'deployment: Str(mock-demo)' \ - 'engine: Str(mock)' \ - 'role: Str(Standalone)' \ - 'cluster: Str(local)'; do - case "$telemetry" in - *"$want"*) ;; - *) missing="$missing [$want]" ;; - esac -done - -# 1024 MiB as bytes. DCGM reports the framebuffer in MiB and the name says -# bytes, so a mapping that forgot the unit reads 1024 here instead. -case "$telemetry" in -*"Value: 1073741824"*) ;; -*) missing="$missing [DCGM_FI_DEV_FB_USED converted from MiB to bytes]" ;; -esac - -# Nothing the mappings didn't rename leaves a cluster, so the engine's own -# names must not appear downstream. -case "$telemetry" in -*"Name: vllm:num_requests_waiting"*) missing="$missing [vllm: names should not leave the cluster]" ;; -esac - -[ -z "$missing" ] || { - echo "verify: telemetry missing:$missing" >&2 - echo "$telemetry" | grep -E "Name: |-> (cluster|deployment|engine|role): |Value: " | tail -40 >&2 || true - exit 1 -} -log "Telemetry OK: the engine's series arrive renamed, converted, and attributed to its deployment" diff --git a/e2e/tests/conftest.py b/e2e/tests/conftest.py new file mode 100644 index 000000000..3347611e0 --- /dev/null +++ b/e2e/tests/conftest.py @@ -0,0 +1,238 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Fixtures the end-to-end tests share. See e2e/README.md. + +The tests run against clusters that are already up. `nix run .#e2e -- --verify` +brings them up first. +""" + +import collections.abc + +import pytest +from kubernetes import client +from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 +from models.ai.modelplane.inferencegateway import v1alpha1 as igv1alpha1 +from models.ai.modelplane.modelservice import v1alpha1 as msv1alpha1 + +from e2e import environment, gateway, kube, wait + +# Where the curl pod the tests send requests from runs, on each cluster. +CLIENT_NAMESPACE = "e2e" +CLIENT_POD = "curl" + + +@pytest.fixture(scope="session") +def control_plane() -> kube.Cluster: + """The control plane, running Crossplane and Modelplane.""" + return kube.Cluster(environment.CONTROL_PLANE_CONTEXT) + + +@pytest.fixture(scope="session") +def workload() -> kube.Cluster: + """The workload cluster, running the serving stack, both gateways and the model.""" + return kube.Cluster(environment.WORKLOAD_CONTEXT) + + +@pytest.fixture(scope="session") +def control_plane_client(control_plane: kube.Cluster) -> collections.abc.Iterator[gateway.Client]: + """A pod on the control plane, which sends requests across clusters to the InferenceGateway.""" + yield from curl_pod(control_plane) + + +@pytest.fixture(scope="session") +def workload_client(workload: kube.Cluster) -> collections.abc.Iterator[gateway.Client]: + """A pod on the workload cluster, which can resolve the cluster gateway's Service name.""" + yield from curl_pod(workload) + + +def curl_pod(cluster: kube.Cluster) -> collections.abc.Iterator[gateway.Client]: + """Start a curl pod on a cluster, and delete it afterwards.""" + # A run that was interrupted can leave the namespace behind. + delete_namespace(cluster) + try: + cluster.core.create_namespace( + client.V1Namespace(metadata=client.V1ObjectMeta(name=CLIENT_NAMESPACE)), + _request_timeout=kube.TIMEOUT_SECONDS, + ) + cluster.core.create_namespaced_pod( + CLIENT_NAMESPACE, + client.V1Pod( + metadata=client.V1ObjectMeta(name=CLIENT_POD), + spec=client.V1PodSpec( + containers=[ + client.V1Container( + name="curl", + # Pinned by digest (a multi-arch manifest list), so + # a moving tag can't flake the tests. + image="curlimages/curl@sha256:7c12af72ceb38b7432ab85e1a265cff6ae58e06f95539d539b654f2cfa64bb13", + command=["sleep", "infinity"], + ) + ], + # As PID 1, sleep ignores SIGTERM, so don't wait for it to + # exit. + termination_grace_period_seconds=0, + ), + ), + _request_timeout=kube.TIMEOUT_SECONDS, + ) + + def ready() -> None: + pod = cluster.core.read_namespaced_pod(CLIENT_POD, CLIENT_NAMESPACE, _request_timeout=kube.TIMEOUT_SECONDS) + conditions = pod.status.conditions or [] + assert any(c.type == "Ready" and c.status == "True" for c in conditions), f"pod is {pod.status.phase}" + + wait.until(ready, timeout=2 * 60, what=f"pod {CLIENT_NAMESPACE}/{CLIENT_POD} to be Ready", retry=kube.RETRY) + yield gateway.Client(cluster, CLIENT_NAMESPACE, CLIENT_POD) + finally: + delete_namespace(cluster) + + +def delete_namespace(cluster: kube.Cluster) -> None: + """Delete the curl pod's namespace if it exists, and wait for it to go. + + Waiting means a run that follows doesn't create the pod in a namespace + that's still terminating. + """ + + def gone() -> None: + try: + ns = cluster.core.read_namespace(CLIENT_NAMESPACE, _request_timeout=kube.TIMEOUT_SECONDS) + except client.ApiException as e: + if e.status == 404: + return + raise + msg = f"namespace {CLIENT_NAMESPACE} is {ns.status.phase}" + raise AssertionError(msg) + + try: + cluster.core.delete_namespace(CLIENT_NAMESPACE, _request_timeout=kube.TIMEOUT_SECONDS) + except client.ApiException as e: + if e.status != 404: + raise + wait.until(gone, timeout=2 * 60, what=f"namespace {CLIENT_NAMESPACE} to be deleted", retry=kube.RETRY) + + +@pytest.fixture(scope="session") +def routed(control_plane: kube.Cluster, workload: kube.Cluster) -> gateway.Serving: + """Wait for the InferenceGateway to route ModelService ml-team/mock. + + Bring-up returns once Modelplane is installed and the manifests are applied, + so the serving stack and the model are still reconciling. On an environment + that's already up, each wait returns at once. + """ + + # RoutingReady means the route is composed and applied on every gateway + # serving the ModelService. Its status.model and the gateway's endpoints + # both publish before that, without a replica, so waiting on either would + # start sending requests while the engine is still rolling out. + def routing_ready() -> msv1alpha1.ModelService: + obj = control_plane.modelplane("modelservices", "mock", "ml-team") + assert obj is not None, "ModelService ml-team/mock doesn't exist" + ms = msv1alpha1.ModelService.model_validate(obj) + conditions = (ms.status.conditions if ms.status else None) or [] + assert any(c.type == "RoutingReady" and c.status == "True" for c in conditions), ( + f"ModelService ml-team/mock isn't RoutingReady: {[(c.type, c.status, c.reason) for c in conditions]}" + ) + return ms + + ms = wait.until(routing_ready, timeout=20 * 60, what="ModelService ml-team/mock to route", retry=kube.RETRY) + assert ms.status is not None + assert ms.status.model, "ModelService ml-team/mock is RoutingReady but publishes no model name" + + # AI Gateway rolls the gateway's proxy pods once the first route reaches + # it, to stamp them with the hash of its sidecar's config, so a fresh + # gateway is still replacing its pods when the route goes ready. Requests + # can fail during that rollout, which starts only once AI Gateway has seen + # the route. So wait for the stamp, then for the rollout to finish. + def proxies_rolled_out() -> None: + deployments = workload.apps.list_namespaced_deployment( + gateway.PROXY_NAMESPACE, label_selector=gateway.PROXY_SELECTOR, _request_timeout=kube.TIMEOUT_SECONDS + ).items + assert deployments, "the InferenceGateway has no proxy Deployment" + for d in deployments: + annotations = d.spec.template.metadata.annotations or {} + assert "aigateway.envoyproxy.io/extproc-config-hash" in annotations, ( + f"AI Gateway hasn't stamped Deployment {d.metadata.name}'s pods" + ) + kube.rolled_out(d) + + wait.until( + proxies_rolled_out, timeout=5 * 60, what="the InferenceGateway's proxy pods to roll out", retry=kube.RETRY + ) + + def endpoints_published() -> igv1alpha1.Endpoints: + obj = control_plane.modelplane("inferencegateways", "local", None) + assert obj is not None, "InferenceGateway local doesn't exist" + ig = igv1alpha1.InferenceGateway.model_validate(obj) + assert ig.status is not None + assert ig.status.endpoints is not None, "InferenceGateway local publishes no endpoints" + return ig.status.endpoints + + endpoints = wait.until( + endpoints_published, timeout=5 * 60, what="InferenceGateway local to publish its endpoints", retry=kube.RETRY + ) + assert endpoints.openAI is not None, "InferenceGateway local publishes no OpenAI endpoint" + assert endpoints.anthropic is not None, "InferenceGateway local publishes no Anthropic endpoint" + return gateway.Serving(model=ms.status.model, openai=endpoints.openAI, anthropic=endpoints.anthropic) + + +@pytest.fixture(scope="session") +def serving(routed: gateway.Serving, control_plane_client: gateway.Client) -> gateway.Serving: + """Wait for the InferenceGateway to serve ModelService ml-team/mock to caller e2e. + + The gateway can publish its endpoints a moment before the route serves, and + a slow CI runner widens that gap. Tests that expect a refusal use routed + instead, so a gateway that refuses everyone still fails only the tests that + need it to serve. + """ + + def serves() -> None: + r = control_plane_client.request( + f"{routed.openai}/chat/completions", + gateway.BEARER, + {"model": routed.model, "messages": [{"role": "user", "content": "ping"}]}, + ) + assert r.status == 200, r + + wait.until(serves, timeout=2 * 60, what=f"the InferenceGateway to serve {routed.model}", retry=kube.RETRY) + return routed + + +@pytest.fixture(scope="session") +def cluster_gateway(control_plane: kube.Cluster, workload_client: gateway.Client) -> str: + """Wait for the gateway fronting the engines on the workload cluster to answer, and return its hostname.""" + + def published() -> str: + obj = control_plane.modelplane("inferenceclusters", "local", None) + assert obj is not None, "InferenceCluster local doesn't exist" + ic = icv1alpha1.InferenceCluster.model_validate(obj) + assert ic.status is not None + assert ic.status.gateway is not None, "InferenceCluster local publishes no gateway" + assert ic.status.gateway.hostname is not None, "InferenceCluster local publishes no gateway hostname" + return ic.status.gateway.hostname + + hostname = wait.until( + published, timeout=20 * 60, what="InferenceCluster local to publish its gateway hostname", retry=kube.RETRY + ) + + # The hostname publishes before the gateway listens. Until it does, a + # connection fails to resolve (curl 6), is refused (7), or times out (28), + # which the tests would take for the gateway refusing it. + def answers() -> None: + code = workload_client.connect(f"https://{hostname}./v1/models") + assert code not in {6, 7, 28}, f"the cluster gateway isn't answering yet (curl exit {code})" + + wait.until(answers, timeout=10 * 60, what="the cluster gateway to answer", retry=kube.RETRY) + return hostname diff --git a/e2e/tests/test_auth.py b/e2e/tests/test_auth.py new file mode 100644 index 000000000..110a89f91 --- /dev/null +++ b/e2e/tests/test_auth.py @@ -0,0 +1,60 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Tests that the InferenceGateway authenticates its callers.""" + +import dataclasses + +import pytest + +from e2e import gateway + + +@dataclasses.dataclass +class Case: + """A request the InferenceGateway refuses.""" + + name: str + reason: str + headers: dict[str, str] + want: int + + +REFUSED_CASES = [ + Case( + name="NoKey", + reason="The InferenceGateway refuses a caller that presents no key.", + headers={}, + want=401, + ), + Case( + name="UnknownKey", + reason="The InferenceGateway refuses a caller whose key no Secret holds.", + headers={"authorization": "Bearer sk-wrong"}, + want=401, + ), +] + + +# These use routed rather than serving, so a gateway that refuses every caller +# still passes them, and fails only the tests that need it to serve. +@pytest.mark.parametrize("case", REFUSED_CASES, ids=lambda case: case.name) +def test_refused(case: Case, routed: gateway.Serving, control_plane_client: gateway.Client) -> None: + """The InferenceGateway refuses a chat completion without a valid key.""" + r = control_plane_client.request( + f"{routed.openai}/chat/completions", + case.headers, + {"model": routed.model, "messages": [{"role": "user", "content": "ping"}]}, + ) + assert r.status == case.want, case.reason diff --git a/e2e/tests/test_cluster_gateway.py b/e2e/tests/test_cluster_gateway.py new file mode 100644 index 000000000..8c9e24c11 --- /dev/null +++ b/e2e/tests/test_cluster_gateway.py @@ -0,0 +1,70 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Tests that the cluster gateway fronting the engines refuses callers without a client certificate. + +The other tests that send requests go through the InferenceGateway, which holds +a certificate, so none of them would notice this lapsing. A ClientTrafficPolicy that stopped +applying, or an HTTP listener beside the HTTPS one, would leave the engines open +to anything that can reach the load balancer. +""" + +import dataclasses + +import pytest + +from e2e import gateway + + +@dataclasses.dataclass +class Case: + """A connection the cluster gateway refuses.""" + + name: str + reason: str + scheme: str + # The curl exit codes that mean the gateway refused the connection. Any + # other code, such as an unresolved name, is a different failure. + want: set[int] + + +REFUSED_CASES = [ + # 35: the TLS handshake failed. 52: an empty reply. 55 and 56: the + # connection broke mid-handshake. + Case( + name="NoClientCertificate", + reason="The cluster gateway refuses a caller that presents no client certificate, during the handshake.", + scheme="https", + want={35, 52, 55, 56}, + ), + # The serving HTTPRoutes carry no sectionName, so they attach to every + # listener there is, and an HTTP listener would serve the engines without a + # certificate. The gateway's only listener is HTTPS, and the load balancer + # publishes a port per listener. 7: the connection was refused. 28: it timed + # out. 52 and 56: something answered port 80 without serving HTTP. + Case( + name="Plaintext", + reason="The cluster gateway serves nothing over plain HTTP on port 80.", + scheme="http", + want={7, 28, 52, 56}, + ), +] + + +@pytest.mark.parametrize("case", REFUSED_CASES, ids=lambda case: case.name) +def test_refused(case: Case, workload_client: gateway.Client, cluster_gateway: str) -> None: + """The cluster gateway refuses a connection that carries no client certificate.""" + # The trailing dot skips the pod's search domains, which ndots:5 would + # otherwise try first. + assert workload_client.connect(f"{case.scheme}://{cluster_gateway}./v1/models") in case.want, case.reason diff --git a/e2e/tests/test_metering.py b/e2e/tests/test_metering.py new file mode 100644 index 000000000..3d03fafa9 --- /dev/null +++ b/e2e/tests/test_metering.py @@ -0,0 +1,54 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Tests that the InferenceGateway meters what its callers use.""" + +from e2e import gateway, kube, wait + + +def test_usage_record(serving: gateway.Serving, control_plane_client: gateway.Client, workload: kube.Cluster) -> None: + """The InferenceGateway's access log attributes a request's tokens to its caller. + + The mock engine reports the same tokens for every request, so every request + the tests send logs an identical record. This counts the matching records + before sending a request, and waits for the count to grow. + """ + # The endpoint is the ModelRoute's backend for the ModelEndpoint, in + # ml-team's mirrored namespace on the workload cluster. + want = { + "caller": "e2e", + "service": "ml-team/mock", + "endpoint": "mp-ml-team-51733/mock-local-934fc-mock-demo-da96c-f8a13", + "served_model": "ml-team/mock-demo", + "input_tokens": 12, + "output_tokens": 9, + "total_tokens": 21, + "status": 200, + } + + def matching() -> int: + return sum(1 for record in gateway.usage_records(workload) if {k: record.get(k) for k in want} == want) + + before = matching() + r = control_plane_client.request( + f"{serving.openai}/chat/completions", + gateway.BEARER, + {"model": serving.model, "messages": [{"role": "user", "content": "ping"}]}, + ) + assert r.status == 200 + + def logged() -> None: + assert matching() > before, f"no new usage record matching {want}" + + wait.until(logged, timeout=60, what="the InferenceGateway to log the request", retry=kube.RETRY) diff --git a/e2e/tests/test_routing.py b/e2e/tests/test_routing.py new file mode 100644 index 000000000..cdf98b58a --- /dev/null +++ b/e2e/tests/test_routing.py @@ -0,0 +1,86 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Tests that the InferenceGateway routes requests to ModelService ml-team/mock.""" + +from e2e import gateway + + +def test_openai(serving: gateway.Serving, control_plane_client: gateway.Client) -> None: + """An OpenAI chat completion naming the ModelService returns 200.""" + r = control_plane_client.request( + f"{serving.openai}/chat/completions", + gateway.BEARER, + {"model": serving.model, "messages": [{"role": "user", "content": "ping"}]}, + ) + assert r.status == 200 + + +def test_anthropic(serving: gateway.Serving, control_plane_client: gateway.Client) -> None: + """An Anthropic Messages API request naming the ModelService returns 200. + + The endpoint's API is OpenAI, so the gateway translates the request. The + mock serves /v1/messages too, the way vLLM does, so a 200 alone doesn't + tell translation from passthrough. Anthropic clients send the key in + x-api-key. + """ + r = control_plane_client.request( + f"{serving.anthropic}/messages", + {"x-api-key": gateway.CALLER_KEY, "anthropic-version": "2023-06-01"}, + {"model": serving.model, "max_tokens": 16, "messages": [{"role": "user", "content": "ping"}]}, + ) + assert r.status == 200 + + +def test_served_model(serving: gateway.Serving, control_plane_client: gateway.Client) -> None: + """The response names the model the engine serves, not the ModelService the caller asked for. + + The engine answers only to the name Modelplane started it under, and refuses + anything else with a 404. So a 200 already shows the gateway rewrote the + caller's ModelService name to the deployment's. This asserts the visible + half of the same mechanism. + """ + r = control_plane_client.request( + f"{serving.openai}/chat/completions", + gateway.BEARER, + {"model": serving.model, "messages": [{"role": "user", "content": "ping"}]}, + ) + assert r.status == 200 + assert r.json()["model"] == "ml-team/mock-demo" + + +def test_unclaimed(serving: gateway.Serving, control_plane_client: gateway.Client) -> None: + """A model no ModelService claims routes nowhere. + + This catches a route that matches too broadly, which would send a caller to + an arbitrary backend. A 404 only means that once the claimed name serves, so + this waits for it. + """ + r = control_plane_client.request( + f"{serving.openai}/chat/completions", + gateway.BEARER, + {"model": "ml-team/nope", "messages": [{"role": "user", "content": "ping"}]}, + ) + assert r.status == 404 + + +def test_models(serving: gateway.Serving, control_plane_client: gateway.Client) -> None: + """/v1/models lists the ModelService. + + It lists only models a route matches exactly, so this also shows the route + matches the name exactly rather than by pattern. + """ + r = control_plane_client.request(f"{serving.openai}/models", gateway.BEARER) + assert r.status == 200 + assert serving.model in [m["id"] for m in r.json()["data"]] diff --git a/e2e/tests/test_telemetry.py b/e2e/tests/test_telemetry.py new file mode 100644 index 000000000..bfc81784e --- /dev/null +++ b/e2e/tests/test_telemetry.py @@ -0,0 +1,171 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Tests that the fleet's telemetry reaches its destination renamed, converted and attributed. + +e2e/manifests/60-telemetry.yaml points a TelemetryDestination at the collector's +debug exporter, which prints what reached it to the collector's own log. So one +log covers the whole path: service discovery found the engine by the labels +compose-model-replica stamps on serving pods, the built-in MetricMappings +renamed its series, the unit conversion ran, and the identity came off the pod. +""" + +import dataclasses +import logging + +import pytest +from models.ai.modelplane.modeldeployment import v1alpha1 as mdv1alpha1 + +from e2e import kube, wait + +log = logging.getLogger(__name__) + +# The collector the serving stack composes on the workload cluster. +COLLECTOR_NAMESPACE = "modelplane-system" +COLLECTOR = "modelplane-collector" + + +@dataclasses.dataclass +class Case: + """A line the collector's debug exporter does or doesn't print.""" + + name: str + reason: str + line: str + want: bool + + +@pytest.fixture(scope="module") +def exported(control_plane: kube.Cluster, workload: kube.Cluster) -> str: + """Wait for the collector to export the engine's series, and return the collector's log. + + The wait is for the DCGM series, which the engine publishes alongside the + rest. If it never arrives, this returns the collector's log anyway, so each + test reports whether its own line is there. + """ + + # The collector scrapes the engine pod directly, so this waits for the + # engine rather than for anything to route to it. + def deployed() -> None: + obj = control_plane.modelplane("modeldeployments", "mock-demo", "ml-team") + assert obj is not None, "ModelDeployment ml-team/mock-demo doesn't exist" + md = mdv1alpha1.ModelDeployment.model_validate(obj) + conditions = (md.status.conditions if md.status else None) or [] + assert any(c.type == "Ready" and c.status == "True" for c in conditions), ( + f"ModelDeployment ml-team/mock-demo isn't Ready: {[(c.type, c.status, c.reason) for c in conditions]}" + ) + + wait.until(deployed, timeout=20 * 60, what="ModelDeployment ml-team/mock-demo to be Ready", retry=kube.RETRY) + + def collector_rolled_out() -> None: + d = workload.apps.read_namespaced_deployment( + COLLECTOR, COLLECTOR_NAMESPACE, _request_timeout=kube.TIMEOUT_SECONDS + ) + kube.rolled_out(d) + + wait.until(collector_rolled_out, timeout=3 * 60, what=f"Deployment {COLLECTOR} to roll out", retry=kube.RETRY) + + def collector_log() -> str: + d = workload.apps.read_namespaced_deployment( + COLLECTOR, COLLECTOR_NAMESPACE, _request_timeout=kube.TIMEOUT_SECONDS + ) + selector = ",".join(f"{k}={v}" for k, v in d.spec.selector.match_labels.items()) + pods = workload.core.list_namespaced_pod( + COLLECTOR_NAMESPACE, label_selector=selector, _request_timeout=kube.TIMEOUT_SECONDS + ) + return "\n".join( + workload.logs(p.metadata.name, COLLECTOR_NAMESPACE, "collector", tail_lines=4000) for p in pods.items + ) + + # The collector's config arrives by reconcile. On a fresh install it can + # roll out once against the destination and again once the MetricMappings + # land, and the restart the config change triggers starts its log over. In + # steady state the first read already has everything. + def scraped() -> str: + logs = collector_log() + assert "modelplane_gpu_memory_used_bytes" in logs, "the collector hasn't exported the engine's series yet" + return logs + + try: + logs = wait.until( + scraped, timeout=5 * 60, interval=10, what="the collector to export the engine's series", retry=kube.RETRY + ) + except AssertionError: + logs = collector_log() + # What a failing test needs to see, without the rest of the log. + summary = [line for line in logs.splitlines() if any(k in line for k in ("Name: ", "-> ", "Value: "))] + log.info("The collector exported:\n%s", "\n".join(summary[-60:])) + return logs + + +EXPORTED_CASES = [ + Case( + name="RequestsWaiting", + reason="The engine's vllm:num_requests_waiting arrives renamed to modelplane_requests_waiting.", + line="Name: modelplane_requests_waiting", + want=True, + ), + Case( + name="GPUMemoryUsed", + reason="DCGM_FI_DEV_FB_USED arrives renamed to modelplane_gpu_memory_used_bytes.", + line="Name: modelplane_gpu_memory_used_bytes", + want=True, + ), + # DCGM reports the framebuffer in MiB and the name says bytes, so a mapping + # that forgot the unit reads 1024 here instead. + Case( + name="MiBToBytes", + reason="The 1024 MiB of framebuffer DCGM reports arrives as 1073741824 bytes.", + line="Value: 1073741824", + want=True, + ), + Case( + name="Deployment", + reason="A series carries the ModelDeployment it belongs to.", + line="deployment: Str(mock-demo)", + want=True, + ), + Case( + name="Engine", + reason="A series carries the engine it belongs to.", + line="engine: Str(mock)", + want=True, + ), + Case( + name="Role", + reason="A series carries the role of the engine member it belongs to.", + line="role: Str(Standalone)", + want=True, + ), + Case( + name="Cluster", + reason="A series carries the InferenceCluster it came from.", + line="cluster: Str(local)", + want=True, + ), + # Nothing the mappings didn't rename leaves a cluster. + Case( + name="EngineNames", + reason="The engine's own vllm: names don't leave the cluster.", + line="Name: vllm:num_requests_waiting", + want=False, + ), +] + + +@pytest.mark.parametrize("case", EXPORTED_CASES, ids=lambda case: case.name) +def test_exported(case: Case, exported: str) -> None: + """The collector's debug exporter prints a line, or doesn't.""" + found = case.line in exported + assert found == case.want, case.reason diff --git a/e2e/wait.py b/e2e/wait.py new file mode 100644 index 000000000..cad098c71 --- /dev/null +++ b/e2e/wait.py @@ -0,0 +1,50 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Wait for an eventually consistent system to converge.""" + +import collections.abc +import logging +import time + +log = logging.getLogger(__name__) + + +def until[T]( + check: collections.abc.Callable[[], T], + *, + timeout: float, + what: str, + retry: tuple[type[Exception], ...] = (AssertionError,), + interval: float = 5, +) -> T: + """Call check until it stops raising one of the retry exceptions, and return what it returns. + + Once timeout seconds pass, re-raise check's last exception, so the failure + says what was still wrong rather than only that time ran out. + """ + log.info("Waiting up to %ds for %s", timeout, what) + start = time.monotonic() + while True: + try: + result = check() + except retry as e: + remaining = timeout - (time.monotonic() - start) + if remaining <= 0: + e.add_note(f"Still failing after waiting {timeout:.0f}s for {what}.") + raise + time.sleep(min(interval, remaining)) + continue + log.info("Done after %.0fs", time.monotonic() - start) + return result diff --git a/flake.nix b/flake.nix index 5af039a62..8a55ee838 100644 --- a/flake.nix +++ b/flake.nix @@ -185,7 +185,7 @@ dockerCredentialUp = pkgs.upbound; }; stop = apps.stop { inherit crossplane; }; - e2e = apps.e2e { inherit crossplane functionsPkg; }; + e2e = apps.e2e { inherit crossplane functionsPkg pythonSet; }; test = apps.test { inherit pythonSet functionNames; }; stacks = apps.stacks { inherit (pkgs) aicr; }; } diff --git a/nix/apps.nix b/nix/apps.nix index 5eb57c432..fa62f6d98 100644 --- a/nix/apps.nix +++ b/nix/apps.nix @@ -31,7 +31,7 @@ -ignore '**/*.toml' \ -ignore '**/*.yaml' \ -ignore '**/*.yml' \ - functions/ docs/utils/validate/ hack/ nix.sh + functions/ docs/utils/validate/ hack/ e2e/ nix.sh echo "Formatting and linting Nix..." statix fix . @@ -46,8 +46,8 @@ find . -name '*.sh' -type f -exec shellcheck {} + echo "Formatting and linting Python..." - ruff format functions/ docs/utils/validate/ hack/ - ruff check --fix functions/ docs/utils/validate/ hack/ + ruff format functions/ docs/utils/validate/ hack/ e2e/ + ruff check --fix functions/ docs/utils/validate/ hack/ e2e/ echo "Refreshing uv.lock..." uv lock @@ -149,8 +149,8 @@ esac done - # Pin Crossplane to the version e2e/run.sh uses: without a pin the - # CLI installs the latest release + # Pin Crossplane to the version e2e/environment.py uses: without a + # pin the CLI installs the latest release version_args=(--crossplane-version=2.4.0) for arg in "$@"; do case "$arg" in @@ -358,21 +358,36 @@ ); }; - # Run the two-cluster local end-to-end test: a workload - # kind cluster registered via source: Existing (serving stack + model) and a - # control-plane cluster (crossplane + the InferenceGateway). Two clusters - # because the control-plane and workload layers both install the Gateway API - # CRDs and collide on a single cluster. See e2e. Tear down with - # `nix run .#e2e -- --clean`. This app just materialises the Nix-built - # function images (as `run` does), then hands off to run.sh, which needs real - # orchestration (a second cluster, a cross-cluster kubeconfig) that - # `crossplane project run` flags can't express — kept a normal shell file so - # it stays shellcheck-clean rather than escaped nix strings. + # Run the two-cluster local end-to-end test: a workload kind cluster + # registered via source: Existing (serving stack + model) and a control-plane + # cluster (crossplane + the Configuration). Two clusters because the + # control-plane and workload layers both install the Gateway API CRDs and + # collide on a single cluster. See e2e/README.md. + # + # With no argument it brings the environment up and applies the manifests, + # --no-apply stops short of the manifests, --verify then runs the tests in + # e2e/tests/ (passing any further arguments to pytest), --test runs them + # against an environment that's already up, and --clean tears it all down. + # This app materialises the Nix-built function images (as `run` does) for + # crossplane project run to load. e2e = { crossplane, functionsPkg, + pythonSet, }: + let + # What e2e/ imports: pytest, the Kubernetes client, and the generated + # models it reads Modelplane's status with. pydantic is declared here + # rather than on crossplane-models, whose pyproject.toml the Crossplane + # CLI generates. + venv = pythonSet.mkVirtualEnv "modelplane-e2e-env" { + pytest = [ ]; + kubernetes = [ ]; + crossplane-models = [ ]; + pydantic = [ ]; + }; + in { type = "app"; meta.description = "Run the local two-cluster end-to-end test"; @@ -381,16 +396,11 @@ name = "modelplane-e2e"; runtimeInputs = [ crossplane + venv pkgs.coreutils - pkgs.gnused - pkgs.gnugrep - pkgs.gawk pkgs.kind pkgs.kubectl - pkgs.curl pkgs.docker-client - pkgs.git - pkgs.bash ]; inheritPath = false; text = '' @@ -398,7 +408,28 @@ rm -f _output/functions ln -s ${functionsPkg} _output/functions - exec bash e2e/run.sh "$@" + case "''${1:-}" in + "") exec python -m e2e.environment up ;; + --no-apply) exec python -m e2e.environment up --no-apply ;; + --clean) exec python -m e2e.environment down ;; + --verify) + shift + python -m e2e.environment up + exec python -m pytest e2e/tests \ + -o log_cli=true --log-cli-level=INFO "$@" + ;; + --test) + shift + exec python -m pytest e2e/tests \ + -o log_cli=true --log-cli-level=INFO "$@" + ;; + *) + echo "usage: nix run .#e2e -- [--no-apply |" \ + "--verify [pytest args...] | --test [pytest args...] |" \ + "--clean]" >&2 + exit 2 + ;; + esac ''; } ); diff --git a/nix/checks.nix b/nix/checks.nix index 61cb6e2b7..3bc640f09 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -114,6 +114,29 @@ in touch $out/.function-test-style-checked ''; + # Type-check the end-to-end tests with ty, against the packages the e2e app + # runs them with (see apps.nix). + ty-e2e = + let + venv = pythonSet.mkVirtualEnv "e2e-ty-env" { + pytest = [ ]; + kubernetes = [ ]; + crossplane-models = [ ]; + pydantic = [ ]; + }; + in + pkgs.runCommand "modelplane-ty-e2e" + { + nativeBuildInputs = [ pkgs.unstable.ty ]; + } + '' + cp -r ${self}/e2e e2e + cp ${self}/pyproject.toml pyproject.toml + ty check e2e --python ${venv} + mkdir -p $out + touch $out/.ty-passed + ''; + python = pkgs.runCommand "modelplane-python-checks" { @@ -123,8 +146,8 @@ in cp -r ${self} src chmod -R u+w src cd src - ruff format --check functions/ docs/utils/validate/ hack/ - ruff check functions/ docs/utils/validate/ hack/ + ruff format --check functions/ docs/utils/validate/ hack/ e2e/ + ruff check functions/ docs/utils/validate/ hack/ e2e/ mkdir -p $out touch $out/.python-checks-passed ''; @@ -159,11 +182,11 @@ in # Fail if any hand-written source file is missing its Apache 2.0 license # header. Scoped to the files we author: the composition functions, the docs - # manifest validator, and the scripts in hack/. Generated models under - # schemas/python carry their own codegen banner, and config (*.toml) and - # vendored upstream CRDs (*.yaml) are excluded. addlicense -check only reads, - # so it runs against the store path directly. Run 'nix run .#fix' to add any - # missing headers. + # manifest validator, the scripts in hack/, and the end-to-end tests. + # Generated models under schemas/python carry their own codegen banner, and + # config (*.toml) and vendored upstream CRDs (*.yaml) are excluded. + # addlicense -check only reads, so it runs against the store path directly. + # Run 'nix run .#fix' to add any missing headers. license = pkgs.runCommand "modelplane-license-check" { @@ -175,7 +198,7 @@ in -ignore '**/*.toml' \ -ignore '**/*.yaml' \ -ignore '**/*.yml' \ - functions/ docs/utils/validate/ hack/ nix.sh + functions/ docs/utils/validate/ hack/ e2e/ nix.sh mkdir -p $out touch $out/.license-check-passed ''; diff --git a/pyproject.toml b/pyproject.toml index b1bffcf3a..a0d46c5cc 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -28,6 +28,8 @@ dev = [ "pydantic>=2.0", # 9.0 reports unittest subtests natively. "pytest>=9.0", + # The end-to-end tests read the clusters through the Kubernetes API. + "kubernetes>=36.0", ] [tool.ruff] @@ -76,11 +78,13 @@ allow-star-arg-any = true [tool.ruff.lint.per-file-ignores] # Generated models are not under our control. "schemas/python/**" = ["ALL"] -# Tests use magic values, many parameters, long hardcoded resource dicts, and -# boolean fixture toggles passed positionally. -"**/tests/**" = ["E501", "PLR2004", "PLR0913", "FBT001"] +# Function tests use magic values, many parameters, long hardcoded resource +# dicts, and boolean fixture toggles passed positionally. +"functions/*/tests/**" = ["E501", "PLR2004", "PLR0913", "FBT001"] +# The end-to-end tests compare HTTP statuses and curl exit codes. +"e2e/**" = ["PLR2004"] # The pytest style rules are for tests. Elsewhere an assert guards an invariant. -"!**/tests/**" = ["PT"] +"!{**/tests/**,e2e/**}" = ["PT"] # fn.py uses gRPC's required PascalCase method name. "functions/*/function/fn.py" = ["N802"] # The docs manifest validator is a CLI script; print is its output. diff --git a/uv.lock b/uv.lock index 15f980c67..ca57061c9 100644 --- a/uv.lock +++ b/uv.lock @@ -31,6 +31,106 @@ members = [ "modelplane", ] +[[package]] +name = "aiohappyeyeballs" +version = "2.7.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/ce/f4/eec0465c2f67b2664688d0240b3212d5196fd89e741df67ddb81f8d35658/aiohappyeyeballs-2.7.1.tar.gz", hash = "sha256:065665c041c42a5938ed220bdcd7230f22527fbec085e1853d2402c8a3615d9d", size = 24757, upload-time = "2026-07-01T17:11:55.501Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/71/43/1947f06babed6b3f1d7f38b0c767f52df66bfb2bc10b468c4a7de9eceff2/aiohappyeyeballs-2.7.1-py3-none-any.whl", hash = "sha256:9243213661e29250eb41368e5daa826fc017156c3b8a11440826b2e3ed376472", size = 15038, upload-time = "2026-07-01T17:11:54.055Z" }, +] + +[[package]] +name = "aiohttp" +version = "3.14.4" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "aiohappyeyeballs" }, + { name = "aiosignal" }, + { name = "attrs" }, + { name = "frozenlist" }, + { name = "multidict" }, + { name = "propcache" }, + { name = "typing-extensions", marker = "python_full_version < '3.13'" }, + { name = "yarl" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/93/2f/6a91adaa2dc26877d6ed2f54c0370c8910f019db7d77c5c6a194611e93ea/aiohttp-3.14.4.tar.gz", hash = "sha256:831fc5bd39ec2517851e348f613ddb5447a47cf4b71cb09845af7ad7ed45d8f9", size = 8077839, upload-time = "2026-10-05T17:43:48.309Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/81/e3/998dc75634ecd2a24f33bab480b57313059d55bcadff8825a46b7c80a8af/aiohttp-3.14.4-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:eb0093ca0817019d539f0d27bdf98e823944e05bb7efd9d2337a45b92e4d1b06", size = 832892, upload-time = "2026-10-05T17:38:16.373Z" }, + { url = "https://files.pythonhosted.org/packages/58/c9/9bacc26397854440d169f70ae00ea51721f504611d3a64cf043643792c32/aiohttp-3.14.4-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:f482dc9c8309801b4259dd218609d3dae0a7bcdedeac56d90500eabaa0c23465", size = 559893, upload-time = "2026-10-05T17:38:18.74Z" }, + { url = "https://files.pythonhosted.org/packages/fa/f3/90c5856014328a8b1e9a753de794edea282182b776ff65f0602ad4ffa520/aiohttp-3.14.4-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:aa800bb7c1d00e166493931f2ddacb982db913292503dc2f4884db1f2bc5f105", size = 552850, upload-time = "2026-10-05T17:38:20.385Z" }, + { url = "https://files.pythonhosted.org/packages/9b/88/e71c29ce6daa78fcee0feb1992f0fdc408f09f42c3311b1f1c68c36f16ea/aiohttp-3.14.4-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d90e4dae71b26597f6d1fd12af7d092ae67551edf0e524b73e0687214b79ee33", size = 1928941, upload-time = "2026-10-05T17:38:22.218Z" }, + { url = "https://files.pythonhosted.org/packages/3a/4f/f0f3bebfa521a2ce5fd4302c937701cbc50cd6d908beaa597388639b0b98/aiohttp-3.14.4-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:5b681054b8c3b86aa6a0384b6ad7f5cb78dccbabab1f4a3a6793f98ad1378959", size = 1882092, upload-time = "2026-10-05T17:38:24.016Z" }, + { url = "https://files.pythonhosted.org/packages/b5/0d/5df041fd25dceb4f4cafbbee49bb2c1e2d54a91aedecaf03b77cbb404c44/aiohttp-3.14.4-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:6dfe80b41bcf8d80d2125fb99a171adeb86db2fcb35c1a68e0226acba0ba7f03", size = 1986051, upload-time = "2026-10-05T17:38:26.022Z" }, + { url = "https://files.pythonhosted.org/packages/1a/07/e95eddafffef7c3a74446f67ac4e477a49cec2c78f2ab6c86d1acfd90d2d/aiohttp-3.14.4-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:71e4a59c6a8c5a6ae8b0b696c32635913e8f526cc01448762eaf5a3c59df8a4f", size = 2107840, upload-time = "2026-10-05T17:38:28.112Z" }, + { url = "https://files.pythonhosted.org/packages/da/6c/1f83990fea249dcb10f48086726953546efc9199c96b6be7114532ac361f/aiohttp-3.14.4-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:f887c3b7fb3058fc1ccd917f52078540c18d938d12018cb9edd64b6811d20c57", size = 1930721, upload-time = "2026-10-05T17:38:30.046Z" }, + { url = "https://files.pythonhosted.org/packages/6f/d1/48184a186ea26df4f81dd466144dcab7f9cdd1e131bfee0f0e41278c6f0d/aiohttp-3.14.4-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:0da4525e9ba145a11617d2cd7e44fe1acaf471f9c72a0cad28e72879c1020fbb", size = 1774508, upload-time = "2026-10-05T17:38:31.8Z" }, + { url = "https://files.pythonhosted.org/packages/0f/ef/a9ec758c6d3850453838c31b56d1e4499eb47c80883753a12dcb203f9afa/aiohttp-3.14.4-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:f7e8d3c5caa44fdbee3eba1aa1417aa58c74c95d4089fe3c451911117fcbb2f3", size = 1902311, upload-time = "2026-10-05T17:38:33.791Z" }, + { url = "https://files.pythonhosted.org/packages/d6/49/9a2be54900f2dea6e63bbf9628fced604686a2e301130ccde0b1aa2e7e53/aiohttp-3.14.4-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:09092cf906c18c824b6b16881da5a89cc9cfa9b15344e496bcf880e9045bee33", size = 1912107, upload-time = "2026-10-05T17:38:35.614Z" }, + { url = "https://files.pythonhosted.org/packages/a2/2b/151254abfc0fe5cbb8e6c41ad04d52122db0af176fb31dbf3bc44e081bda/aiohttp-3.14.4-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:dfd144837e92264878bba6d3a7ffa8206668b21e87748740afd45d193e48c87e", size = 1960207, upload-time = "2026-10-05T17:38:37.322Z" }, + { url = "https://files.pythonhosted.org/packages/23/99/92c8dce336ac5784755a3144623528516f2e3b6b4871d02cad5a0642ae66/aiohttp-3.14.4-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:9413cffe4e0d654b9b524f9c99dde8cbdb37e0ed9a2c4edd2e63adbac6b374e8", size = 1760168, upload-time = "2026-10-05T17:38:39.214Z" }, + { url = "https://files.pythonhosted.org/packages/66/3e/f05fd8d1a9595730db8d522882eea76429b9be2ed5337609f7c6e3d9d49c/aiohttp-3.14.4-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:f50d719f97ba488dfb8e306cf8c8cc172ff0683ec0af53917b288229aa555ba5", size = 1979946, upload-time = "2026-10-05T17:38:40.927Z" }, + { url = "https://files.pythonhosted.org/packages/b9/14/8941aa73ffceabc2a0eb26878b4f1b2958fb7b5358f97cc49b1c00ed9258/aiohttp-3.14.4-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:68ed4824d96b7afec1f3a8d7fba190fe282718041c5772e20298003e1e3e5d66", size = 1923375, upload-time = "2026-10-05T17:38:43.184Z" }, + { url = "https://files.pythonhosted.org/packages/18/b5/ef275b63c8bfe0a21e7e8d3ef0e67c4745ed403f1894fe5205057cedd3b9/aiohttp-3.14.4-cp311-cp311-win32.whl", hash = "sha256:d7a41d1427136828e6b11d3f3fd194d4357d5205a5b610be7ae07b1ba52ab974", size = 488414, upload-time = "2026-10-05T17:38:44.978Z" }, + { url = "https://files.pythonhosted.org/packages/e7/0b/21bdd70a1e072222d9dbd96bed4fbed26f7c46c5a20af11b78d32fd049d6/aiohttp-3.14.4-cp311-cp311-win_amd64.whl", hash = "sha256:2efbdb87e79d596325c4eefaef0f4495d701145e882351310ebc538e854dc7a3", size = 512755, upload-time = "2026-10-05T17:38:47.036Z" }, + { url = "https://files.pythonhosted.org/packages/b7/4c/bb324a7d20a9598187bb028ca8d6bcf66c63589c8632420919dd576964ed/aiohttp-3.14.4-cp311-cp311-win_arm64.whl", hash = "sha256:ac6e6f90d9360e460c873f945f11dc8698db0a71fec6612823f1aaaab2c11faa", size = 492960, upload-time = "2026-10-05T17:38:49.006Z" }, + { url = "https://files.pythonhosted.org/packages/22/02/db5935d45347fa8ede56b777930740fb0e72bbb4473ddb6ff8541f6f2afa/aiohttp-3.14.4-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:55b0b335982b0db7117dff7f6a8e3aad43c4105f9e392ca04f9fc8d958a95cce", size = 817689, upload-time = "2026-10-05T17:38:50.69Z" }, + { url = "https://files.pythonhosted.org/packages/e6/56/f00aa46127e6342f019d567853eeeb932e04fb9b9424b5fb981ad68f4178/aiohttp-3.14.4-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:2d0d8e435582b0d53009dac99e26a3d2be932ed7c9dbfcfc70f3fce1cf943bd5", size = 549599, upload-time = "2026-10-05T17:38:52.525Z" }, + { url = "https://files.pythonhosted.org/packages/3f/bc/825c09b09962253a2103e4a88c04374ee939c0c8ea31c270daa0308efb07/aiohttp-3.14.4-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:bf734016932d68e1324cfb638cac7a0866a83933a33badf15e836a7e24efb8c0", size = 548132, upload-time = "2026-10-05T17:38:54.629Z" }, + { url = "https://files.pythonhosted.org/packages/40/80/8a1bc8036540fe7cb38219cc5c0aa67d32e9068165725e8199ea8200b08b/aiohttp-3.14.4-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:322e3d741b3133bcbb981a41d4f9b8ac3545c18de49c662c2b1f4e1c770b98ad", size = 1898485, upload-time = "2026-10-05T17:38:56.439Z" }, + { url = "https://files.pythonhosted.org/packages/3c/fe/65b464d8520ace0d65b4ca69bc9abe175f842c8ce8d1522d363d9669903f/aiohttp-3.14.4-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:15a3bb5f90a4e515071f3617a45dfe26560beb834a6b6e108cdf4bc9e9719ec3", size = 1875334, upload-time = "2026-10-05T17:38:58.536Z" }, + { url = "https://files.pythonhosted.org/packages/be/51/9c583be502720e927560a8414f4dfb7c7de41945e1a30ccf7003b93c6f85/aiohttp-3.14.4-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:6c044e52df1a466af82cad3518c92ea80cc4ddcfea86966ff083bb34bab29bf4", size = 1937772, upload-time = "2026-10-05T17:39:00.376Z" }, + { url = "https://files.pythonhosted.org/packages/a2/3d/b56512df7ea9cbb5f59099a6b375088dd578356a5a3fb27198ab82e19832/aiohttp-3.14.4-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:d613d7d51bf06fe9a5e3aab86ce06d835bb0ceb1a31ea9e8766df4a403ae62a6", size = 2066220, upload-time = "2026-10-05T17:39:02.54Z" }, + { url = "https://files.pythonhosted.org/packages/11/6c/a7a42701d9ea319140a3d44a24ec72a00b0b46d1b899b26d8a705e65bdd8/aiohttp-3.14.4-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:c5e93ab6e6330dce40f5b43996f1fc5f2543296d263a5e4b59ff09e73870b7d8", size = 1922895, upload-time = "2026-10-05T17:39:04.638Z" }, + { url = "https://files.pythonhosted.org/packages/77/04/f8337bd824ca3fe3968d34dc5037ace1c4872551c4621f8266239311d301/aiohttp-3.14.4-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:20134dc8e68c67e7478124d516449428684c0545b2d3faa575691c8bcbd56678", size = 1712589, upload-time = "2026-10-05T17:39:06.671Z" }, + { url = "https://files.pythonhosted.org/packages/fe/a6/d20e2c940b592d7c4a21690889dbff64fe0ca3facd9190696368694e1488/aiohttp-3.14.4-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:f0daef5ef012369dddc391c61f604111424aa9a96313a69fb899018b09d1b01e", size = 1862251, upload-time = "2026-10-05T17:39:08.523Z" }, + { url = "https://files.pythonhosted.org/packages/f5/e4/d50edd5ae5d024b380a8e892e333840bb22967a9eb9c8492108a403f678a/aiohttp-3.14.4-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:19290a108de0718b73bc69b86e788975c6a59855c671ebe2a98a07a6f968328f", size = 1895155, upload-time = "2026-10-05T17:39:10.305Z" }, + { url = "https://files.pythonhosted.org/packages/c3/d3/f36f2931f807ccf8437ec15e8f8bfaa4748d4792639401cc767a866abaa0/aiohttp-3.14.4-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:1494877625baeffab669f4a25a09ec4eea721b49e35363081a2a8e4231fe141f", size = 1906623, upload-time = "2026-10-05T17:39:12.411Z" }, + { url = "https://files.pythonhosted.org/packages/5d/0a/94821b711493142bf5e11c27b0d7502776778f7792c77ecc6647357d20c1/aiohttp-3.14.4-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:a779ce4bf099ce4ad2cdfed62f1ca536d435f53237c54ff40172fddada24d1b5", size = 1698932, upload-time = "2026-10-05T17:39:14.446Z" }, + { url = "https://files.pythonhosted.org/packages/d9/76/3d5a0d28362f0053c1c457024749911932481e8fd105fddf8305b1f93c81/aiohttp-3.14.4-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:0b4cec9fca876e2d4f4c6cb32b177390132d04618867689f43fefabf3f5c3980", size = 1934528, upload-time = "2026-10-05T17:39:16.354Z" }, + { url = "https://files.pythonhosted.org/packages/a7/51/9784a21dc05d2e0f6acaf864f16db53dab0fbd7b0753cef168601eacb012/aiohttp-3.14.4-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:217707313e23aedde9bea67cf8365c7b138b549ae179121ef3f476cf67e3173a", size = 1901984, upload-time = "2026-10-05T17:39:18.299Z" }, + { url = "https://files.pythonhosted.org/packages/44/4f/f98c88703e6d620243aa08473a946ca4e3756b6288690e38f0d9f9c1d007/aiohttp-3.14.4-cp312-cp312-win32.whl", hash = "sha256:18a9fb9a3f6e6e63f4d27f9adeef313a2d3c69a84d31170a844623824e722534", size = 480278, upload-time = "2026-10-05T17:39:20.206Z" }, + { url = "https://files.pythonhosted.org/packages/b7/bc/8555013f84b53e4fdde423131b641af50e822c6d7d9118b24c52b31f6900/aiohttp-3.14.4-cp312-cp312-win_amd64.whl", hash = "sha256:69dcb02d33dfe415d5ee342cc63b7d320efcaf4b761701c7bd7bfb91c715481a", size = 506317, upload-time = "2026-10-05T17:39:22.706Z" }, + { url = "https://files.pythonhosted.org/packages/21/3f/ca5bad7bfdaa7ddca053a044a1c7c0d4f3ad9d1f12f96dcab10bfb9c50e7/aiohttp-3.14.4-cp312-cp312-win_arm64.whl", hash = "sha256:d36b0263e7c2fbf1750b9f35d4e7e48b0442d8c45e4a89e3e74168821fb014bb", size = 487683, upload-time = "2026-10-05T17:39:24.482Z" }, + { url = "https://files.pythonhosted.org/packages/23/ec/7fee140da700ec3f5aa5e0fd2319f67be5e86de197e2384d27e5e6209e0e/aiohttp-3.14.4-cp313-cp313-android_24_arm64_v8a.whl", hash = "sha256:264b5a7568306590c44ae4a0237ad1715e35d447ea44ebf324e4c8fd9f78f67a", size = 539114, upload-time = "2026-10-05T17:39:26.322Z" }, + { url = "https://files.pythonhosted.org/packages/4e/04/930c01ce9e186e787046433d7dc5b45b188885f8406c14b58435b4431974/aiohttp-3.14.4-cp313-cp313-android_24_x86_64.whl", hash = "sha256:17e5d7c7775d8dd8e894c7ef6ebfa26509de36868a9a6c7c4fcfaf6a1bf43d60", size = 553514, upload-time = "2026-10-05T17:39:28.104Z" }, + { url = "https://files.pythonhosted.org/packages/2c/d7/983e3cea81c07e95f35d34da4d1e182b5c05489c49b7560b198a83f0da79/aiohttp-3.14.4-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:e91dc8fe9cbd052d16f7f1267e285cf8bbac4d2194bd5741f05ef201c77ed09e", size = 521381, upload-time = "2026-10-05T17:39:30.419Z" }, + { url = "https://files.pythonhosted.org/packages/0e/2d/80baed11659093883f51245755105557d3253e45bd0b693d70df0059d68b/aiohttp-3.14.4-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:8df6157d31703972aec9e3b93963f039ff06318301e30b007c3e941ff8a7ac42", size = 528347, upload-time = "2026-10-05T17:39:32.344Z" }, + { url = "https://files.pythonhosted.org/packages/c8/18/bfda255c311096c08dab26dc65f3d219f090df37dc5f5f6345f8c3aefc1f/aiohttp-3.14.4-cp313-cp313-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:220912b351549dc736c59d104dc443fa615753102bb778a4e98f86e4d94bff6a", size = 539914, upload-time = "2026-10-05T17:39:34.265Z" }, + { url = "https://files.pythonhosted.org/packages/51/fc/3b6ca6117c0c42481d719411fff7b5d3970ebd906543d3dc1ba40fe2462f/aiohttp-3.14.4-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:2efa42b39bb3f524d5eca3760639b584afdd1143a6b047fda29910ffb75e4993", size = 813997, upload-time = "2026-10-05T17:39:36.061Z" }, + { url = "https://files.pythonhosted.org/packages/b5/f6/86e3e02a9712aadc1ae0a1076bb2d1adf3d9addfe5542a28190c8f9b6d9b/aiohttp-3.14.4-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:7e7356059c8262d2c1fd95242e2b7b70fbc6eee3135bf470049cbec3e968b5b0", size = 548058, upload-time = "2026-10-05T17:39:38.109Z" }, + { url = "https://files.pythonhosted.org/packages/85/7e/6b9c2e271419d40ca76589613b3c7ab185fd3050a407bd2565b6aa46f695/aiohttp-3.14.4-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:fac247d0cc956d732df1211ac2d8a9999ec8dc1a92a90401de9d52e4ead546f7", size = 545933, upload-time = "2026-10-05T17:39:40.193Z" }, + { url = "https://files.pythonhosted.org/packages/14/fc/33d6e24a5f8ac32423230088c415e2a1c6efee23d902b62f20506631e40b/aiohttp-3.14.4-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:6e2eb0f8b03b4f2bd154ed8cf7c1a119a1ad04808c258bf58f90138c83b8c6d5", size = 1888265, upload-time = "2026-10-05T17:39:42.182Z" }, + { url = "https://files.pythonhosted.org/packages/8d/e8/2eca1d468d83a514621d1280f9992d08649191db2028c9236b4edfa7c723/aiohttp-3.14.4-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:7fbe827d27e0e369fd7bb88ceeaa5b46ea006e73e015b9dd77e3723a53162293", size = 1872300, upload-time = "2026-10-05T17:39:44.195Z" }, + { url = "https://files.pythonhosted.org/packages/d6/f6/5872983de71b0214acdfb86822f5cca295b39be4832ec589fa224aee0f6c/aiohttp-3.14.4-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:a70387146d03af9f047c629b71aa27d9594b3b31ec4c9c91e1a73d0d8bf38e0e", size = 1926213, upload-time = "2026-10-05T17:39:46.199Z" }, + { url = "https://files.pythonhosted.org/packages/ee/65/ebfefa11a546c33c717af63c67868408ecf0b33b23a9d75669af95485407/aiohttp-3.14.4-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:0d0934926fc65744e2fdd44ce68b2d79d5ee608a1e23f0596b35dd696519bb7c", size = 2056609, upload-time = "2026-10-05T17:39:48.221Z" }, + { url = "https://files.pythonhosted.org/packages/69/c0/70667e82ec041f4122a17afce711bbcf94a71bec5888b69a49aed8817be2/aiohttp-3.14.4-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:ff4b366510c733974adb7a4102c094e4a6308e6f369315a4cf14d2d0a619cf82", size = 1914740, upload-time = "2026-10-05T17:39:50.304Z" }, + { url = "https://files.pythonhosted.org/packages/bf/86/559e9dfea676206b20737d7ccb5b3716bff5e58d2a8d53682e5301724cd1/aiohttp-3.14.4-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:85fac7c99e0ac3dcbd6ae1c33774e0603bfc732ddb967a6fce4a731103c1888e", size = 1703390, upload-time = "2026-10-05T17:39:52.385Z" }, + { url = "https://files.pythonhosted.org/packages/fc/91/c98897d4cf23e120b2f3980042835cb34c86bbe3c653f3eaacf4bf986349/aiohttp-3.14.4-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:ef1e9c3b4a2300023a5dc865889ce532cfaf6798ba4ef78fbbefe738a9d23efa", size = 1852642, upload-time = "2026-10-05T17:39:54.391Z" }, + { url = "https://files.pythonhosted.org/packages/14/13/0258f352cb2c236f358a5b57d138e15fa2566e17accc0758d968dbf2ceb6/aiohttp-3.14.4-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:97e65916588be0f952b3307aaa8460c0771fa98dbea2dd27dabc9f5d84d53c3b", size = 1890616, upload-time = "2026-10-05T17:39:56.466Z" }, + { url = "https://files.pythonhosted.org/packages/64/3c/f43e294ce1a06d1234a97bd862c1aa31acfa1a0eea3786c22fd0dc1ea731/aiohttp-3.14.4-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:840775cb39a9424f9edf142133a1192263ac5bd79117ad2a16602cef6a0c1e9d", size = 1896638, upload-time = "2026-10-05T17:39:58.585Z" }, + { url = "https://files.pythonhosted.org/packages/9f/f7/ce1e3a85b73d05037d729d3d32da7d6959b9e433435e9abd46c2fb38410f/aiohttp-3.14.4-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:54f9256430ca040d58affbd551513de13a5cff7d065204fc52b9a1b0af09a049", size = 1693003, upload-time = "2026-10-05T17:40:00.698Z" }, + { url = "https://files.pythonhosted.org/packages/1d/72/45eb900d6a9ee046728f5589f3d07f5d78f049b5fb84f48881fa4174d5df/aiohttp-3.14.4-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:6112d5931bf12c776188cab81b82bbb17d86e2e1381f2ade6edc43e6b240f0cb", size = 1927730, upload-time = "2026-10-05T17:40:02.943Z" }, + { url = "https://files.pythonhosted.org/packages/b2/43/a83ff79dc8375953a1f00b808a21c82590f4e68304fecccdcc42a25fb7d8/aiohttp-3.14.4-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:facc6df39047c4732cbd885bb62e40df20885fcd709fc399e11f81b45ce0fff9", size = 1891177, upload-time = "2026-10-05T17:40:04.989Z" }, + { url = "https://files.pythonhosted.org/packages/5c/ba/fbdc41608d4e2fb6084d960a41b392fa82beb5b126c2466de549b5c67fc8/aiohttp-3.14.4-cp313-cp313-win32.whl", hash = "sha256:7968634fa3a967a2bc0b9aeba1d5a1955501dcd7d0de44b96f7ebfb662797735", size = 479943, upload-time = "2026-10-05T17:40:07.402Z" }, + { url = "https://files.pythonhosted.org/packages/35/51/d75629750e705832fd5cd52516a82943ed58b6e09c30e14830f5d87452fa/aiohttp-3.14.4-cp313-cp313-win_amd64.whl", hash = "sha256:c0a894b0265d139cd4c2c8bd4643822cc289a4a5331e21e810881bd000ffb2dc", size = 506209, upload-time = "2026-10-05T17:40:09.49Z" }, + { url = "https://files.pythonhosted.org/packages/02/64/66452d2ffcc13e5e5ee4ba4a968656fa4a98c29ed5ff0205d4fde02b2943/aiohttp-3.14.4-cp313-cp313-win_arm64.whl", hash = "sha256:da16c3037178e72589d91de59536d271dffc92c2be2b47c38a45538cba54fd29", size = 487163, upload-time = "2026-10-05T17:40:11.526Z" }, + { url = "https://files.pythonhosted.org/packages/2b/27/5e1f8446545be98e9416af59c396b1ed1f9f3294afb0a40ee3a709889273/aiohttp-3.14.4-py3-none-any.whl", hash = "sha256:5c6758ba62aea282c537179cfc8474a90f2b5f7b089cc5ff66d8920d86a9bfdd", size = 279487, upload-time = "2026-10-05T17:43:44.905Z" }, +] + +[[package]] +name = "aiosignal" +version = "1.4.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "frozenlist" }, + { name = "typing-extensions", marker = "python_full_version < '3.13'" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/61/62/06741b579156360248d1ec624842ad0edf697050bbaf7c3e46394e106ad1/aiosignal-1.4.0.tar.gz", hash = "sha256:f47eecd9468083c2029cc99945502cb7708b082c232f9aca65da147157b251c7", size = 25007, upload-time = "2025-07-03T22:54:43.528Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/fb/76/641ae371508676492379f16e2fa48f4e2c11741bd63c48be4b12a6b09cba/aiosignal-1.4.0-py3-none-any.whl", hash = "sha256:053243f8b92b990551949e63930a839ff0cf0b0ebbe0597b0f3fb19e1a0fe82e", size = 7490, upload-time = "2025-07-03T22:54:42.156Z" }, +] + [[package]] name = "annotated-types" version = "0.7.0" @@ -40,6 +140,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/78/b6/6307fbef88d9b5ee7421e68d78a9f162e0da4900bc5f5793f6d3d0e34fb8/annotated_types-0.7.0-py3-none-any.whl", hash = "sha256:1f02e8b43a8fbbc3f3e0d4f0f4bfc8131bcb4eebe8849b8e5c773f3a1c582a53", size = 13643, upload-time = "2024-05-20T21:33:24.1Z" }, ] +[[package]] +name = "attrs" +version = "26.1.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/9a/8e/82a0fe20a541c03148528be8cac2408564a6c9a0cc7e9171802bc1d26985/attrs-26.1.0.tar.gz", hash = "sha256:d03ceb89cb322a8fd706d4fb91940737b6642aa36998fe130a9bc96c985eff32", size = 952055, upload-time = "2026-03-19T14:22:25.026Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/64/b4/17d4b0b2a2dc85a6df63d1157e028ed19f90d4cd97c36717afef2bc2f395/attrs-26.1.0-py3-none-any.whl", hash = "sha256:c647aa4a12dfbad9333ca4e71fe62ddc36f4e63b2d260a37a8b83d2f043ac309", size = 67548, upload-time = "2026-03-19T14:22:23.645Z" }, +] + [[package]] name = "cel-python" version = "0.5.0" @@ -56,6 +165,93 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/1e/f8/38812adc3f787c2c2e8ba56f524185ed379656c10b40347a32796ba61c08/cel_python-0.5.0-py3-none-any.whl", hash = "sha256:d0f85008b89655c2bb18d797d2fa3f96f2ed80f4a3b43b0e8138c6646581e5f6", size = 84950, upload-time = "2026-01-31T19:07:11.821Z" }, ] +[[package]] +name = "certifi" +version = "2026.7.22" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/a3/c2/24167ea9858356b47a87a50d39908bfdb72ceeefe0041586e704e5376b3a/certifi-2026.7.22.tar.gz", hash = "sha256:741e2c3b351ddf169a738da9f2c048608ff7f2c5cc02f1ebc6b118bb090d5d55", size = 138112, upload-time = "2026-07-22T03:35:12.644Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/0b/a7/71ac2cff56fec219ed242bb11b8efb69fcc4bec75db06fb7bfe35de520e6/certifi-2026.7.22-py3-none-any.whl", hash = "sha256:62f22742b58a1a33014a2b6b706588a8d7e2a88ae7bd1a6ebe8c992928483775", size = 136983, upload-time = "2026-07-22T03:35:11.276Z" }, +] + +[[package]] +name = "charset-normalizer" +version = "3.5.2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/33/1c/f41d4e74c28ab327ff3acd36053f7ea506c55872d7a90b0fa71aa3ab0c89/charset_normalizer-3.5.2.tar.gz", hash = "sha256:39de2a259fc954455c57274dc94c79d5842774e1247a016aff30bc0efed0f4ef", size = 172659, upload-time = "2026-09-30T04:39:23.398Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/22/67/6a0b94a7960d5e1b5eacd2fb529f3fccc47db4644f7f0a7cfdcfc3be578a/charset_normalizer-3.5.2-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:3d21b8b13c7592db2ac5e544a6d83187b995257472b0c9e8351b6d507ae37ed6", size = 370642, upload-time = "2026-09-30T04:35:06.91Z" }, + { url = "https://files.pythonhosted.org/packages/fb/94/01009e13b94041599004edf32e56e382c24e570f60f79bab8efe45cfe1eb/charset_normalizer-3.5.2-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d760fe2a4d7c3b226cb9026d6a842868d52a7901bd98420e1baf14e80da85cf5", size = 259280, upload-time = "2026-09-30T04:35:08.448Z" }, + { url = "https://files.pythonhosted.org/packages/66/85/3b5358f60a13210f0b67d3755c168ef758701b021e655d88d4da28554467/charset_normalizer-3.5.2-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:c9790464842f85f437dbbb54417eda1e0e6bfc52dd8d22d6fd1c994b73b2dc74", size = 245335, upload-time = "2026-09-30T04:35:10.104Z" }, + { url = "https://files.pythonhosted.org/packages/74/75/77c1c479b09ecd751d1e767b251ea5c14d4d50ff757bf404afab2692f600/charset_normalizer-3.5.2-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:4685902cf26edf013ed7a3da0f426ebba7a00ebb9541386d835afbf002c11cab", size = 291707, upload-time = "2026-09-30T04:35:11.575Z" }, + { url = "https://files.pythonhosted.org/packages/0b/0d/363f78cacb70f58f15f4b083961bbd9d292f335d3f5c66fc4f1cfe69cb90/charset_normalizer-3.5.2-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:4495c5002a7b28557e7e222e77e0b661183e432b7d6d2e788101e3f240e05b8c", size = 286747, upload-time = "2026-09-30T04:35:13.022Z" }, + { url = "https://files.pythonhosted.org/packages/e4/ed/cf505d3011ffceb12c2067a7a5d3cfe92b875d4d44bb0ff0d69375e2c184/charset_normalizer-3.5.2-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:211d5a3eb6af8f513b8d4ca19a8c1b7accab1b5f0d3175f9826b03c1a920dc1f", size = 269972, upload-time = "2026-09-30T04:35:14.606Z" }, + { url = "https://files.pythonhosted.org/packages/15/d8/f0a93a431d170e7ca681d4f6650fee3de934d18560e474e7267eb4b0f987/charset_normalizer-3.5.2-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:ef4fcbf3327382cd4c9f540babd61248208af7b93eec4de397b4d5f58a09e288", size = 269339, upload-time = "2026-09-30T04:35:16.087Z" }, + { url = "https://files.pythonhosted.org/packages/86/bd/9b2bd1c5b7af02462c9752d33994834ff972a96b4c483eefde9e594488e2/charset_normalizer-3.5.2-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:bd16aabe4a02a297c23417aa17ac6299dbd8c49f673bcd645b4929b11f5a4400", size = 261683, upload-time = "2026-09-30T04:35:17.488Z" }, + { url = "https://files.pythonhosted.org/packages/76/a5/cac540ab0fd61f3fec88ad3dbb64509e71424593d73cfdfff5ab3e4db279/charset_normalizer-3.5.2-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:fb9e68df06293761f9fe66ade60a9bc6d0f5e42b8acf2939a9158af86ab0e5bd", size = 248143, upload-time = "2026-09-30T04:35:18.849Z" }, + { url = "https://files.pythonhosted.org/packages/71/7a/ff467301deef2089fad87f72df9e000a26a78fec7acbb18e1999371b8369/charset_normalizer-3.5.2-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:59f63901b0031c3136cf64704dcb21de0bbae62ce2c9529bc39d27665463de37", size = 292358, upload-time = "2026-09-30T04:35:20.326Z" }, + { url = "https://files.pythonhosted.org/packages/ad/77/22d7e785d1e210afc2e2f58600dd1799d17a35665faf84383f002826c5f8/charset_normalizer-3.5.2-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:304d5463e65a35d7bb0850550e0780395395f6fcf452f04db7d5ca7cecc425ac", size = 268766, upload-time = "2026-09-30T04:35:21.72Z" }, + { url = "https://files.pythonhosted.org/packages/ae/91/e8e946267f1c2d9e2bd651726e2fbd2addf02c4d36cea5069e32ca9d7bb5/charset_normalizer-3.5.2-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:9cf9b1a857e25c4baceeb3624e92a56df3668f398c4acba74e174d81fb4d1d3a", size = 287432, upload-time = "2026-09-30T04:35:23.273Z" }, + { url = "https://files.pythonhosted.org/packages/4e/88/7561d8a88d555e7df6623abe7c0070b4baf47549b9408783a2ae0a1a6cf7/charset_normalizer-3.5.2-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:114e4d0c92d618409ed82a99e22b5c5e768fe995f2973f78265f4524f49d4640", size = 272470, upload-time = "2026-09-30T04:35:24.655Z" }, + { url = "https://files.pythonhosted.org/packages/35/7e/578c702301ec036f01455f30744a08d2b42f6ab35b9b2d4bf8cae0ef2a80/charset_normalizer-3.5.2-cp311-cp311-win32.whl", hash = "sha256:2625388c6c754520c37abaf3b41eb34d1cc4a373f457898f08606c8e362b891d", size = 186688, upload-time = "2026-09-30T04:35:26.225Z" }, + { url = "https://files.pythonhosted.org/packages/e8/fc/fdf8cf52ff21cd5bf158f20978991cf985325842f74283eb6df26c8a39d8/charset_normalizer-3.5.2-cp311-cp311-win_amd64.whl", hash = "sha256:87e50a3e7cb90af586b6c5faf23e302a970415ac73bd7bd90a515a04b427ef96", size = 214932, upload-time = "2026-09-30T04:35:27.796Z" }, + { url = "https://files.pythonhosted.org/packages/97/66/3e45a506d8110b632541faf9a9470185aa9878f1ed44020f31346c1c5e5b/charset_normalizer-3.5.2-cp311-cp311-win_arm64.whl", hash = "sha256:254eb48b9fa5ee9898a3c445825a1f340fe53712a098904b39b0bddba8ea3cb1", size = 202828, upload-time = "2026-09-30T04:35:29.259Z" }, + { url = "https://files.pythonhosted.org/packages/e7/c8/693809898870237d82785a03f3b2b58fe4c9f14669f84a7d4e623c92a59e/charset_normalizer-3.5.2-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:ed2a239c0ea213acc1908150a3037257083c7c083128f1a4cec2ec4b97dca491", size = 367780, upload-time = "2026-09-30T04:35:30.888Z" }, + { url = "https://files.pythonhosted.org/packages/c9/87/2fea8c13dc24b3ca9c6f803a5b2dfdeae73eb4f9e12c7885ed908ff0433c/charset_normalizer-3.5.2-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:b91363207bd9dc966a691e959bb47f64b30f7ac4b072be9968b366982f7db77c", size = 246730, upload-time = "2026-09-30T04:35:32.286Z" }, + { url = "https://files.pythonhosted.org/packages/a8/9e/09efac30b937722f46d3110ba30b875b24b2e3a266ed746cc4e376a94d80/charset_normalizer-3.5.2-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:38a873987f3be698494da8b2e3085e29da02da7b633dce73e79c699a113d7bf0", size = 237707, upload-time = "2026-09-30T04:35:33.709Z" }, + { url = "https://files.pythonhosted.org/packages/9e/18/70d76670b13686237863a379928d60bd10e021f17d243ab3d7014c4a5f4e/charset_normalizer-3.5.2-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:355ad8011081dec5412240c087a9a0c9d4d5039f3ed11a3f13e18c2b29b56c51", size = 273050, upload-time = "2026-09-30T04:35:35.138Z" }, + { url = "https://files.pythonhosted.org/packages/54/e2/77a8b09d5adc013ed07b95b01b8b8fa5441c4e810e83ee7e4aae2fa4d91a/charset_normalizer-3.5.2-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:ee21e28f0430bd6dc9086c6e525d5e818a44a5ad19720c8a0ef766792f3eb5e5", size = 270345, upload-time = "2026-09-30T04:35:36.502Z" }, + { url = "https://files.pythonhosted.org/packages/7f/c5/38806a25ab5e65fc178f39affeda20858efafede2fce1ffc2556cfc9fe73/charset_normalizer-3.5.2-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:3d31298449090ab8d47b7b1b2a555ff73cac7ed438a08b7ac160980c7ebed649", size = 257601, upload-time = "2026-09-30T04:35:37.919Z" }, + { url = "https://files.pythonhosted.org/packages/ae/8d/213565184708fdb263ae55e2c04ee1ff748129dd65d48ed0e3502da9c85a/charset_normalizer-3.5.2-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:5cde776b7cc66e4f6c99612cea4aa7269aa65863f7a15841b2c264f103822f4e", size = 252222, upload-time = "2026-09-30T04:35:39.544Z" }, + { url = "https://files.pythonhosted.org/packages/7e/24/76d2cefc25472531e4c5c7dfff68865eb1c39b78482f0fdc15b46f047830/charset_normalizer-3.5.2-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:ae4f5fea5b8b8ccff88238cc8569303e5ee95efae67fa62922a311397a71f346", size = 248482, upload-time = "2026-09-30T04:35:41.088Z" }, + { url = "https://files.pythonhosted.org/packages/7d/dc/65a801b66ab4c197e22c433ab25e7ac24324ac6f45a2269aca42cce309bf/charset_normalizer-3.5.2-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:f7d486c83842422badd511868fd8a9a20e9407ace71564b6af47ce7e60a336c1", size = 241206, upload-time = "2026-09-30T04:35:42.59Z" }, + { url = "https://files.pythonhosted.org/packages/a7/95/ca9b5eabde673002c6f1e7ada1b223916fe18f6d661da7aabd4d643718f1/charset_normalizer-3.5.2-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:11a4d68a6ecda3292cb1e50239e111543ba5d709bb62a6b4ea1afcfa729d8875", size = 273190, upload-time = "2026-09-30T04:35:44.347Z" }, + { url = "https://files.pythonhosted.org/packages/2d/8b/803b4d2a3f6e1740f63f1e87b04d14b42f3d4fdfe6ed7d4db2d34102b14f/charset_normalizer-3.5.2-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:d6734d2ef8a50fbf8445c139477da401f50d62a0606bf00e20ec6d87773fefb1", size = 253527, upload-time = "2026-09-30T04:35:45.915Z" }, + { url = "https://files.pythonhosted.org/packages/a9/55/93c0e5dbd085ae0471346026abbe7e0db9ea2d6fea74e51f0b5a46f233a7/charset_normalizer-3.5.2-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:a815775b6c38d4e0ff7bcffbeba67feded90202bb6a226b8dd35f1c855217413", size = 271285, upload-time = "2026-09-30T04:35:47.49Z" }, + { url = "https://files.pythonhosted.org/packages/95/69/0dbd0e0b9b16cfa816cdfcb3e2e3854a1f680dc07fb1245ea125e7448060/charset_normalizer-3.5.2-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:23851fb4e1b85ed3f6c2a27b777cdfe2e19fb5b38429a8faf38c7542b7665869", size = 260010, upload-time = "2026-09-30T04:35:48.996Z" }, + { url = "https://files.pythonhosted.org/packages/58/9d/e7b88e7b1bf403590c3b573277b5e1e488c68c7a6fbacca310a2c324e90c/charset_normalizer-3.5.2-cp312-cp312-win32.whl", hash = "sha256:db19d07e2e0129e974a0e65d0064fc222a446cd5122c2fd4184d2af9fc734a9e", size = 184125, upload-time = "2026-09-30T04:35:50.777Z" }, + { url = "https://files.pythonhosted.org/packages/eb/e6/e6e083884cbcfd49c64865af05027fe7011be7b2d9179524f099a1b611f3/charset_normalizer-3.5.2-cp312-cp312-win_amd64.whl", hash = "sha256:780fbe7cab297b81dad9fb8dc5eb003c0468ffb0d9e5f65068c53a34661a96bc", size = 207486, upload-time = "2026-09-30T04:35:52.194Z" }, + { url = "https://files.pythonhosted.org/packages/c4/e3/017aea0911ada7405a825c7d937eb3a13009664e2f5b38e8c4bbf2abf894/charset_normalizer-3.5.2-cp312-cp312-win_arm64.whl", hash = "sha256:e2af3aad578aa6bd1384bcf4750fc285e5a9de53f40b7d41e5a0bf748edeb2b3", size = 196734, upload-time = "2026-09-30T04:35:53.636Z" }, + { url = "https://files.pythonhosted.org/packages/c5/34/68292d68512768591aaff07c59bb53ee31341c87759433a859c4641a50c2/charset_normalizer-3.5.2-cp313-cp313-android_24_arm64_v8a.whl", hash = "sha256:ed905975ab14056a2e5eb1c376cb2e1ebc5396baf84163939c518556fccde9f5", size = 225946, upload-time = "2026-09-30T04:35:55.313Z" }, + { url = "https://files.pythonhosted.org/packages/e3/80/bee0b01b90ccd5322ae1d0abb33fab1bd95b7c2eadaf02aeccf22e04ee83/charset_normalizer-3.5.2-cp313-cp313-android_24_x86_64.whl", hash = "sha256:a66c3bc5ab1f0ff2164fc9965ddd611ff0802173f4b9d24554c563f6ab7e1d6e", size = 238619, upload-time = "2026-09-30T04:35:56.863Z" }, + { url = "https://files.pythonhosted.org/packages/78/6e/60ce52a85a7fd631ae8482ae6d74521014ca2f255892679484dc04d7ef56/charset_normalizer-3.5.2-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:d2374b62878abb00cd8309b32af6c0b715cd02dec0ca74ef12e5069bdc64144a", size = 206701, upload-time = "2026-09-30T04:35:58.639Z" }, + { url = "https://files.pythonhosted.org/packages/36/8c/71aafad23f971afc84c2b295bc0c560739ce1dac558aad9fec22e39f3639/charset_normalizer-3.5.2-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:d376bbd28b3a8999db1a103b3b388aee6f1ddeb3e51bc2172993efdcd86e064d", size = 210506, upload-time = "2026-09-30T04:36:00.147Z" }, + { url = "https://files.pythonhosted.org/packages/91/da/3c5a7798c046df7d2d68ad653cf5b6c5a8bfee225055a843c6f2f42aac1a/charset_normalizer-3.5.2-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:6045373d5a89a5ec71afde535db987ca28e76dfa276c2d4c818265b375d4b055", size = 366626, upload-time = "2026-09-30T04:36:01.77Z" }, + { url = "https://files.pythonhosted.org/packages/e1/16/710ac3de2ee354e2bd1a9c94efe45a2d27b5c6ad39b2d6a905be2c094b6c/charset_normalizer-3.5.2-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:849df64e889b2e17230d58410a03dba311a65b163508fd33679b2b737d4b7858", size = 244624, upload-time = "2026-09-30T04:36:03.389Z" }, + { url = "https://files.pythonhosted.org/packages/d6/39/45c7439f5b63d24f7d5b2a1d760f34af7628782d7144b4cc8ded45c2d4bc/charset_normalizer-3.5.2-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:15c44f7edfd477b06f517a5cc317fc1707edb9de2c865f43d4b6513907473234", size = 235731, upload-time = "2026-09-30T04:36:04.987Z" }, + { url = "https://files.pythonhosted.org/packages/4d/34/38f3154785ce92e9f56eb226f4d35bdfae6b008480dd055f58837a89c810/charset_normalizer-3.5.2-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:a89012d6d5476ee112d20d998570ed58df2260a852afb1758809cd6900411d21", size = 270352, upload-time = "2026-09-30T04:36:06.412Z" }, + { url = "https://files.pythonhosted.org/packages/04/f3/859f74e7babc977705026b30593b3be04049632a522fb7000f83c033d747/charset_normalizer-3.5.2-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:0c951d5e6dd9c2ff60609476752bee49da4206adde960ebc247766937f72e718", size = 267393, upload-time = "2026-09-30T04:36:07.865Z" }, + { url = "https://files.pythonhosted.org/packages/4b/85/41d27f234b82e47c167a5f6c0f62501dc0c640585ff4aba79e08a390336a/charset_normalizer-3.5.2-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:7218e8f32b0956cfcd048fd42d9d5779809745ca1d86113ca56f66e7ae1549c4", size = 254660, upload-time = "2026-09-30T04:36:09.248Z" }, + { url = "https://files.pythonhosted.org/packages/58/ca/5d1a997587febe5b26d8daffe363b5c1a091cece19828eec6502fd09c5ef/charset_normalizer-3.5.2-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:a19a731138fc27d5682277d3b9df22855cea1239bce7fcec5f78f42ef2d1f3c3", size = 249996, upload-time = "2026-09-30T04:36:10.73Z" }, + { url = "https://files.pythonhosted.org/packages/b3/1f/d1e78246f7ed60c8c8d606b4ac27f66ce49cc3e95f24893ccbeba9f77302/charset_normalizer-3.5.2-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:62603db9a7caa0802eaa28c1c46fecd7b3a263a774069c24c3c28c302448721c", size = 247028, upload-time = "2026-09-30T04:36:12.294Z" }, + { url = "https://files.pythonhosted.org/packages/8e/37/eba316edd4f0c4d3a5d945924c4eeeae59abac4056aa815d8a4268f863a2/charset_normalizer-3.5.2-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:b6856554c4f44d79fc2307d5768854310a8f0096e501c75637542c82292b0429", size = 240562, upload-time = "2026-09-30T04:36:13.887Z" }, + { url = "https://files.pythonhosted.org/packages/c8/8e/aaa037d40ca9ef045977f1a661048b1aa33f223adfce3452fe9be9f79d14/charset_normalizer-3.5.2-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:1bc0baf5ef96b6ede57d47f4b8fe4d9d84019c3bfcbeb20a41edc6a6ee341f1f", size = 271699, upload-time = "2026-09-30T04:36:15.41Z" }, + { url = "https://files.pythonhosted.org/packages/26/19/1c1c9f75974adf523b87f34b8a2adc5a435cd65916812bcbd0dfa45f9a29/charset_normalizer-3.5.2-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:56bc200a365efb37383b7852e4cc5898d3b2da5987289b543956cf8cad71018a", size = 252030, upload-time = "2026-09-30T04:36:16.839Z" }, + { url = "https://files.pythonhosted.org/packages/bc/90/0660ef18e18df0a4d2a1a0edff7dfbba42d4e50ef2425557a5bb7051f77b/charset_normalizer-3.5.2-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:2c9ad19a6cfcd5ea5c0d41161d22f9df1dcc277e9bef2751391334546a314c00", size = 268565, upload-time = "2026-09-30T04:36:18.468Z" }, + { url = "https://files.pythonhosted.org/packages/79/ba/57adc269824e8658f1a0f97a9e514c247445a9632b3419b97e0ba37f16dc/charset_normalizer-3.5.2-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:e243bd13217235fc7290c621941c3f5cc8b66e4872495be821d7436ba2fb838d", size = 257627, upload-time = "2026-09-30T04:36:19.938Z" }, + { url = "https://files.pythonhosted.org/packages/9a/85/33abd4315c052d3d4f54c92b1ee49bfbc0dc7115a981e462a793b6d2ab87/charset_normalizer-3.5.2-cp313-cp313-pyemscripten_2025_0_wasm32.whl", hash = "sha256:a090bb2c68df85450502e3e20d665e3a5af9c65a84d6508ed477badd49166fd3", size = 143407, upload-time = "2026-09-30T04:36:21.376Z" }, + { url = "https://files.pythonhosted.org/packages/4f/de/6435e18d1aaa5d910b896d551411c96af1f42a0c56c29afc2016c61ccc2e/charset_normalizer-3.5.2-cp313-cp313-win32.whl", hash = "sha256:2b7b3bbfb4fe8ef40600792d762fbaa9057559f9d3fad209525b7a22b99e91fd", size = 183207, upload-time = "2026-09-30T04:36:22.776Z" }, + { url = "https://files.pythonhosted.org/packages/9c/76/b8ec57f4e9ee3253541abf95e4a462c0175fe8032dcd070f1f2421240942/charset_normalizer-3.5.2-cp313-cp313-win_amd64.whl", hash = "sha256:78456a747de8dc58360ffa581f30a002baf5aa28cb262536545e91f113ed7639", size = 206489, upload-time = "2026-09-30T04:36:24.306Z" }, + { url = "https://files.pythonhosted.org/packages/3e/60/c647c6ae47480221e875ea5d743ff94946f7416e3c69415ab772928e8d32/charset_normalizer-3.5.2-cp313-cp313-win_arm64.whl", hash = "sha256:11912e4bb14baae7c5d8791aa55ba0a3a03ec6729073307b0f57270abaa713d3", size = 195501, upload-time = "2026-09-30T04:36:25.846Z" }, + { url = "https://files.pythonhosted.org/packages/8c/ab/176fbfd5b64939c55d652366aa5b9ef1d767af207a3aa6ebeb0d226c484d/charset_normalizer-3.5.2-cp37-abi3-macosx_10_9_universal2.whl", hash = "sha256:4275811936e2f06feff5e598fb42a1b7ae852da8e39605211892b56b81a34efd", size = 331815, upload-time = "2026-09-30T04:38:26.216Z" }, + { url = "https://files.pythonhosted.org/packages/7e/84/371eac6b30bdbcbf2d632a1a01809103459216fcaae61b8b8d922c1bfb8a/charset_normalizer-3.5.2-cp37-abi3-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:1c50fe28bbc2ced33386f298650d91218076c05420e6cbd790b913adc41659e7", size = 253276, upload-time = "2026-09-30T04:38:28.032Z" }, + { url = "https://files.pythonhosted.org/packages/43/6f/c4fbae58febff71709c51bc7e18fdfa55341dc382704740f9f0cbf03817b/charset_normalizer-3.5.2-cp37-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d19fbd981a488e22cd04883659ca6b08f50b5974f9fd7c95655ef6a043e5893f", size = 241239, upload-time = "2026-09-30T04:38:29.732Z" }, + { url = "https://files.pythonhosted.org/packages/61/71/458c3f42164a07d0c5210798e9e704b39e540a6793b05aba67f3a35243a9/charset_normalizer-3.5.2-cp37-abi3-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:0fed1d06615f022ee3b13caf5e8b180cfea32bb2c5aded8a9d44277afc040f93", size = 231121, upload-time = "2026-09-30T04:38:31.462Z" }, + { url = "https://files.pythonhosted.org/packages/09/54/ab9e89367076f6331bb6c65c4bf14a5361fa5191cb6561bf534f18504e1b/charset_normalizer-3.5.2-cp37-abi3-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:838dcc90063569a0448120554591a1d6c4a4ffe11babf048908793154ab86ade", size = 260350, upload-time = "2026-09-30T04:38:33.239Z" }, + { url = "https://files.pythonhosted.org/packages/7c/c1/061431ecc688d9d76602502cb57cc01e691e682c18f1beb45f9673b5bbd2/charset_normalizer-3.5.2-cp37-abi3-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:2ce45c6627b22c47e390bc91a41c3d13032192e699fa0bea96e9671b373d69b0", size = 255430, upload-time = "2026-09-30T04:38:34.865Z" }, + { url = "https://files.pythonhosted.org/packages/8d/1f/20c8949f0676f7ab811abdeb7f4d7f1cbc6e61ff20bef08b44edeb092bc8/charset_normalizer-3.5.2-cp37-abi3-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:0774bf9bf620249fee3e0b8b9fd3065de213be30f3aa94ce2494b3b638949e26", size = 250612, upload-time = "2026-09-30T04:38:36.649Z" }, + { url = "https://files.pythonhosted.org/packages/2b/9e/46f2fa4c431fc98c4ae76a8cb5bdca54e0341e3cfc3fcfd8e82740250818/charset_normalizer-3.5.2-cp37-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:1db38f4c5496827c1a501846d64d14c3b80c7e6714e406cd7dc36a9899fa1011", size = 242083, upload-time = "2026-09-30T04:38:38.26Z" }, + { url = "https://files.pythonhosted.org/packages/bd/39/559be29a0c0f086e0bba6922babd38916cc5e0b58ced4de13ee01ea05508/charset_normalizer-3.5.2-cp37-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:304d8e4d493af723536393eee0c689eb7813f4a474c8b479dee63f1fdd98f621", size = 232738, upload-time = "2026-09-30T04:38:39.81Z" }, + { url = "https://files.pythonhosted.org/packages/ff/6c/387b0e4f756a282831c1d9fc6aeb6c51ca4507ca202767c8de15ce9b12e2/charset_normalizer-3.5.2-cp37-abi3-musllinux_1_2_ppc64le.whl", hash = "sha256:9b7f416ff0978e2f2249330527f0ad6fa02f4932e6199692d3b52da2048c19e4", size = 260703, upload-time = "2026-09-30T04:38:41.346Z" }, + { url = "https://files.pythonhosted.org/packages/96/92/1fdf015f09ef449f50d3ac4b67c90887c9c318b727daa95cc4f866e6521d/charset_normalizer-3.5.2-cp37-abi3-musllinux_1_2_riscv64.whl", hash = "sha256:01077390b03f7988f11d700a2194e69b119741a86b1a638b1db88891e3eced8e", size = 247622, upload-time = "2026-09-30T04:38:42.937Z" }, + { url = "https://files.pythonhosted.org/packages/dc/3c/8e7b8a5671ad5d433669fb2a76f1a0164df2d9b1718b0206bc2a16d840cc/charset_normalizer-3.5.2-cp37-abi3-musllinux_1_2_s390x.whl", hash = "sha256:7e841fb9010836c992c9f12fcbd43a831de93a5f726fc1ccd8ca1d0268c5014c", size = 257500, upload-time = "2026-09-30T04:38:44.604Z" }, + { url = "https://files.pythonhosted.org/packages/b4/f0/45b579df5cabc1d5d53ea1cc35e8437d3ca768c0acccc7041517cb6fbb32/charset_normalizer-3.5.2-cp37-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:9cae88599c7219005d879f98e5ed53341e9a122af585e1091200358a3003d2a0", size = 255100, upload-time = "2026-09-30T04:38:46.289Z" }, + { url = "https://files.pythonhosted.org/packages/31/68/fdec18a343f5fb3f310588dd478b09ac4799e0b187dbade3a8cd776f03ef/charset_normalizer-3.5.2-cp37-abi3-win32.whl", hash = "sha256:01b0c0d2262a9e28e8484a278c7e1b5d650e3ac8cf2683d2967e25899f208bdf", size = 174499, upload-time = "2026-09-30T04:38:47.999Z" }, + { url = "https://files.pythonhosted.org/packages/9d/8a/b618149cc5207943a0242068d7a27897f56a62947b5a039085f2a22029f8/charset_normalizer-3.5.2-cp37-abi3-win_amd64.whl", hash = "sha256:9f56f72050826f63dcee7a7f55b0a77168cb3bfc553fd405e7f8f9ece75a4036", size = 200092, upload-time = "2026-09-30T04:38:49.707Z" }, + { url = "https://files.pythonhosted.org/packages/03/cf/4c66866fa9e2b1c78e3c911516d1de497a677b7ac60f1eceda74ce777ca3/charset_normalizer-3.5.2-cp37-abi3-win_arm64.whl", hash = "sha256:40ab6bffa02ae10a0581e6c198be7d2d8ca5c2a0c64e4ed3465d766df457573e", size = 294363, upload-time = "2026-09-30T04:38:51.312Z" }, + { url = "https://files.pythonhosted.org/packages/fc/ad/d07d7862a62ffa6d79d68074d14823243dd235a77c45262acbf6adeb28bf/charset_normalizer-3.5.2-py3-none-any.whl", hash = "sha256:b6b751274acb69d77b3323d6b7dbaa3c7fdfc1eb829b7eb61d262f32e1af9685", size = 68872, upload-time = "2026-09-30T04:39:21.828Z" }, +] + [[package]] name = "click" version = "8.4.1" @@ -463,6 +659,88 @@ name = "crossplane-models" version = "0.0.0" source = { editable = "schemas/python" } +[[package]] +name = "durationpy" +version = "0.11" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/b0/5d/5f8571bd5dedc80863191621ac4be001f3f3dd8315d2ec078705dab7dec1/durationpy-0.11.tar.gz", hash = "sha256:181898e1ae282e288f0a2291829656bf1b6b3aadf30a97993b85db4943642905", size = 3582, upload-time = "2026-08-26T13:56:00.991Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e6/c4/ebdf7837bc4ef6fd98cfb013c28855bb358467bf86c1af011bbc21e21df0/durationpy-0.11-py3-none-any.whl", hash = "sha256:a739fe2b8972c250ff72f8e2c488d18cf25f7b852f49ee76048775d5171df30c", size = 4133, upload-time = "2026-08-26T13:55:59.456Z" }, +] + +[[package]] +name = "frozenlist" +version = "1.8.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/2d/f5/c831fac6cc817d26fd54c7eaccd04ef7e0288806943f7cc5bbf69f3ac1f0/frozenlist-1.8.0.tar.gz", hash = "sha256:3ede829ed8d842f6cd48fc7081d7a41001a56f1f38603f9d49bf3020d59a31ad", size = 45875, upload-time = "2025-10-06T05:38:17.865Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/bc/03/077f869d540370db12165c0aa51640a873fb661d8b315d1d4d67b284d7ac/frozenlist-1.8.0-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:09474e9831bc2b2199fad6da3c14c7b0fbdd377cce9d3d77131be28906cb7d84", size = 86912, upload-time = "2025-10-06T05:35:45.98Z" }, + { url = "https://files.pythonhosted.org/packages/df/b5/7610b6bd13e4ae77b96ba85abea1c8cb249683217ef09ac9e0ae93f25a91/frozenlist-1.8.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:17c883ab0ab67200b5f964d2b9ed6b00971917d5d8a92df149dc2c9779208ee9", size = 50046, upload-time = "2025-10-06T05:35:47.009Z" }, + { url = "https://files.pythonhosted.org/packages/6e/ef/0e8f1fe32f8a53dd26bdd1f9347efe0778b0fddf62789ea683f4cc7d787d/frozenlist-1.8.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:fa47e444b8ba08fffd1c18e8cdb9a75db1b6a27f17507522834ad13ed5922b93", size = 50119, upload-time = "2025-10-06T05:35:48.38Z" }, + { url = "https://files.pythonhosted.org/packages/11/b1/71a477adc7c36e5fb628245dfbdea2166feae310757dea848d02bd0689fd/frozenlist-1.8.0-cp311-cp311-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:2552f44204b744fba866e573be4c1f9048d6a324dfe14475103fd51613eb1d1f", size = 231067, upload-time = "2025-10-06T05:35:49.97Z" }, + { url = "https://files.pythonhosted.org/packages/45/7e/afe40eca3a2dc19b9904c0f5d7edfe82b5304cb831391edec0ac04af94c2/frozenlist-1.8.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:957e7c38f250991e48a9a73e6423db1bb9dd14e722a10f6b8bb8e16a0f55f695", size = 233160, upload-time = "2025-10-06T05:35:51.729Z" }, + { url = "https://files.pythonhosted.org/packages/a6/aa/7416eac95603ce428679d273255ffc7c998d4132cfae200103f164b108aa/frozenlist-1.8.0-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:8585e3bb2cdea02fc88ffa245069c36555557ad3609e83be0ec71f54fd4abb52", size = 228544, upload-time = "2025-10-06T05:35:53.246Z" }, + { url = "https://files.pythonhosted.org/packages/8b/3d/2a2d1f683d55ac7e3875e4263d28410063e738384d3adc294f5ff3d7105e/frozenlist-1.8.0-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:edee74874ce20a373d62dc28b0b18b93f645633c2943fd90ee9d898550770581", size = 243797, upload-time = "2025-10-06T05:35:54.497Z" }, + { url = "https://files.pythonhosted.org/packages/78/1e/2d5565b589e580c296d3bb54da08d206e797d941a83a6fdea42af23be79c/frozenlist-1.8.0-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:c9a63152fe95756b85f31186bddf42e4c02c6321207fd6601a1c89ebac4fe567", size = 247923, upload-time = "2025-10-06T05:35:55.861Z" }, + { url = "https://files.pythonhosted.org/packages/aa/c3/65872fcf1d326a7f101ad4d86285c403c87be7d832b7470b77f6d2ed5ddc/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:b6db2185db9be0a04fecf2f241c70b63b1a242e2805be291855078f2b404dd6b", size = 230886, upload-time = "2025-10-06T05:35:57.399Z" }, + { url = "https://files.pythonhosted.org/packages/a0/76/ac9ced601d62f6956f03cc794f9e04c81719509f85255abf96e2510f4265/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:f4be2e3d8bc8aabd566f8d5b8ba7ecc09249d74ba3c9ed52e54dc23a293f0b92", size = 245731, upload-time = "2025-10-06T05:35:58.563Z" }, + { url = "https://files.pythonhosted.org/packages/b9/49/ecccb5f2598daf0b4a1415497eba4c33c1e8ce07495eb07d2860c731b8d5/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:c8d1634419f39ea6f5c427ea2f90ca85126b54b50837f31497f3bf38266e853d", size = 241544, upload-time = "2025-10-06T05:35:59.719Z" }, + { url = "https://files.pythonhosted.org/packages/53/4b/ddf24113323c0bbcc54cb38c8b8916f1da7165e07b8e24a717b4a12cbf10/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:1a7fa382a4a223773ed64242dbe1c9c326ec09457e6b8428efb4118c685c3dfd", size = 241806, upload-time = "2025-10-06T05:36:00.959Z" }, + { url = "https://files.pythonhosted.org/packages/a7/fb/9b9a084d73c67175484ba2789a59f8eebebd0827d186a8102005ce41e1ba/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:11847b53d722050808926e785df837353bd4d75f1d494377e59b23594d834967", size = 229382, upload-time = "2025-10-06T05:36:02.22Z" }, + { url = "https://files.pythonhosted.org/packages/95/a3/c8fb25aac55bf5e12dae5c5aa6a98f85d436c1dc658f21c3ac73f9fa95e5/frozenlist-1.8.0-cp311-cp311-win32.whl", hash = "sha256:27c6e8077956cf73eadd514be8fb04d77fc946a7fe9f7fe167648b0b9085cc25", size = 39647, upload-time = "2025-10-06T05:36:03.409Z" }, + { url = "https://files.pythonhosted.org/packages/0a/f5/603d0d6a02cfd4c8f2a095a54672b3cf967ad688a60fb9faf04fc4887f65/frozenlist-1.8.0-cp311-cp311-win_amd64.whl", hash = "sha256:ac913f8403b36a2c8610bbfd25b8013488533e71e62b4b4adce9c86c8cea905b", size = 44064, upload-time = "2025-10-06T05:36:04.368Z" }, + { url = "https://files.pythonhosted.org/packages/5d/16/c2c9ab44e181f043a86f9a8f84d5124b62dbcb3a02c0977ec72b9ac1d3e0/frozenlist-1.8.0-cp311-cp311-win_arm64.whl", hash = "sha256:d4d3214a0f8394edfa3e303136d0575eece0745ff2b47bd2cb2e66dd92d4351a", size = 39937, upload-time = "2025-10-06T05:36:05.669Z" }, + { url = "https://files.pythonhosted.org/packages/69/29/948b9aa87e75820a38650af445d2ef2b6b8a6fab1a23b6bb9e4ef0be2d59/frozenlist-1.8.0-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:78f7b9e5d6f2fdb88cdde9440dc147259b62b9d3b019924def9f6478be254ac1", size = 87782, upload-time = "2025-10-06T05:36:06.649Z" }, + { url = "https://files.pythonhosted.org/packages/64/80/4f6e318ee2a7c0750ed724fa33a4bdf1eacdc5a39a7a24e818a773cd91af/frozenlist-1.8.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:229bf37d2e4acdaf808fd3f06e854a4a7a3661e871b10dc1f8f1896a3b05f18b", size = 50594, upload-time = "2025-10-06T05:36:07.69Z" }, + { url = "https://files.pythonhosted.org/packages/2b/94/5c8a2b50a496b11dd519f4a24cb5496cf125681dd99e94c604ccdea9419a/frozenlist-1.8.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:f833670942247a14eafbb675458b4e61c82e002a148f49e68257b79296e865c4", size = 50448, upload-time = "2025-10-06T05:36:08.78Z" }, + { url = "https://files.pythonhosted.org/packages/6a/bd/d91c5e39f490a49df14320f4e8c80161cfcce09f1e2cde1edd16a551abb3/frozenlist-1.8.0-cp312-cp312-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:494a5952b1c597ba44e0e78113a7266e656b9794eec897b19ead706bd7074383", size = 242411, upload-time = "2025-10-06T05:36:09.801Z" }, + { url = "https://files.pythonhosted.org/packages/8f/83/f61505a05109ef3293dfb1ff594d13d64a2324ac3482be2cedc2be818256/frozenlist-1.8.0-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:96f423a119f4777a4a056b66ce11527366a8bb92f54e541ade21f2374433f6d4", size = 243014, upload-time = "2025-10-06T05:36:11.394Z" }, + { url = "https://files.pythonhosted.org/packages/d8/cb/cb6c7b0f7d4023ddda30cf56b8b17494eb3a79e3fda666bf735f63118b35/frozenlist-1.8.0-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:3462dd9475af2025c31cc61be6652dfa25cbfb56cbbf52f4ccfe029f38decaf8", size = 234909, upload-time = "2025-10-06T05:36:12.598Z" }, + { url = "https://files.pythonhosted.org/packages/31/c5/cd7a1f3b8b34af009fb17d4123c5a778b44ae2804e3ad6b86204255f9ec5/frozenlist-1.8.0-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:c4c800524c9cd9bac5166cd6f55285957fcfc907db323e193f2afcd4d9abd69b", size = 250049, upload-time = "2025-10-06T05:36:14.065Z" }, + { url = "https://files.pythonhosted.org/packages/c0/01/2f95d3b416c584a1e7f0e1d6d31998c4a795f7544069ee2e0962a4b60740/frozenlist-1.8.0-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:d6a5df73acd3399d893dafc71663ad22534b5aa4f94e8a2fabfe856c3c1b6a52", size = 256485, upload-time = "2025-10-06T05:36:15.39Z" }, + { url = "https://files.pythonhosted.org/packages/ce/03/024bf7720b3abaebcff6d0793d73c154237b85bdf67b7ed55e5e9596dc9a/frozenlist-1.8.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:405e8fe955c2280ce66428b3ca55e12b3c4e9c336fb2103a4937e891c69a4a29", size = 237619, upload-time = "2025-10-06T05:36:16.558Z" }, + { url = "https://files.pythonhosted.org/packages/69/fa/f8abdfe7d76b731f5d8bd217827cf6764d4f1d9763407e42717b4bed50a0/frozenlist-1.8.0-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:908bd3f6439f2fef9e85031b59fd4f1297af54415fb60e4254a95f75b3cab3f3", size = 250320, upload-time = "2025-10-06T05:36:17.821Z" }, + { url = "https://files.pythonhosted.org/packages/f5/3c/b051329f718b463b22613e269ad72138cc256c540f78a6de89452803a47d/frozenlist-1.8.0-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:294e487f9ec720bd8ffcebc99d575f7eff3568a08a253d1ee1a0378754b74143", size = 246820, upload-time = "2025-10-06T05:36:19.046Z" }, + { url = "https://files.pythonhosted.org/packages/0f/ae/58282e8f98e444b3f4dd42448ff36fa38bef29e40d40f330b22e7108f565/frozenlist-1.8.0-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:74c51543498289c0c43656701be6b077f4b265868fa7f8a8859c197006efb608", size = 250518, upload-time = "2025-10-06T05:36:20.763Z" }, + { url = "https://files.pythonhosted.org/packages/8f/96/007e5944694d66123183845a106547a15944fbbb7154788cbf7272789536/frozenlist-1.8.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:776f352e8329135506a1d6bf16ac3f87bc25b28e765949282dcc627af36123aa", size = 239096, upload-time = "2025-10-06T05:36:22.129Z" }, + { url = "https://files.pythonhosted.org/packages/66/bb/852b9d6db2fa40be96f29c0d1205c306288f0684df8fd26ca1951d461a56/frozenlist-1.8.0-cp312-cp312-win32.whl", hash = "sha256:433403ae80709741ce34038da08511d4a77062aa924baf411ef73d1146e74faf", size = 39985, upload-time = "2025-10-06T05:36:23.661Z" }, + { url = "https://files.pythonhosted.org/packages/b8/af/38e51a553dd66eb064cdf193841f16f077585d4d28394c2fa6235cb41765/frozenlist-1.8.0-cp312-cp312-win_amd64.whl", hash = "sha256:34187385b08f866104f0c0617404c8eb08165ab1272e884abc89c112e9c00746", size = 44591, upload-time = "2025-10-06T05:36:24.958Z" }, + { url = "https://files.pythonhosted.org/packages/a7/06/1dc65480ab147339fecc70797e9c2f69d9cea9cf38934ce08df070fdb9cb/frozenlist-1.8.0-cp312-cp312-win_arm64.whl", hash = "sha256:fe3c58d2f5db5fbd18c2987cba06d51b0529f52bc3a6cdc33d3f4eab725104bd", size = 40102, upload-time = "2025-10-06T05:36:26.333Z" }, + { url = "https://files.pythonhosted.org/packages/2d/40/0832c31a37d60f60ed79e9dfb5a92e1e2af4f40a16a29abcc7992af9edff/frozenlist-1.8.0-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:8d92f1a84bb12d9e56f818b3a746f3efba93c1b63c8387a73dde655e1e42282a", size = 85717, upload-time = "2025-10-06T05:36:27.341Z" }, + { url = "https://files.pythonhosted.org/packages/30/ba/b0b3de23f40bc55a7057bd38434e25c34fa48e17f20ee273bbde5e0650f3/frozenlist-1.8.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:96153e77a591c8adc2ee805756c61f59fef4cf4073a9275ee86fe8cba41241f7", size = 49651, upload-time = "2025-10-06T05:36:28.855Z" }, + { url = "https://files.pythonhosted.org/packages/0c/ab/6e5080ee374f875296c4243c381bbdef97a9ac39c6e3ce1d5f7d42cb78d6/frozenlist-1.8.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:f21f00a91358803399890ab167098c131ec2ddd5f8f5fd5fe9c9f2c6fcd91e40", size = 49417, upload-time = "2025-10-06T05:36:29.877Z" }, + { url = "https://files.pythonhosted.org/packages/d5/4e/e4691508f9477ce67da2015d8c00acd751e6287739123113a9fca6f1604e/frozenlist-1.8.0-cp313-cp313-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:fb30f9626572a76dfe4293c7194a09fb1fe93ba94c7d4f720dfae3b646b45027", size = 234391, upload-time = "2025-10-06T05:36:31.301Z" }, + { url = "https://files.pythonhosted.org/packages/40/76/c202df58e3acdf12969a7895fd6f3bc016c642e6726aa63bd3025e0fc71c/frozenlist-1.8.0-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:eaa352d7047a31d87dafcacbabe89df0aa506abb5b1b85a2fb91bc3faa02d822", size = 233048, upload-time = "2025-10-06T05:36:32.531Z" }, + { url = "https://files.pythonhosted.org/packages/f9/c0/8746afb90f17b73ca5979c7a3958116e105ff796e718575175319b5bb4ce/frozenlist-1.8.0-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:03ae967b4e297f58f8c774c7eabcce57fe3c2434817d4385c50661845a058121", size = 226549, upload-time = "2025-10-06T05:36:33.706Z" }, + { url = "https://files.pythonhosted.org/packages/7e/eb/4c7eefc718ff72f9b6c4893291abaae5fbc0c82226a32dcd8ef4f7a5dbef/frozenlist-1.8.0-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:f6292f1de555ffcc675941d65fffffb0a5bcd992905015f85d0592201793e0e5", size = 239833, upload-time = "2025-10-06T05:36:34.947Z" }, + { url = "https://files.pythonhosted.org/packages/c2/4e/e5c02187cf704224f8b21bee886f3d713ca379535f16893233b9d672ea71/frozenlist-1.8.0-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:29548f9b5b5e3460ce7378144c3010363d8035cea44bc0bf02d57f5a685e084e", size = 245363, upload-time = "2025-10-06T05:36:36.534Z" }, + { url = "https://files.pythonhosted.org/packages/1f/96/cb85ec608464472e82ad37a17f844889c36100eed57bea094518bf270692/frozenlist-1.8.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:ec3cc8c5d4084591b4237c0a272cc4f50a5b03396a47d9caaf76f5d7b38a4f11", size = 229314, upload-time = "2025-10-06T05:36:38.582Z" }, + { url = "https://files.pythonhosted.org/packages/5d/6f/4ae69c550e4cee66b57887daeebe006fe985917c01d0fff9caab9883f6d0/frozenlist-1.8.0-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:517279f58009d0b1f2e7c1b130b377a349405da3f7621ed6bfae50b10adf20c1", size = 243365, upload-time = "2025-10-06T05:36:40.152Z" }, + { url = "https://files.pythonhosted.org/packages/7a/58/afd56de246cf11780a40a2c28dc7cbabbf06337cc8ddb1c780a2d97e88d8/frozenlist-1.8.0-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:db1e72ede2d0d7ccb213f218df6a078a9c09a7de257c2fe8fcef16d5925230b1", size = 237763, upload-time = "2025-10-06T05:36:41.355Z" }, + { url = "https://files.pythonhosted.org/packages/cb/36/cdfaf6ed42e2644740d4a10452d8e97fa1c062e2a8006e4b09f1b5fd7d63/frozenlist-1.8.0-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:b4dec9482a65c54a5044486847b8a66bf10c9cb4926d42927ec4e8fd5db7fed8", size = 240110, upload-time = "2025-10-06T05:36:42.716Z" }, + { url = "https://files.pythonhosted.org/packages/03/a8/9ea226fbefad669f11b52e864c55f0bd57d3c8d7eb07e9f2e9a0b39502e1/frozenlist-1.8.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:21900c48ae04d13d416f0e1e0c4d81f7931f73a9dfa0b7a8746fb2fe7dd970ed", size = 233717, upload-time = "2025-10-06T05:36:44.251Z" }, + { url = "https://files.pythonhosted.org/packages/1e/0b/1b5531611e83ba7d13ccc9988967ea1b51186af64c42b7a7af465dcc9568/frozenlist-1.8.0-cp313-cp313-win32.whl", hash = "sha256:8b7b94a067d1c504ee0b16def57ad5738701e4ba10cec90529f13fa03c833496", size = 39628, upload-time = "2025-10-06T05:36:45.423Z" }, + { url = "https://files.pythonhosted.org/packages/d8/cf/174c91dbc9cc49bc7b7aab74d8b734e974d1faa8f191c74af9b7e80848e6/frozenlist-1.8.0-cp313-cp313-win_amd64.whl", hash = "sha256:878be833caa6a3821caf85eb39c5ba92d28e85df26d57afb06b35b2efd937231", size = 43882, upload-time = "2025-10-06T05:36:46.796Z" }, + { url = "https://files.pythonhosted.org/packages/c1/17/502cd212cbfa96eb1388614fe39a3fc9ab87dbbe042b66f97acb57474834/frozenlist-1.8.0-cp313-cp313-win_arm64.whl", hash = "sha256:44389d135b3ff43ba8cc89ff7f51f5a0bb6b63d829c8300f79a2fe4fe61bcc62", size = 39676, upload-time = "2025-10-06T05:36:47.8Z" }, + { url = "https://files.pythonhosted.org/packages/d2/5c/3bbfaa920dfab09e76946a5d2833a7cbdf7b9b4a91c714666ac4855b88b4/frozenlist-1.8.0-cp313-cp313t-macosx_10_13_universal2.whl", hash = "sha256:e25ac20a2ef37e91c1b39938b591457666a0fa835c7783c3a8f33ea42870db94", size = 89235, upload-time = "2025-10-06T05:36:48.78Z" }, + { url = "https://files.pythonhosted.org/packages/d2/d6/f03961ef72166cec1687e84e8925838442b615bd0b8854b54923ce5b7b8a/frozenlist-1.8.0-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:07cdca25a91a4386d2e76ad992916a85038a9b97561bf7a3fd12d5d9ce31870c", size = 50742, upload-time = "2025-10-06T05:36:49.837Z" }, + { url = "https://files.pythonhosted.org/packages/1e/bb/a6d12b7ba4c3337667d0e421f7181c82dda448ce4e7ad7ecd249a16fa806/frozenlist-1.8.0-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:4e0c11f2cc6717e0a741f84a527c52616140741cd812a50422f83dc31749fb52", size = 51725, upload-time = "2025-10-06T05:36:50.851Z" }, + { url = "https://files.pythonhosted.org/packages/bc/71/d1fed0ffe2c2ccd70b43714c6cab0f4188f09f8a67a7914a6b46ee30f274/frozenlist-1.8.0-cp313-cp313t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:b3210649ee28062ea6099cfda39e147fa1bc039583c8ee4481cb7811e2448c51", size = 284533, upload-time = "2025-10-06T05:36:51.898Z" }, + { url = "https://files.pythonhosted.org/packages/c9/1f/fb1685a7b009d89f9bf78a42d94461bc06581f6e718c39344754a5d9bada/frozenlist-1.8.0-cp313-cp313t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:581ef5194c48035a7de2aefc72ac6539823bb71508189e5de01d60c9dcd5fa65", size = 292506, upload-time = "2025-10-06T05:36:53.101Z" }, + { url = "https://files.pythonhosted.org/packages/e6/3b/b991fe1612703f7e0d05c0cf734c1b77aaf7c7d321df4572e8d36e7048c8/frozenlist-1.8.0-cp313-cp313t-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:3ef2d026f16a2b1866e1d86fc4e1291e1ed8a387b2c333809419a2f8b3a77b82", size = 274161, upload-time = "2025-10-06T05:36:54.309Z" }, + { url = "https://files.pythonhosted.org/packages/ca/ec/c5c618767bcdf66e88945ec0157d7f6c4a1322f1473392319b7a2501ded7/frozenlist-1.8.0-cp313-cp313t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:5500ef82073f599ac84d888e3a8c1f77ac831183244bfd7f11eaa0289fb30714", size = 294676, upload-time = "2025-10-06T05:36:55.566Z" }, + { url = "https://files.pythonhosted.org/packages/7c/ce/3934758637d8f8a88d11f0585d6495ef54b2044ed6ec84492a91fa3b27aa/frozenlist-1.8.0-cp313-cp313t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:50066c3997d0091c411a66e710f4e11752251e6d2d73d70d8d5d4c76442a199d", size = 300638, upload-time = "2025-10-06T05:36:56.758Z" }, + { url = "https://files.pythonhosted.org/packages/fc/4f/a7e4d0d467298f42de4b41cbc7ddaf19d3cfeabaf9ff97c20c6c7ee409f9/frozenlist-1.8.0-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:5c1c8e78426e59b3f8005e9b19f6ff46e5845895adbde20ece9218319eca6506", size = 283067, upload-time = "2025-10-06T05:36:57.965Z" }, + { url = "https://files.pythonhosted.org/packages/dc/48/c7b163063d55a83772b268e6d1affb960771b0e203b632cfe09522d67ea5/frozenlist-1.8.0-cp313-cp313t-musllinux_1_2_armv7l.whl", hash = "sha256:eefdba20de0d938cec6a89bd4d70f346a03108a19b9df4248d3cf0d88f1b0f51", size = 292101, upload-time = "2025-10-06T05:36:59.237Z" }, + { url = "https://files.pythonhosted.org/packages/9f/d0/2366d3c4ecdc2fd391e0afa6e11500bfba0ea772764d631bbf82f0136c9d/frozenlist-1.8.0-cp313-cp313t-musllinux_1_2_ppc64le.whl", hash = "sha256:cf253e0e1c3ceb4aaff6df637ce033ff6535fb8c70a764a8f46aafd3d6ab798e", size = 289901, upload-time = "2025-10-06T05:37:00.811Z" }, + { url = "https://files.pythonhosted.org/packages/b8/94/daff920e82c1b70e3618a2ac39fbc01ae3e2ff6124e80739ce5d71c9b920/frozenlist-1.8.0-cp313-cp313t-musllinux_1_2_s390x.whl", hash = "sha256:032efa2674356903cd0261c4317a561a6850f3ac864a63fc1583147fb05a79b0", size = 289395, upload-time = "2025-10-06T05:37:02.115Z" }, + { url = "https://files.pythonhosted.org/packages/e3/20/bba307ab4235a09fdcd3cc5508dbabd17c4634a1af4b96e0f69bfe551ebd/frozenlist-1.8.0-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:6da155091429aeba16851ecb10a9104a108bcd32f6c1642867eadaee401c1c41", size = 283659, upload-time = "2025-10-06T05:37:03.711Z" }, + { url = "https://files.pythonhosted.org/packages/fd/00/04ca1c3a7a124b6de4f8a9a17cc2fcad138b4608e7a3fc5877804b8715d7/frozenlist-1.8.0-cp313-cp313t-win32.whl", hash = "sha256:0f96534f8bfebc1a394209427d0f8a63d343c9779cda6fc25e8e121b5fd8555b", size = 43492, upload-time = "2025-10-06T05:37:04.915Z" }, + { url = "https://files.pythonhosted.org/packages/59/5e/c69f733a86a94ab10f68e496dc6b7e8bc078ebb415281d5698313e3af3a1/frozenlist-1.8.0-cp313-cp313t-win_amd64.whl", hash = "sha256:5d63a068f978fc69421fb0e6eb91a9603187527c86b7cd3f534a5b77a592b888", size = 48034, upload-time = "2025-10-06T05:37:06.343Z" }, + { url = "https://files.pythonhosted.org/packages/16/6c/be9d79775d8abe79b05fa6d23da99ad6e7763a1d080fbae7290b286093fd/frozenlist-1.8.0-cp313-cp313t-win_arm64.whl", hash = "sha256:bf0a7e10b077bf5fb9380ad3ae8ce20ef919a6ad93b4552896419ac7e1d8e042", size = 41749, upload-time = "2025-10-06T05:37:07.431Z" }, + { url = "https://files.pythonhosted.org/packages/9a/9a/e35b4a917281c0b8419d4207f4334c8e8c5dbf4f3f5f9ada73958d937dcc/frozenlist-1.8.0-py3-none-any.whl", hash = "sha256:0c18a16eab41e82c295618a77502e17b195883241c563b00f0aa5106fc4eaa0d", size = 13409, upload-time = "2025-10-06T05:38:16.721Z" }, +] + [[package]] name = "google-re2" version = "1.1.20251105" @@ -558,6 +836,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/82/54/acc6a6e684827b0f6bb4e2c27f3d7e25b71322c4078ef5b455c07c43260e/grpcio_reflection-1.62.3-py3-none-any.whl", hash = "sha256:a48ef37df81a3bada78261fc92ef382f061112f989d1312398b945cc69838b9c", size = 22232, upload-time = "2024-08-06T00:30:13.131Z" }, ] +[[package]] +name = "idna" +version = "3.20" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/f5/08/8eea9d4b8302028f3abb2c0813953f7aec26d33b7a8960ed760e65ff29fa/idna-3.20.tar.gz", hash = "sha256:a7db850025b95ded1eae8a46181a1a6c56c92c96f0e2b005d9ff8dc0210cab44", size = 216463, upload-time = "2026-09-17T14:11:04.752Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/58/a2/bb081bab032533a855d44de1d56f8e8426114ff1ba5d1f07a438a0a654f8/idna-3.20-py3-none-any.whl", hash = "sha256:ab7ae7122974553370f0bdb919e1a960b2cd1bc1ef0276416d896db81c14582c", size = 69583, upload-time = "2026-09-17T14:11:03.168Z" }, +] + [[package]] name = "iniconfig" version = "2.3.0" @@ -576,6 +863,27 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/14/2f/967ba146e6d58cf6a652da73885f52fc68001525b4197effc174321d70b4/jmespath-1.1.0-py3-none-any.whl", hash = "sha256:a5663118de4908c91729bea0acadca56526eb2698e83de10cd116ae0f4e97c64", size = 20419, upload-time = "2026-01-22T16:35:24.919Z" }, ] +[[package]] +name = "kubernetes" +version = "36.0.3" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "aiohttp" }, + { name = "certifi" }, + { name = "durationpy" }, + { name = "python-dateutil" }, + { name = "pyyaml" }, + { name = "requests" }, + { name = "requests-oauthlib" }, + { name = "six" }, + { name = "urllib3" }, + { name = "websocket-client" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/ca/57/b07b96353f902aa1bdbe00e878e3a12a137977d03a962479785576aa8ec9/kubernetes-36.0.3.tar.gz", hash = "sha256:36993ed25ce59b789c9341473a228fcf268504a2fec7c2b2b1531d73072e5ce7", size = 2337528, upload-time = "2026-07-13T20:38:12.128Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/5b/30/a96d47df739689ac0001ade0afefc16e3b477fc2fb426b568515fdc8afce/kubernetes-36.0.3-py2.py3-none-any.whl", hash = "sha256:8fde9241c4b298e6374a069dcf728359b4e72c2fb29489a975ba4e1c047cf10f", size = 4618066, upload-time = "2026-07-13T20:38:10.172Z" }, +] + [[package]] name = "lark" version = "1.3.1" @@ -593,6 +901,7 @@ source = { virtual = "." } [package.dev-dependencies] dev = [ { name = "crossplane-function-sdk-python" }, + { name = "kubernetes" }, { name = "pydantic" }, { name = "pytest" }, { name = "pyyaml" }, @@ -604,12 +913,94 @@ dev = [ [package.metadata.requires-dev] dev = [ { name = "crossplane-function-sdk-python", specifier = ">=0.14.0" }, + { name = "kubernetes", specifier = ">=36.0" }, { name = "pydantic", specifier = ">=2.0" }, { name = "pytest", specifier = ">=9.0" }, { name = "pyyaml", specifier = ">=6.0" }, { name = "types-protobuf", specifier = ">=4.24" }, ] +[[package]] +name = "multidict" +version = "6.9.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/d6/99/1d4d69c3512d0ddbfa3a1b69cfd9a151012ab2eb4eabbb096201b1f0b7d8/multidict-6.9.1.tar.gz", hash = "sha256:0f06e60fa190aa7abd0914c2a766736fdc8e9f34878c4346338534b73d1b20e2", size = 182404, upload-time = "2026-09-21T17:59:05.362Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e3/24/3823efc330630a1f132bcd0a9b182ddec3c53f452f551c8e8d3aaf5a2d20/multidict-6.9.1-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:910d4260512660484c0dc1588a316fbb35a40c081c36fc51d1225351af17cfe4", size = 96739, upload-time = "2026-09-21T17:54:42.815Z" }, + { url = "https://files.pythonhosted.org/packages/6a/69/36331fe1d3aeb0c525a9972d6cf82791075a395aa96f026f060009d691ae/multidict-6.9.1-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:33fa55b990f81c2927e01399ace0d18926c69d69baa8cdaa819424132fb97987", size = 58514, upload-time = "2026-09-21T17:54:44.064Z" }, + { url = "https://files.pythonhosted.org/packages/8c/ca/5860651f782078ef13f2761c2593f4ba066d76f626b2f30f23a7f838b775/multidict-6.9.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:369b5aa01b241cd3fea6890bdbb11a1425d87bf1515831500d518f4223e9d72c", size = 57303, upload-time = "2026-09-21T17:54:45.304Z" }, + { url = "https://files.pythonhosted.org/packages/c7/9f/c56e2fa223cb5d182660e1a09eec38880f96b7cd22769582f74447c60418/multidict-6.9.1-cp311-cp311-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:803f8b575a71b1b299d677c28db0653459c79b5308874efec813f17b7457c7f1", size = 320076, upload-time = "2026-09-21T17:54:46.826Z" }, + { url = "https://files.pythonhosted.org/packages/59/97/eda4fe0cad51f96096363ed32d3e0f8df28dab00beff4d92adc9db517726/multidict-6.9.1-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:b8a65621b98984a62e59403009591b8a5a7736273aefe1cab64cfb85b365cc07", size = 315489, upload-time = "2026-09-21T17:54:48.25Z" }, + { url = "https://files.pythonhosted.org/packages/24/14/10b1ded0a4085a51c3054125f854da4f47ce4fec724a2d2503231b83f3db/multidict-6.9.1-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:13849a1d4f54c3809ae721e9e83ab28f5ea602f33660cb84eb6ef261eac706c1", size = 296716, upload-time = "2026-09-21T17:54:49.749Z" }, + { url = "https://files.pythonhosted.org/packages/dc/14/04d155dd18528443cb9009f3a9f35f7417773797d072c95104ff17b8b9b7/multidict-6.9.1-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:dd9a137a4a9becda3094f3831cd026380f75f6855e051eefe4c73ade524f1cc3", size = 330244, upload-time = "2026-09-21T17:54:51.26Z" }, + { url = "https://files.pythonhosted.org/packages/8b/90/284d10b3a9e5b312a8ec5b7a175fc7d560eefdc99886320449891a197992/multidict-6.9.1-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:9ae9614c317c50836689ce2dfde07c05fe0b16378562e2221746c3913ede3c80", size = 331288, upload-time = "2026-09-21T17:54:52.595Z" }, + { url = "https://files.pythonhosted.org/packages/27/50/f420de3683f9b047fc5588064d84043ef57a7152ef451b8373dfc7068d22/multidict-6.9.1-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:e8f1e362c9352b50ed120f001046fdbb80810c9d56580f4c3fc13bbe30823387", size = 320663, upload-time = "2026-09-21T17:54:53.91Z" }, + { url = "https://files.pythonhosted.org/packages/01/ab/0120a650d7ce4fe167299c13a13cc4ee2dc72458a9c0e9e9330b824b73c9/multidict-6.9.1-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:0dd655518f136febd96c05131a76a863e32fc2a1d7acd4e3c959e3ceb77d8345", size = 291139, upload-time = "2026-09-21T17:54:55.313Z" }, + { url = "https://files.pythonhosted.org/packages/b3/3f/c66342f73ee5cd2c9b1dda6cfce89b5f348c48921b8c8b6e59aafbbfe31a/multidict-6.9.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:2c5d675da8f1cb5650271c8ad5e95c0a3e5a183c105e72d953b12877b1c8d0fd", size = 310044, upload-time = "2026-09-21T17:54:56.745Z" }, + { url = "https://files.pythonhosted.org/packages/05/0d/e44b90d9e77e44ec3473c4c7ac7ea95764d7e8fbf6d5a085ef9bff7ce0c2/multidict-6.9.1-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:658f90f49cf5af2441cad0a2b801c3ef520471989a1ec55bcb25b255b2ca8d2f", size = 303279, upload-time = "2026-09-21T17:54:58.178Z" }, + { url = "https://files.pythonhosted.org/packages/97/8a/7741f7c23c211fae31ab546f0ba41569dae7671cf7ca30d5afb59e9d92aa/multidict-6.9.1-cp311-cp311-musllinux_1_2_i686.whl", hash = "sha256:43124fe172ada86d03ac3c8dc8179091341f6724d5e5d5b160e1587e4cd3761b", size = 326177, upload-time = "2026-09-21T17:54:59.825Z" }, + { url = "https://files.pythonhosted.org/packages/e8/63/0cdbfbbe36316b2ef4b8e011afaca2cf291a73fe4fabf34dd0b22e3ee48a/multidict-6.9.1-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:b2f0adc22a4eb31e545221d93fc73a0f6a8cc2379d0f4309f71d1d17ba938b82", size = 323526, upload-time = "2026-09-21T17:55:01.43Z" }, + { url = "https://files.pythonhosted.org/packages/98/05/0c86e9678e78f4596df0ed7f59d2aa755107da3f2a92f1b7fa150e94cbb8/multidict-6.9.1-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:7fc59e9ba821b220944ccfe0f89c9dc4745f6d097569992356eb03869f21e953", size = 290278, upload-time = "2026-09-21T17:55:02.993Z" }, + { url = "https://files.pythonhosted.org/packages/95/8e/71bd7f43c4883cba6c5f96db88bd2311d52e74ad7680be5915938490d56f/multidict-6.9.1-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:87cc632c88ee5dc80e12681047839304d98ee5c9a708d686505767001c9b8b9a", size = 321142, upload-time = "2026-09-21T17:55:04.48Z" }, + { url = "https://files.pythonhosted.org/packages/2d/00/bf59d6bb22a4c152e4039574c29490caf3c65107715121807b4bd7cde244/multidict-6.9.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:b828cf64d62dc09ac183f03c1aeedd164ade96a2ce4934109452edf29de1dd13", size = 319045, upload-time = "2026-09-21T17:55:06.119Z" }, + { url = "https://files.pythonhosted.org/packages/3a/3c/f27f4f045198bf1c4bcd6eb3b6cf06be2dceaceb6e67222b903fce7a6749/multidict-6.9.1-cp311-cp311-win32.whl", hash = "sha256:2c1aeb92eea59d824f004341b26d5e4b47a8a441cf9726769b9a90abf9d0e08f", size = 51309, upload-time = "2026-09-21T17:55:07.531Z" }, + { url = "https://files.pythonhosted.org/packages/91/17/c2064aef64efc007bc89a571f843e5772f27946eb9bc018731bf8016e319/multidict-6.9.1-cp311-cp311-win_amd64.whl", hash = "sha256:5f89dad732280e7a10b74d40b91364f88e13c3f2c08c2ef83a8cd42f7a61af2e", size = 59148, upload-time = "2026-09-21T17:55:08.936Z" }, + { url = "https://files.pythonhosted.org/packages/45/a1/3bab1827edd813dfb90712af5cbb7fc1473ed4d9c871d103df4f4bb950c2/multidict-6.9.1-cp311-cp311-win_arm64.whl", hash = "sha256:5800368526647146978389dfaa46da3356291e9f0fff9a4ef12e8c2bef964a0d", size = 54258, upload-time = "2026-09-21T17:55:10.191Z" }, + { url = "https://files.pythonhosted.org/packages/d9/0d/4b5afb6d3e545c9af0cdfe2db8f6f4c6664568c23d863d888674e447e6a4/multidict-6.9.1-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:29138fef49828542e859828107e42e50d0e587c513b7eb4b2d92bade2b0860fe", size = 98360, upload-time = "2026-09-21T17:55:11.508Z" }, + { url = "https://files.pythonhosted.org/packages/89/ab/1b9ca66251899981b21138b87da9d5a9c2c81af12b1ea7d19466972f7fe2/multidict-6.9.1-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:19e31815d41cefc365489e591d105d2baceb2f65aa75d29471fbdbda8651e006", size = 59979, upload-time = "2026-09-21T17:55:12.866Z" }, + { url = "https://files.pythonhosted.org/packages/36/eb/6ae44062466c26c8469ef43f2481a6a48d8cea0587b2d54514ec92e2adfd/multidict-6.9.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:6ed30be8918e18c8bed0a2e8b70639ecf02feb61ed00ca2e41cfcb2a50fa3f42", size = 57736, upload-time = "2026-09-21T17:55:14.223Z" }, + { url = "https://files.pythonhosted.org/packages/b9/5c/a67817593019257a4ac8b0d1b4c426030e637c047b0692ef405439ecea7a/multidict-6.9.1-cp312-cp312-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:637f4ae36264bd7b8d9a60193acddc1d735ad52e8ed53a19931ea6d921fea8e5", size = 333097, upload-time = "2026-09-21T17:55:15.591Z" }, + { url = "https://files.pythonhosted.org/packages/90/bf/599ae2e6222822d88a247a8a7ae82fe6fd25d5700757b79603d5edafe6a0/multidict-6.9.1-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:35fc236507fb1b3138f0af5ecd5f94ed752d4d6d826248eae425f86204013eea", size = 334196, upload-time = "2026-09-21T17:55:17.015Z" }, + { url = "https://files.pythonhosted.org/packages/15/10/d8aac5acacbe7f5c117866c776ec26d5f15a868759b6f37ad8e7ed3b5b02/multidict-6.9.1-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:e58952f04772f59f11c6e007471449809a30165188669bca8fdb19dde40a8f24", size = 317465, upload-time = "2026-09-21T17:55:18.412Z" }, + { url = "https://files.pythonhosted.org/packages/19/0a/714f796f7293a8b1c5c3f465a26996231d5450c54e85d64ef1258c091134/multidict-6.9.1-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:d35a4f1c63f07fbb8c8f9946dea98b21eddf6c57421585f71d91864be3ba2a24", size = 345914, upload-time = "2026-09-21T17:55:20.306Z" }, + { url = "https://files.pythonhosted.org/packages/da/b1/e37fbf769c567be277bcf32df6234035a4384677fcc1bd852752be3d6b93/multidict-6.9.1-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:b5f8771aaaed7f80e84a4e471d2f29ab6721e4595075e54d03ae1ed951b2000a", size = 351163, upload-time = "2026-09-21T17:55:22.21Z" }, + { url = "https://files.pythonhosted.org/packages/ed/5b/db08419c1e1f7c9d60cfd2787b2b517d7ae4ebbda8281b48a33eb5141467/multidict-6.9.1-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:976fd7689d69ec78d67d31d38d396d8adb562f7e8368279f76aed4aa451fa06d", size = 336881, upload-time = "2026-09-21T17:55:23.699Z" }, + { url = "https://files.pythonhosted.org/packages/9b/07/cc9bc8a62651d2d53ab93ee4993b3a71b7cb78eb8ebfc9c757a5b6698617/multidict-6.9.1-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:95052e8777a86bae87c0bd0b5ab22d809e3d1d02bf69e3e66ddda5ba75a05805", size = 303405, upload-time = "2026-09-21T17:55:25.305Z" }, + { url = "https://files.pythonhosted.org/packages/b6/0c/8e912afafa70e944dbb8bec4b66ca6e008511395278c0d3dd0e89567536a/multidict-6.9.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:1a8adfcaf96f587ab138476eaddef95f29b8a2a8a9afbfea8d2fd62180995d02", size = 324265, upload-time = "2026-09-21T17:55:26.989Z" }, + { url = "https://files.pythonhosted.org/packages/66/6a/62c2af80fb085e6234805017af857b8e913dfac7a55e9c1349c27768c58a/multidict-6.9.1-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:1a53de2772cfb74559df2eb4456ec4eeb908435ec55a84b69370d9d745d62aa8", size = 322085, upload-time = "2026-09-21T17:55:28.732Z" }, + { url = "https://files.pythonhosted.org/packages/f8/6b/35bf801b336fd960811207203ffdcdc24acc3558b3e7ff2e1c914b141e70/multidict-6.9.1-cp312-cp312-musllinux_1_2_i686.whl", hash = "sha256:0f2ce963299d42fa3f22a90adc0fdf174792ffef5ff4c7ffb68260548fb05580", size = 338107, upload-time = "2026-09-21T17:55:30.303Z" }, + { url = "https://files.pythonhosted.org/packages/80/41/495ef65bf5bba29d142d81b3fe1b8154b919bed70e96491b1e93c3c26f0f/multidict-6.9.1-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:5b30ddf7234e611ca877575b62840e6af5977f92f1f9d532eedbb05a44ff8004", size = 339631, upload-time = "2026-09-21T17:55:32.012Z" }, + { url = "https://files.pythonhosted.org/packages/19/bd/057fdff5f4e04dcd40a960e38f77d19d3c4b67dd243ffa5f43718edb29fc/multidict-6.9.1-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:63ada7ee2e9345695f9e9bc4c65d72222253f07b1ac94fd0e37555cc6f3c7f60", size = 302080, upload-time = "2026-09-21T17:55:33.817Z" }, + { url = "https://files.pythonhosted.org/packages/3f/de/9ace933ee8dad808632523726f42255b09600087219e3d4ead7369820910/multidict-6.9.1-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:3c95601ed98fad3f6e2f8fe809c3b526b0fab31ef525e00a155e227f3d17f58a", size = 341107, upload-time = "2026-09-21T17:55:35.471Z" }, + { url = "https://files.pythonhosted.org/packages/d8/ac/7c1204406097bfc5c283d4a3287d61166807d189917567df4c1318484fc4/multidict-6.9.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:c148e8b596000dd3e4bfe206e70f3e666be18d72032e0012555f2373c52e35d6", size = 333482, upload-time = "2026-09-21T17:55:37.002Z" }, + { url = "https://files.pythonhosted.org/packages/ff/c0/a70c32ea3299ebe00f44533740cb46905717c51faf29bbd3ce8bb5886d9d/multidict-6.9.1-cp312-cp312-win32.whl", hash = "sha256:f9dad513626a33670f17cddc6078e30e311f444c957e8dbfc5b2b4603c8b4edb", size = 52548, upload-time = "2026-09-21T17:55:38.529Z" }, + { url = "https://files.pythonhosted.org/packages/d7/2e/8c9c2591df01e5692ad1bf92febb082a17b1c6c3a19aa8b7ff49988fd989/multidict-6.9.1-cp312-cp312-win_amd64.whl", hash = "sha256:a16a1dc8529f9e734a41c3b856f3eae7ebacdc061dde3f8a844e0c7889c97203", size = 59622, upload-time = "2026-09-21T17:55:39.86Z" }, + { url = "https://files.pythonhosted.org/packages/d0/95/59f4472ec512bc180fd207899594c7803ece96262ff9aaa3a4b6d7da6940/multidict-6.9.1-cp312-cp312-win_arm64.whl", hash = "sha256:361f7206cf341ba94fb015688f5c8b480f8e63bd58a4c14a48aeca7851a241cc", size = 54930, upload-time = "2026-09-21T17:55:41.155Z" }, + { url = "https://files.pythonhosted.org/packages/eb/45/ddf7c76860f5f553a23ad3ca38463ccf5247081618977742c8dc4625355e/multidict-6.9.1-cp313-cp313-android_24_x86_64.whl", hash = "sha256:d7bf9e43282d69561618e8a0ea33368d532ebef42f15c096f427090521dd74f3", size = 63266, upload-time = "2026-09-21T17:55:42.676Z" }, + { url = "https://files.pythonhosted.org/packages/6d/d3/f4ae5945ea2de597eaeb4eca3a74c59783ace54482c9b49435442b63efa8/multidict-6.9.1-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:03d47df72f084f757c1cb771188d5f4e3a805e4abc4d67e32509272343ae9382", size = 55771, upload-time = "2026-09-21T17:55:44.016Z" }, + { url = "https://files.pythonhosted.org/packages/93/7d/15468239920040d01c686e5ae669e6382f7bf31fc3e5e0a6b8f1c3b32d3b/multidict-6.9.1-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:6bc94fe17c3c56e5418f79515b786b101845f70609b0d19d0c1ba13448e5633a", size = 57116, upload-time = "2026-09-21T17:55:45.315Z" }, + { url = "https://files.pythonhosted.org/packages/17/1b/b958f06aac2d8b1e485eb1c105b1b159cbf3868249e5142142451577dad2/multidict-6.9.1-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:8f2973bbd2bebd9d2e0cd6394c1292a1a19ccd56bdcbe1e174059f1a39be5b40", size = 97692, upload-time = "2026-09-21T17:55:46.674Z" }, + { url = "https://files.pythonhosted.org/packages/93/6c/d6cfe18e61010166d7237d7527c775eb9843e7078feadea45b7628751b60/multidict-6.9.1-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:de7738b8c0bb74c4cc16bbd7fb49fc2bcf6430dba11b3432cd52768ae40933e8", size = 59556, upload-time = "2026-09-21T17:55:48.22Z" }, + { url = "https://files.pythonhosted.org/packages/46/46/9b4c1127cece0289fb207d04aae5116382ddfcb2a4dc8e0cb49a33f3c7b3/multidict-6.9.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:e5ccad4b7bac125722f48d6f862bed3b514d8526deea06316bb72f152cd30a7c", size = 57357, upload-time = "2026-09-21T17:55:49.998Z" }, + { url = "https://files.pythonhosted.org/packages/bc/fc/c18b07100a6064573e49e538d13d47eb2b5221f48080512df78aef54ab04/multidict-6.9.1-cp313-cp313-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:a49ff5cdb33654cb7d6a3c377aa2a83ddefaa1db31eb10bcf3c180aa84f9af8a", size = 330433, upload-time = "2026-09-21T17:55:51.435Z" }, + { url = "https://files.pythonhosted.org/packages/62/ff/52a0082adeb69656b634609d4fb0ae65456deaae4cca3d5a12656626dc50/multidict-6.9.1-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:b22ff30006a2f28f8bff878fb93413cbe3a4d1fd517c28081d848c90e9cfd2c8", size = 332487, upload-time = "2026-09-21T17:55:52.937Z" }, + { url = "https://files.pythonhosted.org/packages/39/3e/80bd4729635a7af371ff4dcc312594cf96e73f98854491be8740792de430/multidict-6.9.1-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:3adf06c66041aa21eeb8a71e82379b74773298c8e6d3d839b151aae441a99b94", size = 316911, upload-time = "2026-09-21T17:55:54.751Z" }, + { url = "https://files.pythonhosted.org/packages/15/f4/7ea4a907e053e924e2e1694d90560f79f72056504bfd5e8483695f9527d2/multidict-6.9.1-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:963a8d8f97057082679523d0fd4c53a38f86bc58cabe4556faef682ae53fa2fa", size = 344950, upload-time = "2026-09-21T17:55:56.5Z" }, + { url = "https://files.pythonhosted.org/packages/0d/16/9642ae41fbfbdc546aa481c2b07ae1bc087a94ab5f5020b4d9ab77da401f/multidict-6.9.1-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:b66ccc5c2cdd26e74fa5d4c29ffae424cc6148bf93ce574821783fb3b6d452c5", size = 350264, upload-time = "2026-09-21T17:55:58.073Z" }, + { url = "https://files.pythonhosted.org/packages/93/f2/e06c8e42074d0a4b8419bfe92afbd1d262190b69549ecbcd2dffa4f103c2/multidict-6.9.1-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:7fd79c521f6290c69125fa2b85fa65d9e657e6a8ffaf722dc881b926bef4aa5c", size = 336213, upload-time = "2026-09-21T17:55:59.723Z" }, + { url = "https://files.pythonhosted.org/packages/98/43/d7cf9ef4700e7c25244590d1b4f20cf12cbed86beba6c17be4d9e48c1099/multidict-6.9.1-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:1d804e4caf5d5da37d6dac1325da5629ebef1e27a294c2b568b295814aa36c7b", size = 302555, upload-time = "2026-09-21T17:56:01.666Z" }, + { url = "https://files.pythonhosted.org/packages/7d/3a/54209920324bc3928f5f2ba28b01a6dd957a9d48c3aaff77e38faaf5668d/multidict-6.9.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:bb0d664505f4b112f384cffeee82e91e3f6448d8e574989479db4439b68cba05", size = 322247, upload-time = "2026-09-21T17:56:03.265Z" }, + { url = "https://files.pythonhosted.org/packages/33/50/df96f961b178b621ab0adbf9221d4fd21534e1064fd9f5e85c1fa27fea55/multidict-6.9.1-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:9e14d17773b1b3c758ff153659a1824608a0cb562c45f484b5ed8a433428444a", size = 321787, upload-time = "2026-09-21T17:56:04.894Z" }, + { url = "https://files.pythonhosted.org/packages/a7/9b/37f354562a8f82f9c1f63d3a94300fc82126d50320175f0702bbd544f87c/multidict-6.9.1-cp313-cp313-musllinux_1_2_i686.whl", hash = "sha256:083735b7f395894e43adb278d5dae901448883a835ff8f1977e285fefdb10418", size = 335498, upload-time = "2026-09-21T17:56:06.783Z" }, + { url = "https://files.pythonhosted.org/packages/1e/44/78e366efc6c004185295cc9eb9711c10f13eadf3b547603b9c5526329976/multidict-6.9.1-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:b65121091567847a8cb520d364ab22ba90e00d3cc55fa9eb34bb439f0684bcd1", size = 338244, upload-time = "2026-09-21T17:56:09.162Z" }, + { url = "https://files.pythonhosted.org/packages/ad/2b/ab5bd3964691abe4d14b43bdbdb8a621b77cea2ca352dc93159ec0a4575b/multidict-6.9.1-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:ea999ae6e80e66ad5eea287860951b033d0104ca34d6d87c7b5125ebe0e12721", size = 301641, upload-time = "2026-09-21T17:56:11.227Z" }, + { url = "https://files.pythonhosted.org/packages/ea/a8/f6bc899f5aafaba755edc8e4bb934bc00b1e405f8d3fbc1dd10283738074/multidict-6.9.1-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:095d900c242e00fbe5f321ee072e7278b4153e78c5ce9c1efde167d62c1e4771", size = 340162, upload-time = "2026-09-21T17:56:13.076Z" }, + { url = "https://files.pythonhosted.org/packages/eb/03/a8fc809ef364b8c231c065ea2ea2f97ae86d4f9a0e36b5721759fd606778/multidict-6.9.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:6441cc837aea58be7d9baef1b2383eb8311ab9303f500f99ac90b584cd78bb14", size = 332826, upload-time = "2026-09-21T17:56:15.175Z" }, + { url = "https://files.pythonhosted.org/packages/f2/27/32dde80245e2024e9bb7a6ca3c2fb7c695dc74014cba557cc35be6f91e24/multidict-6.9.1-cp313-cp313-win32.whl", hash = "sha256:9c4880d017555d70dea367dd49271830842d48e3891c2da97da7ce8c4abcee40", size = 52378, upload-time = "2026-09-21T17:56:16.959Z" }, + { url = "https://files.pythonhosted.org/packages/be/78/1bac2987edac273a6b27fa50e9204bc3466c94a7b4f6a44828d730f8810f/multidict-6.9.1-cp313-cp313-win_amd64.whl", hash = "sha256:ac51cd64bae51c462ea58ad2492c9b8209667a4ef60c45c4a304518b67598d5d", size = 59449, upload-time = "2026-09-21T17:56:18.336Z" }, + { url = "https://files.pythonhosted.org/packages/e1/32/2a77ce19eea48cb9b3521a202b513ba3349aafa702c8317be0b645ffb74b/multidict-6.9.1-cp313-cp313-win_arm64.whl", hash = "sha256:37a9ebe00c698279213d56e6c64e1962ab1e092918270649b397cac3dc196ca4", size = 54713, upload-time = "2026-09-21T17:56:19.774Z" }, + { url = "https://files.pythonhosted.org/packages/be/59/e26cb779be4c591d1a910f59d29aca9fba4de70349840a833beba2652371/multidict-6.9.1-py3-none-any.whl", hash = "sha256:7bf6478188f4e47bf5686e8a33da4ae28bf43b1b2528d9ee144d28492bfac60b", size = 19176, upload-time = "2026-09-21T17:59:03.501Z" }, +] + +[[package]] +name = "oauthlib" +version = "4.0.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/7a/d8/a1bcc8ba112a627f8ffbdc212a78ce18d3ac07e91a5ca65d27918eee25a1/oauthlib-4.0.0.tar.gz", hash = "sha256:efb274799819440f95b4ab3b818869f1ce9ae26c5beacba0201d1a1b76b54f86", size = 187232, upload-time = "2026-09-28T06:01:18.77Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/d9/f4/78229a1066068ca14fc60fb26cf7381cabe4382261392b90e5f9552722d4/oauthlib-4.0.0-py3-none-any.whl", hash = "sha256:624c28c13a0a59cabf9747dfa52af63be3e512a7f2714df16e91b5b3a145e6cd", size = 159715, upload-time = "2026-09-28T06:01:17.008Z" }, +] + [[package]] name = "packaging" version = "26.3" @@ -678,6 +1069,66 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/54/20/4d324d65cc6d9205fabedc306948156824eb9f0ee1633355a8f7ec5c66bf/pluggy-1.6.0-py3-none-any.whl", hash = "sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746", size = 20538, upload-time = "2025-05-15T12:30:06.134Z" }, ] +[[package]] +name = "propcache" +version = "0.5.4" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/b3/9a/9fbf4e4ec0c2d7f1c32519fff782ef467859b8faa9fbc5331a96f6395d43/propcache-0.5.4.tar.gz", hash = "sha256:ff6b113f50bc066a698db5d944d2c6dc7507168dd3341e255a8892fd0715a558", size = 61545, upload-time = "2026-09-16T00:17:14.386Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/1e/40/14b21e505b7921617466576423f188a5c9caddfdaa1cf4b2b8a83d8fe216/propcache-0.5.4-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:897d1ddf6716e8f47200f7aad9a0efa6cc7586df66c6defa572f9eab379c078e", size = 86393, upload-time = "2026-09-16T00:14:06.9Z" }, + { url = "https://files.pythonhosted.org/packages/e7/4b/5a52e1a7b43563f7d408814194bb23cc8bf214eb6b86639b667a33a8d0d0/propcache-0.5.4-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:9cbfff4423eef4cc6cafc021469641a2b835f610b2647a6c5281903e21b8670d", size = 50431, upload-time = "2026-09-16T00:14:08.025Z" }, + { url = "https://files.pythonhosted.org/packages/05/cf/b5248180bf056cc76acc60c9c6e8c0ebbfdbd1c6cffd31fd14996927b7c8/propcache-0.5.4-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:3fc24f209c1b7f7f688b66b98293954f5504279760999b58920ee12dd8471c1d", size = 52116, upload-time = "2026-09-16T00:14:09.114Z" }, + { url = "https://files.pythonhosted.org/packages/86/a8/7c6cd6bfead1a11f2e411e688640e6d26574cb0bde7dcaa7423b0b65ed7a/propcache-0.5.4-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:62530ca89187827e4a4fe733f971abe81a7542eeea48ff61995f19b64d7199c8", size = 238729, upload-time = "2026-09-16T00:14:10.357Z" }, + { url = "https://files.pythonhosted.org/packages/5c/b4/442715b2e980df51be52d203549279e027728f24c80b00b5e525e31cd5ea/propcache-0.5.4-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:56fc3f7599528db40b1efa0889a620116e2704144495273d66066e8164e45838", size = 246121, upload-time = "2026-09-16T00:14:11.735Z" }, + { url = "https://files.pythonhosted.org/packages/bc/5d/df0684fc2b1732a01a7bec26d7897369022712422d25b09c37ce7dbc88a2/propcache-0.5.4-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:4f2d880ff60f45898f4acfa152aac8d04e3ee627d90ff4003491bf92239d5757", size = 251735, upload-time = "2026-09-16T00:14:13.053Z" }, + { url = "https://files.pythonhosted.org/packages/c7/06/519a5ebb48b6f94beb48396e55c905f12246a25c3a3608a7ec7bceabf50e/propcache-0.5.4-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:6e9368e87a3efc285e559131092c5db643eb8e56de4ee42064d5baec22ef2bb5", size = 235381, upload-time = "2026-09-16T00:14:14.398Z" }, + { url = "https://files.pythonhosted.org/packages/3c/07/1e0a9bb310830f2245edbd5cd3c6d24a783c053c4efd8e08e386e513c940/propcache-0.5.4-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:004e685b315646c410771836e72a44f143bbe624f29653a42687815069a303d5", size = 208973, upload-time = "2026-09-16T00:14:15.715Z" }, + { url = "https://files.pythonhosted.org/packages/62/5c/9324fab27d6088eecc47fe4332bf7aaf8c1ded93c36f558391e8a06d41a7/propcache-0.5.4-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:594eb4c6ec35e7179b058481f4e9f02521b56de16fa577c4b85c76fb1bf8a9f8", size = 233897, upload-time = "2026-09-16T00:14:17.25Z" }, + { url = "https://files.pythonhosted.org/packages/9c/a3/570d92fc952eae93b676f3a1568f4b89264102abd3c982ab6a9ebec58dcf/propcache-0.5.4-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:2dba2f02d2d5c09ef8a0e6c1a42aeaa451f4be9898cb00b04fe98717da2eb23b", size = 223512, upload-time = "2026-09-16T00:14:18.87Z" }, + { url = "https://files.pythonhosted.org/packages/65/47/26810d889d89bba31db397e6a88f8984af775f5ed6bad0a29dce84324cff/propcache-0.5.4-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:c3ef2818d63bc86071e9d2989ae75a1bc32b8f7059cfd9f5abbbee70c32e2ed6", size = 239043, upload-time = "2026-09-16T00:14:20.366Z" }, + { url = "https://files.pythonhosted.org/packages/89/2d/f9c47691aa024c8299a3afacd78d22a01ab57eb627b481b6089708e71017/propcache-0.5.4-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:dd2ac8f5b643454c2cc6b6118b13da16e88f4a6434fc3ba61aca384029f04f36", size = 208218, upload-time = "2026-09-16T00:14:21.801Z" }, + { url = "https://files.pythonhosted.org/packages/74/6b/d510c0c378cabbf9d0ac7b663af6d00f2e9074073d93b20c85f24aa5c071/propcache-0.5.4-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:4054acf80d40456a0537f2913b349718649d8d6458a14ab7f48d0ce28c30869d", size = 240301, upload-time = "2026-09-16T00:14:23.121Z" }, + { url = "https://files.pythonhosted.org/packages/3d/80/c80f6adaaa1e51f0db2dce8c9b3714d94ec45e358a21f9a1910b10b40a80/propcache-0.5.4-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:40e94adb1e7d39ff28a8bd8d8b8fbd1df6b9f40976dbe379134f1ce058e532dd", size = 230785, upload-time = "2026-09-16T00:14:24.458Z" }, + { url = "https://files.pythonhosted.org/packages/d9/e2/c32a7df3f39caa7f11b2eb37ea5b6960a6946f2bc7c4b8ff97bbdf6d6b6e/propcache-0.5.4-cp311-cp311-win32.whl", hash = "sha256:9f86f7259efe2c951f43e57d471c9b41daa5bfc7db9f67189059cf1ae6d77fd9", size = 42747, upload-time = "2026-09-16T00:14:25.715Z" }, + { url = "https://files.pythonhosted.org/packages/0a/8a/3db6a3543d8101263b4c52978b6276a04ead2caff2c5ab880d934f47bd89/propcache-0.5.4-cp311-cp311-win_amd64.whl", hash = "sha256:e904d4d01f36bd6e197590be1533c44e06058771e0746dd073a8ebb3ef880858", size = 46268, upload-time = "2026-09-16T00:14:26.996Z" }, + { url = "https://files.pythonhosted.org/packages/41/07/5222e2665bbf6e45847492ecbf3b9f3e4975a0ae300e5fd465df7d48ce55/propcache-0.5.4-cp311-cp311-win_arm64.whl", hash = "sha256:d42a9a856a4a6e2f6c10f1318c07e7daa498d6593abe745c71dae4521a26ca39", size = 43547, upload-time = "2026-09-16T00:14:28.143Z" }, + { url = "https://files.pythonhosted.org/packages/71/cd/348d58f142aebc4873345c6b31087629182ca6e0f2b3caeaa528cf882eba/propcache-0.5.4-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:b28f41fa3b8c6900457f858ec5b03998f3a6d535fbc1bb2edec5961ea05ec429", size = 87285, upload-time = "2026-09-16T00:14:29.362Z" }, + { url = "https://files.pythonhosted.org/packages/df/f4/f3ffaee281b276da854ac1d7a6a506d26cbc62ea2e623756f1d0a4a1ba1a/propcache-0.5.4-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:dcbf346a318a5e30063f547630b02bb787ce2f45b6368d5da143660b6a3835d8", size = 50984, upload-time = "2026-09-16T00:14:30.473Z" }, + { url = "https://files.pythonhosted.org/packages/25/88/1d7df7201750b37765ef2b23bc1c526c028dadde80afa0f57a118fc01182/propcache-0.5.4-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:87a3caecf8095e48dc72f84bfa42e23a848cf410cc9cc13031fba4869b706a21", size = 52460, upload-time = "2026-09-16T00:14:31.692Z" }, + { url = "https://files.pythonhosted.org/packages/83/4f/48865bd02a16ee5236bc46166b2946f37b93e07b0eae355dac0be0b216ca/propcache-0.5.4-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:60a64cbccaa11b7760ce705a14ada17ba459e7ca9f23ba587eb013821032d7ef", size = 251768, upload-time = "2026-09-16T00:14:32.908Z" }, + { url = "https://files.pythonhosted.org/packages/b0/19/3742a5eed62317b03b4002ee865dc9fd720308bdd0da1f29a5786c630311/propcache-0.5.4-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:a74bfa37147cc08fb29df10bd9c16f40fa7f860cd3a6d2fff853323a94f6e17f", size = 257723, upload-time = "2026-09-16T00:14:34.267Z" }, + { url = "https://files.pythonhosted.org/packages/cb/d5/ee6350fb0be9122bb6c67082a876d34b90d980d100c106af4b81023e04f4/propcache-0.5.4-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:a4d7a54719b67338a305dca2ce6aafe366817df94ddfd4b5514374356f5ca546", size = 265597, upload-time = "2026-09-16T00:14:35.56Z" }, + { url = "https://files.pythonhosted.org/packages/85/9f/83a07b6ec0e043c050cfdd35fb0cf1b7897b91d554d6eea293740309afe7/propcache-0.5.4-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:2814ecd8e818f487bee4b0f921bc4d1c176cc5fc71ac0f072d0fa67eda4ac14b", size = 250424, upload-time = "2026-09-16T00:14:36.894Z" }, + { url = "https://files.pythonhosted.org/packages/33/2c/a763a8251f50fba042af0fb1f02bfec4b31381e40aff760db2be7b2e1f84/propcache-0.5.4-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:6af4693716bfb03f1752ef1b30faa593db2c01d5272e9b8564a1549452a979ab", size = 216748, upload-time = "2026-09-16T00:14:38.369Z" }, + { url = "https://files.pythonhosted.org/packages/6a/e2/4d11bea8fd6a777149c6c20645f873952eab5de3a2497aa11648ec9ab6ab/propcache-0.5.4-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:4fbc1a15dc8cd1689508758d626b372b1f09d28d9577667feaf9e6bfcd8efcbc", size = 246533, upload-time = "2026-09-16T00:14:39.82Z" }, + { url = "https://files.pythonhosted.org/packages/9f/36/6683597de4907e70c717e3588c541202c66086a72ff3db58be49de66e72c/propcache-0.5.4-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:cdee8205a44d0be91bbac4c41b95d86641b72dfc7aef1279400e4fda3f26a937", size = 238173, upload-time = "2026-09-16T00:14:41.259Z" }, + { url = "https://files.pythonhosted.org/packages/85/84/cb08d79f1762daafeb2b030c470cd0c725c97b8ad67412457c6f35c53e9d/propcache-0.5.4-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:9a2a8a50a93dee0268a860a07fa3b4bd968f8ce4dbd794957da772f395368526", size = 251128, upload-time = "2026-09-16T00:14:42.652Z" }, + { url = "https://files.pythonhosted.org/packages/c2/0d/41b848036db6621370c1f2e5471a7da8149c730f8552a5257567721f4576/propcache-0.5.4-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:7ffafcbfc7b549ab940047e505c831eabac5e67de53e1bc174adbc5285c55944", size = 214821, upload-time = "2026-09-16T00:14:44.112Z" }, + { url = "https://files.pythonhosted.org/packages/f1/b7/adfae4bf9c63bccf12e2d9690a175c6579047a6eec3b5a6a5f51428c15e2/propcache-0.5.4-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:d1f5a500bfcbb2c0ab85e98a0dcd70f5899d34efe365a0187700369a79603031", size = 254793, upload-time = "2026-09-16T00:14:45.429Z" }, + { url = "https://files.pythonhosted.org/packages/51/6f/eeca9647245d5f92e87d53e5f14335bb42fce1a7e6842c8045b364eded8b/propcache-0.5.4-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:8a235f73d6e020855dc29dff012d920c02ee0feab8d73a24185a7569f4be1161", size = 247134, upload-time = "2026-09-16T00:14:46.976Z" }, + { url = "https://files.pythonhosted.org/packages/5d/a9/424e38838793d37160b4379c702f61c74c598fc6cd17204adbe3c554f7a8/propcache-0.5.4-cp312-cp312-win32.whl", hash = "sha256:b3083bfe87f95c756e610bd8025f26cbd1cd4aaa03a422f2d65efb7a97cd53d8", size = 43073, upload-time = "2026-09-16T00:14:48.338Z" }, + { url = "https://files.pythonhosted.org/packages/58/7b/6e8ef26f6d510a7916064fec68d55fcbfbdf7eb01e377480d66a122152d8/propcache-0.5.4-cp312-cp312-win_amd64.whl", hash = "sha256:98914de2c4d7f0f9f4a8c6ea4bf05841f4175796941e3ef7d47eb718f22311fb", size = 46190, upload-time = "2026-09-16T00:14:49.99Z" }, + { url = "https://files.pythonhosted.org/packages/08/b9/72028c5b56ced97f456de6aefa79435ca64d7f77af78ea8cf3c76fc5195f/propcache-0.5.4-cp312-cp312-win_arm64.whl", hash = "sha256:8876b39961e33d912afe3c1bee18ee564fdad0206f873cc15d522756b7f50737", size = 43075, upload-time = "2026-09-16T00:14:51.155Z" }, + { url = "https://files.pythonhosted.org/packages/78/4c/3b1365d58a667689e067e13d055fcd92bdf8d9a2fca3d9201b47ed5b3631/propcache-0.5.4-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:36c0d9db44b523ef93d03341b1c42d69ff01d673c053d1b1c6c3a363bcaa39ba", size = 85290, upload-time = "2026-09-16T00:14:52.342Z" }, + { url = "https://files.pythonhosted.org/packages/8f/61/5f9c29c3aa67c30238c4eadf95149b1d983a48f69b86b0cff927a7d6df13/propcache-0.5.4-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:e1d52a05dc417279f7e5c7618c5dfbbc29923aaf9bc0a5c1802ddcebf54c61a0", size = 50027, upload-time = "2026-09-16T00:14:53.67Z" }, + { url = "https://files.pythonhosted.org/packages/25/7d/c1ab1ef09e9d4d835be5d58c0a32a1e1de8397abaa4e502a9d4141328cad/propcache-0.5.4-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:44149f46500a0a41b95b4d99c2e586a77319539730607b9892974a092788b111", size = 51425, upload-time = "2026-09-16T00:14:54.826Z" }, + { url = "https://files.pythonhosted.org/packages/73/36/0093091ebb270fcd1bc1f6e095f93b2e0ed7f1011c28837dc2dbe5f96b99/propcache-0.5.4-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:dbab5f5ff6897c81f355d079010cdae85b02e5a0b518b5251523b8ad8ae9ac3c", size = 233595, upload-time = "2026-09-16T00:14:56.09Z" }, + { url = "https://files.pythonhosted.org/packages/ae/8f/0de9d4c8e05ce0be71b436919a216bd7fc5cc6e2691c0602295efb22b9ed/propcache-0.5.4-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:c3e98c55bde2bcf7db3c70d1aed7ae9aa8aebbf19a250c66645cde44cdb8b867", size = 240318, upload-time = "2026-09-16T00:14:57.674Z" }, + { url = "https://files.pythonhosted.org/packages/7d/71/2b35e91455209b85ee98f7859583e0814fab57d3af0f2381aaee34c37304/propcache-0.5.4-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:db3ae52ccc150dbc84704e9d642743897f3e1c54742ff34cacb661e52e3818a9", size = 246649, upload-time = "2026-09-16T00:14:59.352Z" }, + { url = "https://files.pythonhosted.org/packages/ed/74/08e6c1faf26ee2732023a3828787ba535557122774f4a386b1f715cbd8e0/propcache-0.5.4-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:f85915e00dcb1cd9f2f890ead064ed40a27df06f0db65be427b29482ae357572", size = 234316, upload-time = "2026-09-16T00:15:00.696Z" }, + { url = "https://files.pythonhosted.org/packages/5c/9a/08385733c9321c9bb78039d3ff31045e4fca962d9665023c4eb70f998819/propcache-0.5.4-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:c2ba30a89035b57b73e00475de948521602f543d79ce01db10b04b36c4c76fc8", size = 204666, upload-time = "2026-09-16T00:15:02.019Z" }, + { url = "https://files.pythonhosted.org/packages/1d/f4/e87bc7629af9a14a752b218764a78742d73c2c563ac58315da6841f0cbe4/propcache-0.5.4-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:ae58f361bd5dae942717c65d3413b478c70aea9c462599e7b9adad3731db3894", size = 225900, upload-time = "2026-09-16T00:15:03.394Z" }, + { url = "https://files.pythonhosted.org/packages/d9/6d/11014938d3fe9bea2ea2dcf930f26ed565bfb2f5be3c756362ea48c92636/propcache-0.5.4-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:96f7c5c15656040ddcbc51e56dc59b58aa25999d743c126abd425b9766ab43e9", size = 219988, upload-time = "2026-09-16T00:15:04.811Z" }, + { url = "https://files.pythonhosted.org/packages/dc/72/fbf17c589f92c0b3bbf6709a425661f8ef2ed0d46b38985a7d7b5a0f6b91/propcache-0.5.4-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:7cc528e760a8af06f2b13e9b9f362cd90c7c718ea61228a96dbd31ba16ed7f47", size = 233611, upload-time = "2026-09-16T00:15:06.498Z" }, + { url = "https://files.pythonhosted.org/packages/55/7e/dbd637572a279692e5518d117274a9331bf5faac59f191d30e82521a3ec7/propcache-0.5.4-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:425f8cc86ab5018b4b8d4a23bc8e74d964bd3d757c3702e301aa79be76c53f6c", size = 204333, upload-time = "2026-09-16T00:15:07.961Z" }, + { url = "https://files.pythonhosted.org/packages/ba/5a/f99c92068f1e0f5c886899ce0e4a619db376ca98c5279d93f95bd86906af/propcache-0.5.4-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:a5793c7698a53f56f4a1889a4737c7eeb1b7ad0842fa6b1abca22913ff79c8c1", size = 235177, upload-time = "2026-09-16T00:15:09.334Z" }, + { url = "https://files.pythonhosted.org/packages/ee/28/95456fabd2daf6be89049a13fbf03341756014d2959c83d12957d4c49694/propcache-0.5.4-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:c02c0e570c5c7e077b0181a9f3cdb7d4c3617d1cda6b5c95bd5d34022923d82c", size = 228982, upload-time = "2026-09-16T00:15:10.729Z" }, + { url = "https://files.pythonhosted.org/packages/b1/bb/df90f62c9cf7c93ea235f6f9405143bba802914607317266dd81fc8d737e/propcache-0.5.4-cp313-cp313-win32.whl", hash = "sha256:3e413d7a4a9b4866b7a761d6060d434b64d23cd35122eda3b026a0bbe8196b25", size = 42611, upload-time = "2026-09-16T00:15:12.111Z" }, + { url = "https://files.pythonhosted.org/packages/01/bc/e0a7b84af04ec02d73a48aa71f091e1e4a2107e3074b7ce12195b66901f4/propcache-0.5.4-cp313-cp313-win_amd64.whl", hash = "sha256:0c889f6fa84957bc7e8b4eab71fd16a0455068d5045e3aa40c733071d2b2fd77", size = 45342, upload-time = "2026-09-16T00:15:13.519Z" }, + { url = "https://files.pythonhosted.org/packages/9a/70/50b031cafe72a5c1878b903ee87303f71313345566bf3d6ec202e5ddc9ec/propcache-0.5.4-cp313-cp313-win_arm64.whl", hash = "sha256:69fc35c0779522da366c563e5faf203ffc1f8ff0021d5b1337fa4efa5be73177", size = 42408, upload-time = "2026-09-16T00:15:14.788Z" }, + { url = "https://files.pythonhosted.org/packages/f5/cd/785c64ed382f3f04201870267b02783f63b4678c2acfddc177a3ebcc2727/propcache-0.5.4-py3-none-any.whl", hash = "sha256:62c60aec739ed00124573cce1178138fd690c7676352d67a37328c1cf51d7468", size = 16338, upload-time = "2026-09-16T00:17:13.106Z" }, +] + [[package]] name = "protobuf" version = "7.35.0" @@ -854,6 +1305,34 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/73/e8/2bdf3ca2090f68bb3d75b44da7bbc71843b19c9f2b9cb9b0f4ab7a5a4329/pyyaml-6.0.3-cp313-cp313-win_arm64.whl", hash = "sha256:5498cd1645aa724a7c71c8f378eb29ebe23da2fc0d7a08071d89469bf1d2defb", size = 140246, upload-time = "2025-09-25T21:32:34.663Z" }, ] +[[package]] +name = "requests" +version = "2.34.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "certifi" }, + { name = "charset-normalizer" }, + { name = "idna" }, + { name = "urllib3" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/ac/c3/e2a2b89f2d3e2179abd6d00ebd70bff6273f37fb3e0cc209f48b39d00cbf/requests-2.34.2.tar.gz", hash = "sha256:f288924cae4e29463698d6d60bc6a4da69c89185ad1e0bcc4104f584e960b9ed", size = 142856, upload-time = "2026-05-14T19:25:27.735Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a0/f4/c67b0b3f1b9245e8d266f0f112c500d50e5b4e83cb6f3b71b6528104182a/requests-2.34.2-py3-none-any.whl", hash = "sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0", size = 73075, upload-time = "2026-05-14T19:25:26.443Z" }, +] + +[[package]] +name = "requests-oauthlib" +version = "2.0.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "oauthlib" }, + { name = "requests" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/42/f2/05f29bc3913aea15eb670be136045bf5c5bbf4b99ecb839da9b422bb2c85/requests-oauthlib-2.0.0.tar.gz", hash = "sha256:b3dffaebd884d8cd778494369603a9e7b58d29111bf6b41bdc2dcd87203af4e9", size = 55650, upload-time = "2024-03-22T20:32:29.939Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/3b/5d/63d4ae3b9daea098d5d6f5da83984853c1bbacd5dc826764b249fe119d24/requests_oauthlib-2.0.0-py2.py3-none-any.whl", hash = "sha256:7dd8a5c40426b779b0868c404bdef9768deccf22749cde15852df527e6269b36", size = 24179, upload-time = "2024-03-22T20:32:28.055Z" }, +] + [[package]] name = "six" version = "1.17.0" @@ -910,3 +1389,86 @@ sdist = { url = "https://files.pythonhosted.org/packages/ba/19/1b9b0e29f30c6d35c wheels = [ { url = "https://files.pythonhosted.org/packages/ce/e4/dccd7f47c4b64213ac01ef921a1337ee6e30e8c6466046018326977efd95/tzdata-2026.2-py2.py3-none-any.whl", hash = "sha256:bbe9af844f658da81a5f95019480da3a89415801f6cc966806612cc7169bffe7", size = 349321, upload-time = "2026-04-24T15:22:05.876Z" }, ] + +[[package]] +name = "urllib3" +version = "2.8.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/e3/05/b17359e1cefb4f909b5e40b1b90a496d987258916dbbf88e842c729f510e/urllib3-2.8.0.tar.gz", hash = "sha256:63bf2ead4c879426ebf22ef2a781eeb4aa3b4ae798a0435506f8687fd5bb9b63", size = 458972, upload-time = "2026-09-15T19:29:36.253Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/92/9d/c4e665119135114480843e7ab388fa94d8480650450e6f8e26b70d323a4c/urllib3-2.8.0-py3-none-any.whl", hash = "sha256:0cf3cae568d36aa9576b28dfb35f11328f1cb974ca7647d9475ebb86c75ac6e3", size = 135717, upload-time = "2026-09-15T19:29:34.577Z" }, +] + +[[package]] +name = "websocket-client" +version = "1.9.2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/d8/cb/a5abcc2891249f393827c650c6296660ce40374ac22d99ab9aea41f9d2a2/websocket_client-1.9.2.tar.gz", hash = "sha256:0fcb57545848be86992e128218fd96dd87a6769ffdb1a968dff79632b85604d0", size = 84110, upload-time = "2026-08-31T14:08:40.964Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/d5/d2/cc4dc1271e464942db7ee278baae2daa99ee77cb2af744025c04da585a3e/websocket_client-1.9.2-py3-none-any.whl", hash = "sha256:e1a673830a9c7bfa47b1cd3d5e4178f4c9651d80a4eab02c9c23a1c3ec6250ce", size = 95786, upload-time = "2026-08-31T14:08:39.899Z" }, +] + +[[package]] +name = "yarl" +version = "1.25.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "idna" }, + { name = "multidict" }, + { name = "propcache" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/75/16/e8be8e2fb175bbf41a0680381a319f1199fae256588241a2ac8677eafb49/yarl-1.25.1.tar.gz", hash = "sha256:03dd38de09bc213e9a8b29761eec33ee1d5318dac0e49d8af36e4d27830e23a7", size = 246245, upload-time = "2026-09-15T19:35:02.264Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/83/b3/2cea721d495ca57f8f414aa4867ee263f486b274e07463158cf52f02ba7f/yarl-1.25.1-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:9d693bf4bf534e9ba3ae2780cfd577f5135629f7b5ac653490859d0b77864865", size = 143794, upload-time = "2026-09-15T19:30:26.946Z" }, + { url = "https://files.pythonhosted.org/packages/55/e6/cd145cff8e5cf60b8b3c41fbecfa2a45028a8dec3fbc52bec03595ff3d3b/yarl-1.25.1-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:ab2054c5531af2a9ba7b69b8ec91e4f884420e83a8c5e579b013084cb57e5e5d", size = 103993, upload-time = "2026-09-15T19:30:28.661Z" }, + { url = "https://files.pythonhosted.org/packages/ce/7c/fbb40fe2d53747c40aa36a9e2bd2178a202f942bda0b670f3306d4aefbce/yarl-1.25.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:564fdc7085d2245ab84f88882fdb1d6ac0723124bff6ded35bfb1c00f812630d", size = 104010, upload-time = "2026-09-15T19:30:31.069Z" }, + { url = "https://files.pythonhosted.org/packages/f0/0b/5a516f70641092283f57cf3670bdb75e7327bcc0dcb5038697e4dfbfd569/yarl-1.25.1-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:acae6b45d1ace09b6ba3876da43b88366ef368f73b988c7f57e14231753d4420", size = 116444, upload-time = "2026-09-15T19:30:32.894Z" }, + { url = "https://files.pythonhosted.org/packages/b6/8a/d1a627f827b0a404ae0f5647cab0534959081cde13b0756a783469fb3b5e/yarl-1.25.1-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:1fb2a01ba8cd9c5d2c5dc1ec35e0fc951d04b4f037541d4ac090c993ce58b3d7", size = 107565, upload-time = "2026-09-15T19:30:35.188Z" }, + { url = "https://files.pythonhosted.org/packages/aa/cf/c8e0aaec886840a6c4480fe44eaec7cd4319f79a563b79321473be79c56f/yarl-1.25.1-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:e92b6bcc741b86d67606c40d3cb9c7cc8e6c737f81e31f4a94efc204456c92e3", size = 125006, upload-time = "2026-09-15T19:30:37.1Z" }, + { url = "https://files.pythonhosted.org/packages/50/26/0cce366d54a93cdc8342965dc7663385e4db161625cc1e8b18e786753d24/yarl-1.25.1-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:72c34ac7ad4314c19362d5ce27626dcc8429bd30bbf8c179f4234078851f9492", size = 128717, upload-time = "2026-09-15T19:30:38.904Z" }, + { url = "https://files.pythonhosted.org/packages/95/0c/a71501bbc1a674ff72c4d6c2b75f4d9a5af819f5244c3a7558080a8802c5/yarl-1.25.1-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d5add7b4ca7afeea91d52e4d4e4db3b1fe9885b71f07054560d8c4296b7441a2", size = 117728, upload-time = "2026-09-15T19:30:40.809Z" }, + { url = "https://files.pythonhosted.org/packages/e7/6f/c3267ca01defeed9ed9c4ff9b17bd54915432c405945233265b707475d1c/yarl-1.25.1-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:def538065f9e4d4cf1ae164bd59aba00dfa84f03923e0de4c3788f252d6bcd17", size = 116224, upload-time = "2026-09-15T19:30:42.818Z" }, + { url = "https://files.pythonhosted.org/packages/8f/69/fad57ee52d648431718ee0f1f68966a99c1352c3924688ffbdcc9d3fe51a/yarl-1.25.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:0a191bfdb30a79b98e5d175d75285f9fcb78bf0e46ba5efda042e1c72071a0de", size = 116299, upload-time = "2026-09-15T19:30:44.966Z" }, + { url = "https://files.pythonhosted.org/packages/2a/d4/6a8c1e29f33338687ca278cba0a8fbf6525a322c2c02a9a500ccbe041152/yarl-1.25.1-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:71f42c5b9a948c113bbdebfa544598321431d064ff959d32e99b1feb61d68345", size = 108625, upload-time = "2026-09-15T19:30:46.874Z" }, + { url = "https://files.pythonhosted.org/packages/e8/d2/3a35ae791c9cb6522c106923ff25c3230d999091e5e65511bca23bbd9914/yarl-1.25.1-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:72849d892954be4d09e569b8b831ac39ce58417fedc767d4308a0fe542018a40", size = 124515, upload-time = "2026-09-15T19:30:49.082Z" }, + { url = "https://files.pythonhosted.org/packages/be/fd/2b022109a6b4af0f7dc371cf7500af380b0d4f034010243e1b0ce218dc93/yarl-1.25.1-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:efb01a106f971cb3752856bca2318bbdf7f01bd8823779c461586cbe5ffd5258", size = 115711, upload-time = "2026-09-15T19:30:51.208Z" }, + { url = "https://files.pythonhosted.org/packages/18/59/f7586271136c3ddb0126bbfe661844699369b76b555fe38c4efe86870b2e/yarl-1.25.1-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:a1daf47cd95a7c3a63456336bc5aaa8c86dd3a47d07ed3d0e76132ae4666a5a1", size = 122751, upload-time = "2026-09-15T19:30:53.535Z" }, + { url = "https://files.pythonhosted.org/packages/a0/0b/07f7a2d881f7e16c385b47fc1753382600847cad305dcc7fa0c25828acf8/yarl-1.25.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:9489e6abf47ba37f332075a91444c7cfedb03e6ce99fbb2f116bfe1ce810da3b", size = 117983, upload-time = "2026-09-15T19:30:55.277Z" }, + { url = "https://files.pythonhosted.org/packages/aa/9d/8cdceec66a9b940700cb45931741403f045b162afd79bcb93c41cadd0972/yarl-1.25.1-cp311-cp311-win_amd64.whl", hash = "sha256:d7306dee25b8a0e737363f347362b875094b4dc4e367311470656ae420fdbf8e", size = 102894, upload-time = "2026-09-15T19:30:57.606Z" }, + { url = "https://files.pythonhosted.org/packages/17/f1/7ec357db1d3ad2863542d71e8fe64a126bbec17b134cd7d30951196809a5/yarl-1.25.1-cp311-cp311-win_arm64.whl", hash = "sha256:abb1384477f5901d436b5d2e5465954de46ea6098f59163d243660b5c4461d35", size = 98609, upload-time = "2026-09-15T19:30:59.705Z" }, + { url = "https://files.pythonhosted.org/packages/75/b3/cd32ac66ae622b854c2df0ac52106dda220d361b65a64fde7d5b3684aa3f/yarl-1.25.1-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:94d7aa6debf92a1dd14cb5280b083a764169a13cfb23a452111160274ed989f4", size = 144798, upload-time = "2026-09-15T19:31:01.821Z" }, + { url = "https://files.pythonhosted.org/packages/61/fb/a2c52a8007c2051ba74662afb112ecf3d00346af4c25e33df9d80fd14fb8/yarl-1.25.1-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:83d4a37e4b95da4d8bda930d6d35b75b4cdadbacbb4980cae290ea3100b5d51d", size = 104583, upload-time = "2026-09-15T19:31:04.05Z" }, + { url = "https://files.pythonhosted.org/packages/be/dd/ee38aec8e09fdf957e50d4085453fbe202f56c6c3b4cf07b81cdb4f09ee9/yarl-1.25.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:e029648f9c951db30e98a7d7ec90835db88ec4b32820efe2a9bdc2287e032eb6", size = 104325, upload-time = "2026-09-15T19:31:06.338Z" }, + { url = "https://files.pythonhosted.org/packages/1e/b3/058dbfb1857b484c9cf9cc135659f50b85ce66e03c99e44dc2f7b6161f55/yarl-1.25.1-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:4d781294bb815ecb5ea57ff6bbf8038e0a31a95fdf3e1788f66e0dc100d64b58", size = 115358, upload-time = "2026-09-15T19:31:08.593Z" }, + { url = "https://files.pythonhosted.org/packages/db/39/29693446cf0cf6b15a0e2f75a5d40f93c56819b05b0622196f45e95b5cc0/yarl-1.25.1-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:e12c538e00e7c1b286a07061046b90e8124e6a9793efae2c70db6a4aad07faad", size = 107658, upload-time = "2026-09-15T19:31:10.802Z" }, + { url = "https://files.pythonhosted.org/packages/86/b3/3c4dd7e1af43b931fba95e0a722737f2ea94a6d199c802585282831d7abd/yarl-1.25.1-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:7e4de3ac4adbad3d0bc7c6f4360a7dbff5de2f15e3b723be3198074e17fd9c40", size = 122660, upload-time = "2026-09-15T19:31:12.84Z" }, + { url = "https://files.pythonhosted.org/packages/bd/b5/1b60dbc3cfc9c5712b15148c206748f2bc93953ffdbe25ea75b63dfc89c9/yarl-1.25.1-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:419f392a1da624877975709e3864dfe833af6cc7671b39318086d456e288380c", size = 126506, upload-time = "2026-09-15T19:31:15.088Z" }, + { url = "https://files.pythonhosted.org/packages/bc/7b/ca212cbe170ac8b96e45317ecbcf9c3c3ecf0cdec98d5b088a9c4088929b/yarl-1.25.1-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:c6f117789d22dce188e5754e8bc65b7e6ebf8cb73963b9fa761f672a5883769d", size = 117050, upload-time = "2026-09-15T19:31:17.241Z" }, + { url = "https://files.pythonhosted.org/packages/cb/c3/72b4938cdbe619ad71ac156182faef4908846b84dc3ca4dbb4c4e6f84014/yarl-1.25.1-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:80e47012e730da131c9f059c80936783f9659aae22dc31c03c0595590d11ed54", size = 114174, upload-time = "2026-09-15T19:31:19.294Z" }, + { url = "https://files.pythonhosted.org/packages/e8/43/268717870f9ba0cc9701a95181587f6dc8c5f387aab4aeecc83158f38a79/yarl-1.25.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:e80f557716fd765439577131e526b8942ffc2c07bdbc5e39fa62f660ba1e963f", size = 114944, upload-time = "2026-09-15T19:31:21.414Z" }, + { url = "https://files.pythonhosted.org/packages/da/84/baa5bf504d51fe062c4bcaf62936da97fffb43285978d0b39984824231fd/yarl-1.25.1-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:f61964f235a43738bfac50da46fc4254943a7eea3051aeb0b6fc7c992c29fadc", size = 108263, upload-time = "2026-09-15T19:31:23.388Z" }, + { url = "https://files.pythonhosted.org/packages/a4/28/779a2ed9e0152a601a27039bed9aead3f0b79797a67e2c44bfa444622dd8/yarl-1.25.1-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:e546fe1d4a93ebc2910f0d768baff19faa09843ab3f2036a67ed6e69fae4419d", size = 122184, upload-time = "2026-09-15T19:31:25.343Z" }, + { url = "https://files.pythonhosted.org/packages/f8/1f/118e9e5b8f07694d63fd3222e801d7782270003f1a222aa798df3f8d5933/yarl-1.25.1-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:cce0727fd5ac04d372fa9bbfde9febc2bcf209aadfcf0468e45dec72719895d1", size = 114001, upload-time = "2026-09-15T19:31:27.465Z" }, + { url = "https://files.pythonhosted.org/packages/0f/ae/a4cf1cf372313734b17996d4007f9f73596e7a178b9485802e5494ecf484/yarl-1.25.1-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:af4ea5b37403ef4e30f3927eaed540db942bde01d8d3ff083527c0704d1c9c68", size = 120565, upload-time = "2026-09-15T19:31:29.47Z" }, + { url = "https://files.pythonhosted.org/packages/05/79/ad94f93ca731bc9e44d321833ab96b82a4f9f5f63cf773f81a4aeea5ecc1/yarl-1.25.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:68782fdb4027b8d1eee25ec35e9a6db05e863b899eb0310b3a33b6c3fef55707", size = 117060, upload-time = "2026-09-15T19:31:31.367Z" }, + { url = "https://files.pythonhosted.org/packages/bb/cc/51a7b4abf4ac593b8e7eb3794b28e5a35ae26eed8bc04787628d215af82f/yarl-1.25.1-cp312-cp312-win_amd64.whl", hash = "sha256:7d575b54cb3863ef9bc290ea4b009999d55dc237326131e4853cf33e888fee03", size = 102593, upload-time = "2026-09-15T19:31:33.329Z" }, + { url = "https://files.pythonhosted.org/packages/9d/21/0941a6b93a58b59a1ec75e5333bf06929b671309c43c0cd201c172d9c39f/yarl-1.25.1-cp312-cp312-win_arm64.whl", hash = "sha256:bc3ac7bf569f6b64dad04dd7808c7872dae8a97df657856eac05e9b7e3614a85", size = 97697, upload-time = "2026-09-15T19:31:35.855Z" }, + { url = "https://files.pythonhosted.org/packages/7b/ed/2f3129bbcc9a5c8ba12cc2b29d8060a3bab9c8043c456cfd4b5ca3188890/yarl-1.25.1-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:25868beca8b6765f8f7d0e11fe6dd7c66dd4b0793b9500286d20cc92352126a5", size = 143623, upload-time = "2026-09-15T19:31:37.966Z" }, + { url = "https://files.pythonhosted.org/packages/17/e1/f1bc3390fdca352826676b531d0712736f156919090206700421d46b2c37/yarl-1.25.1-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:10b2fd95332f0d716d5eee3c9fb2ce8eada19082de7fee83d32e37992fd75c26", size = 104011, upload-time = "2026-09-15T19:31:40.25Z" }, + { url = "https://files.pythonhosted.org/packages/a8/aa/50acc5c3e5da04172ae3c281c75405af4d2ca911e16120ab0563f4dffb66/yarl-1.25.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:0f12afda4eea8c8994a76d4df1875c765194f5fbe8a9d197929ea303caee29ec", size = 103677, upload-time = "2026-09-15T19:31:42.46Z" }, + { url = "https://files.pythonhosted.org/packages/30/d2/7d1e0ab9f8390e1fbcede5a6dbf70d23c96ad09b8c5567f3a514d1ddb0e2/yarl-1.25.1-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:14b79a30a93a3ce2e8832603fd0ab780ada281b0ba5110b519a634f2d7d7d1fc", size = 115392, upload-time = "2026-09-15T19:31:44.371Z" }, + { url = "https://files.pythonhosted.org/packages/71/e1/5ba1e3a2a22139213655e760919038e8ed7e2d4a99826d0bbddb3beb96e5/yarl-1.25.1-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:4bd6340d20ae2c7ca719b87b426e808e90743b676d05d4c26c4fb5ca71f41184", size = 107493, upload-time = "2026-09-15T19:31:46.273Z" }, + { url = "https://files.pythonhosted.org/packages/f5/53/780653d5e0f73831f467cf13548912e5eec97f21dc49fc8daf21da027df4/yarl-1.25.1-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:126a2533570c554719ca40a1288fdee1700b6bc82e7131aa69fa85252d92e651", size = 122537, upload-time = "2026-09-15T19:31:48.654Z" }, + { url = "https://files.pythonhosted.org/packages/03/92/d54fa70236c6036271c9c9c09fd978df5cbe3ef49ef6c46e9b833476d215/yarl-1.25.1-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:a3faadac7d812ddac258feb57b9846b60c1b437c4f4b9ad42595c6f6fe4390df", size = 126170, upload-time = "2026-09-15T19:31:50.872Z" }, + { url = "https://files.pythonhosted.org/packages/0e/b7/a82a49bf88340b837ef6972b508a1604ae377b9e6904b46b10cf5f1cf925/yarl-1.25.1-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:be80550d9bfe83d9b62398a37081a90434e6df2d978ec345c3d2820de6beddab", size = 117012, upload-time = "2026-09-15T19:31:53.189Z" }, + { url = "https://files.pythonhosted.org/packages/ef/78/5d684b411e3f3602464ee9b538db48205038f8605872985f61efb809ced0/yarl-1.25.1-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:e07595c7d6f4db270ceede356a1bd1c07a34f1c26f958d1ed0cd7b48e0d2bba3", size = 114950, upload-time = "2026-09-15T19:31:55.694Z" }, + { url = "https://files.pythonhosted.org/packages/2f/11/51d82b852c64f7fad0fc7a7ff3031517204887e874c722bbca839c0b23ac/yarl-1.25.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:eb96ed1ae6c7d072d60840c0434aef07a2df611812810807fbc54263a6053e9a", size = 115428, upload-time = "2026-09-15T19:31:57.966Z" }, + { url = "https://files.pythonhosted.org/packages/e4/49/9d1978049bf646b9ea918313926453c6901b71c92f097467777d47d36a88/yarl-1.25.1-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:3feb99222553a8cbedfa52c2f59dd84c3f50d5b582c728d522caf8d72769a54b", size = 108428, upload-time = "2026-09-15T19:32:00.048Z" }, + { url = "https://files.pythonhosted.org/packages/43/35/7b8f1ebb45d7ec3dda7d1909bf44f458de41ef91e2937f107733582a5166/yarl-1.25.1-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:a2ed0ba415ccdf08f14bf544cb78346d0f76086707ffee24921a2c84dbf1305a", size = 121961, upload-time = "2026-09-15T19:32:02.436Z" }, + { url = "https://files.pythonhosted.org/packages/63/d6/d8b689ab7ca26edeb85f6ff28812aac7a25376eefc1780e303a7bfbaceff/yarl-1.25.1-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:2b49375d22299b0a834c2bca72f39aaecc270d96fb24c30424899676f487b22a", size = 114961, upload-time = "2026-09-15T19:32:04.456Z" }, + { url = "https://files.pythonhosted.org/packages/cf/d5/1a1798ea4dc6b7ee3260010a27907ebc697c95dae99817d817ed446d24aa/yarl-1.25.1-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:ef74070ac553c59eb4f04258722066d6c6135b7baa03b2e9f2da65c096e96d98", size = 120036, upload-time = "2026-09-15T19:32:06.5Z" }, + { url = "https://files.pythonhosted.org/packages/91/8d/b1b35ed7903da6669b1d367cb2c09436acd4ff508029b4f39a0c0c2058fc/yarl-1.25.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:0a66db89ea473abeac4b70523cafd94db3772380e565f9d28af7a179b7af71fa", size = 117276, upload-time = "2026-09-15T19:32:09.401Z" }, + { url = "https://files.pythonhosted.org/packages/3a/8f/4db01cef62caff0d7a4593ed694fb8a41a27a11158cab80d290221f13e57/yarl-1.25.1-cp313-cp313-win_amd64.whl", hash = "sha256:1f51020b2eb8a003c84925638ec63c21a750a4bddd3a22ec8eac6a742dadf1b9", size = 101945, upload-time = "2026-09-15T19:32:11.545Z" }, + { url = "https://files.pythonhosted.org/packages/c0/5e/3ce00497c5c0babb74d4130c10c3828ccd215b4819d12020c42429f991ac/yarl-1.25.1-cp313-cp313-win_arm64.whl", hash = "sha256:b10dd0557ba422715b5206b3743192135a6022acca8baec51aa127d0a75db8fe", size = 97270, upload-time = "2026-09-15T19:32:14.127Z" }, + { url = "https://files.pythonhosted.org/packages/54/22/318c7980066769c6bcd9221ed2248294f5698811da099013098c670565ed/yarl-1.25.1-py3-none-any.whl", hash = "sha256:681c758b0490f9e96b78e5fa8e8dc6e648e9185bb6eaebe73183c33ea0c445f3", size = 63617, upload-time = "2026-09-15T19:34:59.616Z" }, +] From e20f40878f4ba69224f38f73864f41ed333ab349 Mon Sep 17 00:00:00 2001 From: Nic Cope Date: Tue, 6 Oct 2026 20:19:52 -0700 Subject: [PATCH 7/7] Type-check functions, e2e tests and hack/ in one ty flake check ty ran in 21 checks: a ty- check per function, ty-e2e, and function-test-style, which type-checked hack/check_function_tests.py before running it. This commit replaces them with one ty check. It still runs ty once per function, against that function's venv, because every function's package is named function and one run could resolve one function's import to another's package. Then it checks e2e/ and hack/. A failed run doesn't stop the rest, and the build ends by naming every directory that failed. The runs are serial where the old checks built in parallel, which costs about a second once the venvs are built. Towards #473. Signed-off-by: Nic Cope --- nix/checks.nix | 114 ++++++++++++++++++++++++++----------------------- 1 file changed, 60 insertions(+), 54 deletions(-) diff --git a/nix/checks.nix b/nix/checks.nix index 3bc640f09..c75a0d758 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -32,39 +32,6 @@ let mkdir -p $out touch $out/.tests-passed ''; - - # Type-check each function with ty. Each function exports its own 'function' - # module, so checking all functions at once would let ty resolve one - # function's `function.fn` import to another's package. We check each in - # isolation against a venv that provides its dependencies, pytest, which the - # tests import, and the protobuf type stubs ty needs to resolve the SDK's - # generated Struct and Duration. - # - # Unlike mkFunctionTest, which runs the function module from the venv, ty - # checks the source, so we copy function/ and tests/ from the tree. We also - # copy pyproject.toml: the sandbox has no parent tree for ty to discover the - # [tool.ty] config in, where the target Python version is set. - mkFunctionTypeCheck = - name: - let - venv = pythonSet.mkVirtualEnv "${name}-ty-env" { - ${name} = [ ]; - pytest = [ ]; - types-protobuf = [ ]; - }; - in - pkgs.runCommand "modelplane-ty-${name}" - { - nativeBuildInputs = [ pkgs.unstable.ty ]; - } - '' - cp -r ${self}/functions/${name}/function function - cp -r ${self}/functions/${name}/tests tests - cp ${self}/pyproject.toml pyproject.toml - ty check function tests --python ${venv} - mkdir -p $out - touch $out/.ty-passed - ''; in { # Lint docs prose with Vale. The site that renders this content is the @@ -96,43 +63,88 @@ in # Check the function unit tests against the rules in CONTRIBUTING.md's Tests # section that an AST can decide, such as how a table and its test are laid # out and what a case's name and reason look like. The checker uses only the - # standard library, so it runs on the plain interpreter and ty needs no venv - # to type-check it. + # standard library, so it runs on the plain interpreter. function-test-style = pkgs.runCommand "modelplane-function-test-style" { - nativeBuildInputs = [ - pkgs.python3 - pkgs.unstable.ty - ]; + nativeBuildInputs = [ pkgs.python3 ]; } '' cd ${self} - ty check hack/check_function_tests.py python3 hack/check_function_tests.py mkdir -p $out touch $out/.function-test-style-checked ''; - # Type-check the end-to-end tests with ty, against the packages the e2e app - # runs them with (see apps.nix). - ty-e2e = + # Type-check our Python with ty. Every function's package is named + # `function`, so in one ty run one function's `from function import fn` could + # resolve to another function's package. ty instead runs once per function, + # against a venv that provides the function's dependencies, pytest, which the + # tests import, and the protobuf type stubs ty needs to resolve the SDK's + # generated Struct and Duration. It then runs on e2e/, against the packages + # the e2e app runs the tests with (see apps.nix), and on hack/, whose checker + # uses only the standard library and so needs no venv. + # + # Unlike mkFunctionTest, which runs the function module from the venv, ty + # checks the source, so we copy it from the tree: each function's function/ + # and tests/ to a directory of its own that ty takes as the project, and e2e/ + # and hack/ to the build directory. Each of those gets a copy of + # pyproject.toml for the [tool.ty] config, where the target Python version is + # set. ty also resolves first-party imports from the directory holding that + # config, so without a copy in a function's directory `function` would + # resolve to the package installed in the venv instead of the copied source. + # + # A failed run doesn't stop the rest, so one build reports every failure, and + # the log names each directory that failed. + ty = let - venv = pythonSet.mkVirtualEnv "e2e-ty-env" { + functionVenvs = map (name: { + inherit name; + venv = pythonSet.mkVirtualEnv "${name}-ty-env" { + ${name} = [ ]; + pytest = [ ]; + types-protobuf = [ ]; + }; + }) functionNames; + e2eVenv = pythonSet.mkVirtualEnv "e2e-ty-env" { pytest = [ ]; kubernetes = [ ]; crossplane-models = [ ]; pydantic = [ ]; }; in - pkgs.runCommand "modelplane-ty-e2e" + pkgs.runCommand "modelplane-ty" { nativeBuildInputs = [ pkgs.unstable.ty ]; } '' - cp -r ${self}/e2e e2e - cp ${self}/pyproject.toml pyproject.toml - ty check e2e --python ${venv} + cp -r ${self}/e2e ${self}/hack ${self}/pyproject.toml . + + failed=() + check() { + local name="$1" + shift + echo "ty check $name" + ty check --no-progress "$@" || failed+=("$name") + } + + check_function() { + local dir="functions/$1" venv="$2" + mkdir -p "$dir" + cp -r "${self}/$dir/function" "${self}/$dir/tests" pyproject.toml "$dir" + check "$dir" --project "$dir" --python "$venv" "$dir" + } + + ${pkgs.lib.concatMapStrings (f: '' + check_function ${f.name} ${f.venv} + '') functionVenvs} + check e2e --python ${e2eVenv} e2e + check hack hack + + if [ ''${#failed[@]} -gt 0 ]; then + echo "ty found errors in: ''${failed[*]}" >&2 + exit 1 + fi mkdir -p $out touch $out/.ty-passed ''; @@ -274,9 +286,3 @@ in value = mkFunctionTest name; }) functionNames ) -// builtins.listToAttrs ( - map (name: { - name = "ty-${name}"; - value = mkFunctionTypeCheck name; - }) functionNames -)