From 50c918242bd0596b32e9be133882cda5f875df5d Mon Sep 17 00:00:00 2001 From: Nic Cope Date: Wed, 30 Sep 2026 16:55:54 -0700 Subject: [PATCH 1/7] Run the function unit tests with pytest The e2e tests are moving to pytest, and one test runner for the whole repo is simpler than two. pytest 9 runs the existing unittest suites unchanged and reports each subTest on its own, so this commit switches the runner without touching a test. It configures pytest in strict mode, and to print whole assertion diffs, since the tests compare whole responses. Towards #473. Signed-off-by: Nic Cope --- nix/checks.nix | 10 +++++++--- pyproject.toml | 10 ++++++++++ uv.lock | 54 ++++++++++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 71 insertions(+), 3 deletions(-) diff --git a/nix/checks.nix b/nix/checks.nix index 29993c8af..000ff36d9 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -25,18 +25,22 @@ let ); # Each function exports a 'function' Python module, so tests must run from - # a directory where that module is importable via the venv. We copy tests/ - # from the source tree and run unittest against the venv's Python. + # a directory where that module is importable via the venv, and one pytest + # session can't hold two functions' tests. We copy tests/ from the source + # tree and run pytest against the venv's Python. We also copy pyproject.toml + # for its [tool.pytest] config, which pytest finds in its rootdir. mkFunctionTest = name: let venv = pythonSet.mkVirtualEnv "${name}-test-env" { ${name} = [ ]; + pytest = [ ]; }; in pkgs.runCommand "modelplane-test-${name}" { } '' cp -r ${self}/functions/${name}/tests tests - ${venv}/bin/python -m unittest discover -s tests -v + cp ${self}/pyproject.toml pyproject.toml + ${venv}/bin/python -m pytest tests mkdir -p $out touch $out/.tests-passed ''; diff --git a/pyproject.toml b/pyproject.toml index d808d3ae1..1cdafcfa1 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -26,6 +26,8 @@ dev = [ # schemas/ tree, so an edit there survives until the next build. "pyyaml>=6.0", "pydantic>=2.0", + # 9.0 reports unittest subtests natively. + "pytest>=9.0", ] [tool.ruff] @@ -94,3 +96,11 @@ python-version = "3.12" [tool.ty.src] # The generated Pydantic models aren't under our control and aren't checked. exclude = ["schemas/python"] + +[tool.pytest] +# Fail on unknown config, unregistered markers, duplicate parametrize IDs and +# passing xfails. +strict = true +# Tests compare whole responses, and by default pytest cuts the diff between two +# large values down to a summary line. +verbosity_assertions = "2" diff --git a/uv.lock b/uv.lock index 72b629398..ddc2973d6 100644 --- a/uv.lock +++ b/uv.lock @@ -498,6 +498,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/82/54/acc6a6e684827b0f6bb4e2c27f3d7e25b71322c4078ef5b455c07c43260e/grpcio_reflection-1.62.3-py3-none-any.whl", hash = "sha256:a48ef37df81a3bada78261fc92ef382f061112f989d1312398b945cc69838b9c", size = 22232, upload-time = "2024-08-06T00:30:13.131Z" }, ] +[[package]] +name = "iniconfig" +version = "2.3.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/72/34/14ca021ce8e5dfedc35312d08ba8bf51fdd999c576889fc2c24cb97f4f10/iniconfig-2.3.0.tar.gz", hash = "sha256:c76315c77db068650d49c5b56314774a7804df16fee4402c1f19d6d15d8c4730", size = 20503, upload-time = "2025-10-18T21:55:43.219Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/cb/b1/3846dd7f199d53cb17f49cba7e651e9ce294d8497c8c150530ed11865bb8/iniconfig-2.3.0-py3-none-any.whl", hash = "sha256:f631c04d2c48c52b84d0d0549c99ff3859c98df65b3101406327ecc7d53fbf12", size = 7484, upload-time = "2025-10-18T21:55:41.639Z" }, +] + [[package]] name = "jmespath" version = "1.1.0" @@ -525,6 +534,7 @@ source = { virtual = "." } dev = [ { name = "crossplane-function-sdk-python" }, { name = "pydantic" }, + { name = "pytest" }, { name = "pyyaml" }, { name = "types-protobuf" }, ] @@ -535,10 +545,20 @@ dev = [ dev = [ { name = "crossplane-function-sdk-python", specifier = ">=0.14.0" }, { name = "pydantic", specifier = ">=2.0" }, + { name = "pytest", specifier = ">=9.0" }, { name = "pyyaml", specifier = ">=6.0" }, { name = "types-protobuf", specifier = ">=4.24" }, ] +[[package]] +name = "packaging" +version = "26.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/7d/fa/3944b40b07da9ce895c0e6303a5ab7d53da063554f534556b134a54d6093/packaging-26.3.tar.gz", hash = "sha256:94edc256424af38762eb31306eed28beb9f0efc50a8837492c9d6fd6004aed79", size = 313412, upload-time = "2026-08-04T18:15:28.737Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/63/34/ba1c580383c9eada3711951fef0795c80b829a078d72188184bcab9dd527/packaging-26.3-py3-none-any.whl", hash = "sha256:d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c", size = 129956, upload-time = "2026-08-04T18:15:27.159Z" }, +] + [[package]] name = "pendulum" version = "3.2.0" @@ -589,6 +609,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/02/fb/d65db067a67df7252f18b0cb7420dda84078b9e8bfb375215469c14a50be/pendulum-3.2.0-py3-none-any.whl", hash = "sha256:f3a9c18a89b4d9ef39c5fa6a78722aaff8d5be2597c129a3b16b9f40a561acf3", size = 114111, upload-time = "2026-01-30T11:22:22.361Z" }, ] +[[package]] +name = "pluggy" +version = "1.6.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/f9/e2/3e91f31a7d2b083fe6ef3fa267035b518369d9511ffab804f839851d2779/pluggy-1.6.0.tar.gz", hash = "sha256:7dcc130b76258d33b90f61b658791dede3486c3e6bfb003ee5c9bfb396dd22f3", size = 69412, upload-time = "2025-05-15T12:30:07.975Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/54/20/4d324d65cc6d9205fabedc306948156824eb9f0ee1633355a8f7ec5c66bf/pluggy-1.6.0-py3-none-any.whl", hash = "sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746", size = 20538, upload-time = "2025-05-15T12:30:06.134Z" }, +] + [[package]] name = "protobuf" version = "7.35.0" @@ -691,6 +720,31 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/4b/2d/69abac8f838090bbecd5df894befb2c2619e7996a98ddb949db9f3b93225/pydantic_core-2.46.4-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:d51026d73fcfd93610abc7b27789c26b313920fcfb20e27462d74a7f8b06e983", size = 2193071, upload-time = "2026-05-06T13:38:08.682Z" }, ] +[[package]] +name = "pygments" +version = "2.21.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/49/2e/ced460408999b33da6b31b0021b0f37d329e202d4169aeb164493778f25b/pygments-2.21.0.tar.gz", hash = "sha256:610ca751c9bc2492b38eb9a38a7fbc93edbbb2d7182edaf34e66ae493dee5c8c", size = 5005329, upload-time = "2026-08-17T08:02:48.824Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/71/46/17f022dd3e953bf20a04a028a21ec746d942f8d2af30fa0f124fa0e6a684/pygments-2.21.0-py3-none-any.whl", hash = "sha256:2363c69b61c4a97c838da3b130dcd6468f4848992b21a82f2a63ec34377137d9", size = 1250147, upload-time = "2026-08-17T08:02:44.912Z" }, +] + +[[package]] +name = "pytest" +version = "9.1.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "colorama", marker = "sys_platform == 'win32'" }, + { name = "iniconfig" }, + { name = "packaging" }, + { name = "pluggy" }, + { name = "pygments" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/e4/47/b9efed96c114afcfa3c9d3fe98a76a1d14c74a9e266d397cf6eb64be5e01/pytest-9.1.1.tar.gz", hash = "sha256:1088fbde8f2b49d95a549a195707afa7a76a3ce9bcadc26b6d71f0ffda5fe313", size = 1636369, upload-time = "2026-06-19T10:58:32.857Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/24/25/1de2678b631f5a49215c6c96fff41ba892b0a34df68d6d80292b1b48aa7f/pytest-9.1.1-py3-none-any.whl", hash = "sha256:37a86b45efb9a47a61a36449063e8e18d0cab3161329fc099eb21783169c4f0c", size = 386536, upload-time = "2026-06-19T10:58:31.347Z" }, +] + [[package]] name = "python-dateutil" version = "2.9.0.post0" From bb4de98afd104eed006be162663fe447ae3bb1c6 Mon Sep 17 00:00:00 2001 From: Nic Cope Date: Wed, 30 Sep 2026 18:43:24 -0700 Subject: [PATCH 2/7] Write the function unit tests in pytest's style The previous commit runs the unittest suites under pytest unchanged. This commit ports them to pytest's own style, so each case in a table is a test of its own that -k can select. Async tests call RunFunction with asyncio.run rather than needing a plugin. pytest captures a test's output and shows it only when the test fails, so the tests no longer disable the functions' logging. pytest diffs two dicts in insertion order, where unittest sorted their keys. The Structs a function returns rarely list their keys in the same order as the test's expected ones, so the tests compare dicts with sorted keys, which keeps a diff to the fields that differ. No case's request or expected response changes. Towards #473. Signed-off-by: Nic Cope --- CONTRIBUTING.md | 102 +- .../compose-aks-cluster/tests/test_fn.py | 540 +- .../compose-eks-cluster/tests/__init__.py | 14 - .../compose-eks-cluster/tests/test_fn.py | 898 ++- .../compose-gke-cluster/tests/__init__.py | 15 - .../compose-gke-cluster/tests/test_fn.py | 637 ++- .../compose-inference-class/tests/__init__.py | 14 - .../compose-inference-class/tests/test_fn.py | 118 +- .../tests/__init__.py | 15 - .../tests/test_fn.py | 4942 ++++++++--------- .../tests/__init__.py | 15 - .../tests/test_fn.py | 1882 +++---- .../compose-model-cache/tests/test_fn.py | 1110 ++-- .../tests/__init__.py | 14 - .../tests/test_cel.py | 461 +- .../compose-model-deployment/tests/test_fn.py | 2391 ++++---- .../tests/test_quantity.py | 329 +- .../tests/test_scheduling.py | 2602 +++++---- .../tests/test_semver.py | 155 +- .../compose-model-endpoint/tests/__init__.py | 14 - .../compose-model-endpoint/tests/test_fn.py | 260 +- .../compose-model-replica/tests/__init__.py | 14 - .../tests/test_backends.py | 1953 +++---- .../compose-model-replica/tests/test_fn.py | 968 ++-- .../compose-model-route/tests/__init__.py | 14 - .../compose-model-route/tests/test_fn.py | 895 ++- .../compose-model-service/tests/__init__.py | 14 - .../compose-model-service/tests/test_fn.py | 501 +- .../compose-nebius-cluster/tests/test_fn.py | 721 ++- .../compose-serving-stack/tests/__init__.py | 15 - .../compose-serving-stack/tests/test_fn.py | 647 ++- .../tests/test_stacks.py | 217 +- functions/compose-usages/tests/__init__.py | 14 - functions/compose-usages/tests/test_fn.py | 241 +- .../compose-vultr-cluster/tests/test_fn.py | 516 +- nix/checks.nix | 6 +- pyproject.toml | 3 + 37 files changed, 11426 insertions(+), 11841 deletions(-) delete mode 100644 functions/compose-eks-cluster/tests/__init__.py delete mode 100644 functions/compose-gke-cluster/tests/__init__.py delete mode 100644 functions/compose-inference-class/tests/__init__.py delete mode 100644 functions/compose-inference-cluster/tests/__init__.py delete mode 100644 functions/compose-inference-gateway/tests/__init__.py delete mode 100644 functions/compose-model-deployment/tests/__init__.py delete mode 100644 functions/compose-model-endpoint/tests/__init__.py delete mode 100644 functions/compose-model-replica/tests/__init__.py delete mode 100644 functions/compose-model-route/tests/__init__.py delete mode 100644 functions/compose-model-service/tests/__init__.py delete mode 100644 functions/compose-serving-stack/tests/__init__.py delete mode 100644 functions/compose-usages/tests/__init__.py diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 0c27ee61e..3cc00dc18 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -245,7 +245,7 @@ functions// main.py # CLI entrypoint (boilerplate) fn.py # FunctionRunner gRPC service and Composer logic tests/ - test_fn.py # unittest-based tests for fn.py + test_fn.py # pytest tests for fn.py ``` The `Composer.compose()` method in `fn.py` reads the XR from the request, @@ -274,12 +274,15 @@ XRDs or dependencies you've removed don't linger. ### Tests -Every function has tests under `functions//tests/test_fn.py`. The -canonical form is a table of `Case`s, each running the function on a +Every function has tests under `functions//tests/`, run with +[pytest](https://docs.pytest.org/). `test_fn.py` tests the function as a whole. +A module with logic of its own, such as `compose-model-deployment`'s scheduler, +can have its own `test_.py` too. + +The canonical form is a table of `Case`s, each running the function on a `RunFunctionRequest` and comparing the whole `RunFunctionResponse` against an -expected one — not asserting on individual fields. `compose-usages` is a clean -example; `compose-model-cache` shows the same form scaled up to a multi-pass -reconcile. The skeleton: +expected one, rather than asserting on individual fields. `compose-usages` is a +small example. The skeleton: ```python @dataclasses.dataclass @@ -289,50 +292,59 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - cases = [ - Case( - name="describes what this case exercises", - req=fnv1.RunFunctionRequest(...), - want=fnv1.RunFunctionResponse(...), - ), - ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) +COMPOSE_CASES = [ + Case( + name="describes what this case exercises", + req=fnv1.RunFunctionRequest(...), + want=fnv1.RunFunctionResponse(...), + ), +] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes the resources an XR needs.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) ``` -Build the XR with -`resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json"))` from a -generated Pydantic model; build other observed, desired, and required resources -as plain dicts. Because `want` is the whole response, it must include the parts -the function always emits: `meta.ttl` (60s), an empty `context`, and any -conditions, results, and requirements. Give observed conditions a fixed -`lastTransitionTime` so the input is deterministic. Protobuf maps -(`desired.resources`, `requirements.resources`) compare order-independently, but -repeated fields (`conditions`, `results`, status arrays) must match the order -the function emits. +Name a table for the test that runs it, and put it just above that test. Each +case becomes its own test, named for the case, so `pytest -k` can select it. +With `got` on the left, pytest's diff shows the expected lines as `-` and the +actual lines as `+`, the same way round as Go's `cmp.Diff(want, got)`. Tests are +plain functions, with no classes, fixtures, or `conftest.py`. They call the +async `RunFunction` with `asyncio.run` rather than needing a plugin, and check +errors with `pytest.raises(..., match=...)`. + +Build the XR with `resource.dict_to_struct(xr.model_dump(exclude_none=True, +mode="json"))` from a generated Pydantic model; build other observed, desired, +and required resources as plain dicts. Because `want` is the whole response, it +must include the parts the function always emits: `meta.ttl` (60s), an empty +`context`, and any conditions, results, and requirements. Give observed +conditions a fixed `lastTransitionTime` so the input is deterministic. Protobuf +maps (`desired.resources`, `requirements.resources`) compare +order-independently, but repeated fields (`conditions`, `results`, status +arrays) must match the order the function emits. Some existing tests (`compose-serving-stack`, the second method in `compose-eks-cluster`) predate this form and assert on individual fields. Don't -model new tests on them. Add new cases to the function's `test_fn.py` and run -`nix flake check` to verify they pass. +model new tests on them. + +`nix flake check` runs every function's tests. To run one function's while +you work on it: + +```bash +uv run --isolated --package compose-usages --group dev pytest functions/compose-usages/tests +``` + +`--isolated` gives each run its own environment, because every function names +its package `function`, so two can't share one. That's also why each function +runs in its own pytest session. ### Running locally diff --git a/functions/compose-aks-cluster/tests/test_fn.py b/functions/compose-aks-cluster/tests/test_fn.py index d3ea4c9fa..e5a28a466 100644 --- a/functions/compose-aks-cluster/tests/test_fn.py +++ b/functions/compose-aks-cluster/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-aks-cluster function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.akscluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -36,10 +38,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - # Names derived like the function derives them - the hash suffix depends only # on the input names. _CLUSTER_NAME = resource.child_name("modelplane-system", "test-cluster", "aks") @@ -344,285 +342,277 @@ def _observed_ready(desired: dict) -> fnv1.Resource: ) -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function composes AKS cluster infrastructure.""" - cases = [ - Case( - name="first pass composes infra; gated resources wait for the cluster", - req=_req([_GPU_POOL]), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - # The StorageClass isn't composed yet: the cluster - # isn't observed, so the ProviderConfigs can't - # reach it. - "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, +COMPOSE_CASES = [ + Case( + name="first pass composes infra; gated resources wait for the cluster", + req=_req([_GPU_POOL]), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), + "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), + # The StorageClass isn't composed yet: the cluster + # isn't observed, so the ProviderConfigs can't + # reach it. + "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, ), - context=structpb.Struct(), - ), - ), - Case( - name="zones pass through to the node pool", - req=_req( - [ - v1alpha1.NodePool( - name="gpuh100", - role="GPU", - vmSize="Standard_ND96isr_H100_v5", - diskSizeGb=200, - nodeCount=1, - minNodeCount=1, - maxNodeCount=4, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), - zones=[v1alpha1.Zone("1")], + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), ), - ] - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "nodepool-gpuh100": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu(zones=["1"])), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, + ready=fnv1.READY_TRUE, ), - context=structpb.Struct(), - ), + }, ), - Case( - name="InfiniBand pool composes the network operator once the cluster is observed", - req=_req( - [_GPU_POOL_INFINIBAND], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, + context=structpb.Struct(), + ), + ), + Case( + name="zones pass through to the node pool", + req=_req( + [ + v1alpha1.NodePool( + name="gpuh100", + role="GPU", + vmSize="Standard_ND96isr_H100_v5", + diskSizeGb=200, + nodeCount=1, + minNodeCount=1, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + zones=[v1alpha1.Zone("1")], ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), - "release-network-operator": fnv1.Resource( - resource=resource.dict_to_struct(_network_operator_release()), - ), - "storage-class-rwx-fs": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class()), - ready=fnv1.READY_TRUE, - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, + ] + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), + "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), + "nodepool-gpuh100": fnv1.Resource( + resource=resource.dict_to_struct(_nodepool_gpu(zones=["1"])), ), - context=structpb.Struct(), - ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + }, ), - Case( - name="InfiniBand pool before the cluster is observed gates the network operator", - req=_req([_GPU_POOL_INFINIBAND]), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, + context=structpb.Struct(), + ), + ), + Case( + name="InfiniBand pool composes the network operator once the cluster is observed", + req=_req( + [_GPU_POOL_INFINIBAND], + observed_resources={ + "cluster": _observed_ready(_cluster()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), + "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, ), - context=structpb.Struct(), - ), + "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), + "release-network-operator": fnv1.Resource( + resource=resource.dict_to_struct(_network_operator_release()), + ), + "storage-class-rwx-fs": fnv1.Resource( + resource=resource.dict_to_struct(_storage_class()), + ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + }, ), - Case( - name="custom credentials flow through to all cloud MRs", - req=_req( - [_GPU_POOL], - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="my-azure-account", + context=structpb.Struct(), + ), + ), + Case( + name="InfiniBand pool before the cluster is observed gates the network operator", + req=_req([_GPU_POOL_INFINIBAND]), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), + "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), + "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "resource-group": fnv1.Resource( - resource=resource.dict_to_struct(_resource_group("ProviderConfig", "my-azure-account")), - ), - "virtual-network": fnv1.Resource( - resource=resource.dict_to_struct( - _virtual_network("ProviderConfig", "my-azure-account") - ), - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet("ProviderConfig", "my-azure-account")), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster("ProviderConfig", "my-azure-account")), - ), - "nodepool-gpuh100": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu("ProviderConfig", "my-azure-account")), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, ), - context=structpb.Struct(), - ), + }, ), - Case( - name="marks managed resources ready from observed conditions", - req=_req( - [_GPU_POOL], - observed_resources={ - "resource-group": _observed_ready(_resource_group()), - "virtual-network": _observed_ready(_virtual_network()), - "subnet": _observed_ready(_subnet()), - "cluster": _observed_ready(_cluster()), - "nodepool-gpuh100": _observed_ready(_nodepool_gpu()), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "resource-group": fnv1.Resource( - resource=resource.dict_to_struct(_resource_group()), - ready=fnv1.READY_TRUE, - ), - "virtual-network": fnv1.Resource( - resource=resource.dict_to_struct(_virtual_network()), - ready=fnv1.READY_TRUE, - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ready=fnv1.READY_TRUE, - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - # The cluster is observed, so the StorageClass is - # composed too. - "storage-class-rwx-fs": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class()), - ready=fnv1.READY_TRUE, - ), - "nodepool-gpuh100": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu()), - ready=fnv1.READY_TRUE, - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, + context=structpb.Struct(), + ), + ), + Case( + name="custom credentials flow through to all cloud MRs", + req=_req( + [_GPU_POOL], + credentials=v1alpha1.Credentials( + type="ProviderConfig", + name="my-azure-account", + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "resource-group": fnv1.Resource( + resource=resource.dict_to_struct(_resource_group("ProviderConfig", "my-azure-account")), ), - context=structpb.Struct(), - ), + "virtual-network": fnv1.Resource( + resource=resource.dict_to_struct(_virtual_network("ProviderConfig", "my-azure-account")), + ), + "subnet": fnv1.Resource( + resource=resource.dict_to_struct(_subnet("ProviderConfig", "my-azure-account")), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster("ProviderConfig", "my-azure-account")), + ), + "nodepool-gpuh100": fnv1.Resource( + resource=resource.dict_to_struct(_nodepool_gpu("ProviderConfig", "my-azure-account")), + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + ), + ), + Case( + name="marks managed resources ready from observed conditions", + req=_req( + [_GPU_POOL], + observed_resources={ + "resource-group": _observed_ready(_resource_group()), + "virtual-network": _observed_ready(_virtual_network()), + "subnet": _observed_ready(_subnet()), + "cluster": _observed_ready(_cluster()), + "nodepool-gpuh100": _observed_ready(_nodepool_gpu()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "resource-group": fnv1.Resource( + resource=resource.dict_to_struct(_resource_group()), + ready=fnv1.READY_TRUE, + ), + "virtual-network": fnv1.Resource( + resource=resource.dict_to_struct(_virtual_network()), + ready=fnv1.READY_TRUE, + ), + "subnet": fnv1.Resource( + resource=resource.dict_to_struct(_subnet()), + ready=fnv1.READY_TRUE, + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, + ), + # The cluster is observed, so the StorageClass is + # composed too. + "storage-class-rwx-fs": fnv1.Resource( + resource=resource.dict_to_struct(_storage_class()), + ready=fnv1.READY_TRUE, + ), + "nodepool-gpuh100": fnv1.Resource( + resource=resource.dict_to_struct(_nodepool_gpu()), + ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, + ), + }, ), - ] - - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) + context=structpb.Struct(), + ), + ), +] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """The function composes AKS cluster infrastructure.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-eks-cluster/tests/__init__.py b/functions/compose-eks-cluster/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-eks-cluster/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-eks-cluster/tests/test_fn.py b/functions/compose-eks-cluster/tests/test_fn.py index d0ba254ba..9658abec0 100644 --- a/functions/compose-eks-cluster/tests/test_fn.py +++ b/functions/compose-eks-cluster/tests/test_fn.py @@ -14,15 +14,17 @@ """Tests for the compose-eks-cluster function.""" +import asyncio import dataclasses -import unittest +import json from typing import Any -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.ekscluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -37,10 +39,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - _KUBECONFIG_SECRET = "test-cluster-kubeconfig-55b57" _SUBNET_A = "test-cluster-subnet-us-west-2a-952dc" _SUBNET_B = "test-cluster-subnet-us-west-2b-2b80f" @@ -1139,403 +1137,386 @@ def _expected_resources() -> dict: } -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function composes EKS cluster infrastructure.""" - # Second pass: cluster and cluster-auth observed Ready, function flips - # those two desired resources ready while still emitting everything. - ready_resources = _expected_resources() - ready_resources["cluster"] = fnv1.Resource( - resource=ready_resources["cluster"].resource, - ready=fnv1.READY_TRUE, - ) - ready_resources["cluster-auth"] = fnv1.Resource( - resource=ready_resources["cluster-auth"].resource, - ready=fnv1.READY_TRUE, - ) - ready_resources["efs-filesystem"] = fnv1.Resource( - resource=ready_resources["efs-filesystem"].resource, - ready=fnv1.READY_TRUE, - ) - # Once the EFS filesystem id is observed, the managed StorageClass Object - # is composed (and marked ready) against the cluster's own ProviderConfig. - ready_resources["storage-class-rwx-efs"] = fnv1.Resource( - resource=resource.dict_to_struct(_storage_class_object("fs-0abc123")), - ready=fnv1.READY_TRUE, - ) - # With the cluster observed, the autoscaler Helm release is composed (it's - # gated on the cluster existing so provider-helm can reach it). It carries - # no Ready condition yet, so it stays not-ready this pass. - ready_resources["release-cluster-autoscaler"] = fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler_release()), - ) +def _compose_cases() -> list[Case]: + """The cases test_compose runs.""" + # Second pass: cluster and cluster-auth observed Ready, function flips + # those two desired resources ready while still emitting everything. + ready_resources = _expected_resources() + ready_resources["cluster"] = fnv1.Resource( + resource=ready_resources["cluster"].resource, + ready=fnv1.READY_TRUE, + ) + ready_resources["cluster-auth"] = fnv1.Resource( + resource=ready_resources["cluster-auth"].resource, + ready=fnv1.READY_TRUE, + ) + ready_resources["efs-filesystem"] = fnv1.Resource( + resource=ready_resources["efs-filesystem"].resource, + ready=fnv1.READY_TRUE, + ) + # Once the EFS filesystem id is observed, the managed StorageClass Object + # is composed (and marked ready) against the cluster's own ProviderConfig. + ready_resources["storage-class-rwx-efs"] = fnv1.Resource( + resource=resource.dict_to_struct(_storage_class_object("fs-0abc123")), + ready=fnv1.READY_TRUE, + ) + # With the cluster observed, the autoscaler Helm release is composed (it's + # gated on the cluster existing so provider-helm can reach it). It carries + # no Ready condition yet, so it stays not-ready this pass. + ready_resources["release-cluster-autoscaler"] = fnv1.Resource( + resource=resource.dict_to_struct(_autoscaler_release()), + ) - cases = [ - Case( - name="first pass composes infra resources; none ready", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr().model_dump(exclude_none=True, mode="json"), - ), + return [ + Case( + name="first pass composes infra resources; none ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr().model_dump(exclude_none=True, mode="json"), ), ), ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources=_expected_resources(), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct(_expected_status()), ), - context=structpb.Struct(), + resources=_expected_resources(), ), + context=structpb.Struct(), ), - Case( - name="second pass with observed cluster ready marks cluster resources ready", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( + ), + Case( + name="second pass with observed cluster ready marks cluster resources ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr().model_dump(exclude_none=True, mode="json"), + ), + ), + resources={ + "cluster": fnv1.Resource( resource=resource.dict_to_struct( - _xr().model_dump(exclude_none=True, mode="json"), + { + **_eks_cluster(), + "status": {"conditions": [_ready_condition()]}, + }, ), ), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_eks_cluster(), - "status": {"conditions": [_ready_condition()]}, - }, - ), - ), - "cluster-auth": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_cluster_auth(), - "status": {"conditions": [_ready_condition()]}, - }, - ), + "cluster-auth": fnv1.Resource( + resource=resource.dict_to_struct( + { + **_cluster_auth(), + "status": {"conditions": [_ready_condition()]}, + }, ), - "efs-filesystem": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_efs_filesystem(), - "metadata": {"annotations": {"crossplane.io/external-name": "fs-0abc123"}}, - "status": {"conditions": [_ready_condition()]}, - }, - ), + ), + "efs-filesystem": fnv1.Resource( + resource=resource.dict_to_struct( + { + **_efs_filesystem(), + "metadata": {"annotations": {"crossplane.io/external-name": "fs-0abc123"}}, + "status": {"conditions": [_ready_condition()]}, + }, ), - }, - ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), ), - resources=ready_resources, - ), - context=structpb.Struct(), + }, ), ), - ] - - for case in cases: - with self.subTest(name=case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - ) - - async def test_compose_capacity_block(self) -> None: - """A Capacity Block pool composes a launch template and a CAPACITY_BLOCK node group. - - The GPU node group must not set instanceTypes (EKS takes the type - from the launch template), must set capacityType=CAPACITY_BLOCK, and - must reference the launch template. The launch template targets the - reservation via the capacity-block market type. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr_capacity_block().model_dump(exclude_none=True, mode="json"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct(_expected_status()), ), + resources=ready_resources, ), + context=structpb.Struct(), ), - ) + ), + ] - got = await self.runner.RunFunction(req, None) - resources = got.desired.resources - # The launch template is composed and targets the reservation. - self.assertIn("launch-template-gpu-h200", resources) - self.assertEqual( - _launch_template(), - resource.struct_to_dict(resources["launch-template-gpu-h200"].resource), - ) +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - # The GPU node group uses CAPACITY_BLOCK + the launch template and - # carries no instanceTypes. - self.assertEqual( - _gpu_node_group_capacity_block(), - resource.struct_to_dict(resources["nodegroup-gpu-h200"].resource), - ) - async def test_compose_efa(self) -> None: - """An EFA GPU pool composes EFA infrastructure end to end. - - The node group's launch template carries one EFA interface per network - card (card 0 keeps device index 0 for the node's IP traffic, the rest - device index 1 for RDMA), the cluster gets an EFA security group with - self-referencing all-traffic ingress and egress rules, and the node - group references the launch template instead of setting instanceTypes. - """ - want_resources = { - "vpc": fnv1.Resource(resource=resource.dict_to_struct(_vpc())), - "subnet-0": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_A, "us-west-2a", "10.0.0.0/20")), - ), - "subnet-1": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_B, "us-west-2b", "10.0.16.0/20")), - ), - "subnet-2": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_C, "us-west-2c", "10.0.32.0/20")), - ), - "private-subnet-0": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_A, "us-west-2a", "10.0.48.0/20")), - ), - "private-subnet-1": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_B, "us-west-2b", "10.0.64.0/20")), - ), - "private-subnet-2": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_C, "us-west-2c", "10.0.80.0/20")), - ), - "internet-gateway": fnv1.Resource(resource=resource.dict_to_struct(_internet_gateway())), - "nat-eip": fnv1.Resource(resource=resource.dict_to_struct(_nat_eip())), - "nat-gateway": fnv1.Resource(resource=resource.dict_to_struct(_nat_gateway("us-west-2a"))), - "route-table": fnv1.Resource(resource=resource.dict_to_struct(_route_table())), - "route-default": fnv1.Resource(resource=resource.dict_to_struct(_route_default())), - "private-route-table": fnv1.Resource(resource=resource.dict_to_struct(_private_route_table())), - "private-route-default": fnv1.Resource(resource=resource.dict_to_struct(_private_route_default())), - "route-table-association-0": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2a")), - ), - "route-table-association-1": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2b")), - ), - "route-table-association-2": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2c")), - ), - "private-route-table-association-0": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2a")), - ), - "private-route-table-association-1": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2b")), - ), - "private-route-table-association-2": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2c")), - ), - "iam-role-cluster": fnv1.Resource( - resource=resource.dict_to_struct(_role("cluster", _ASSUME_CLUSTER)), - ), - "iam-attach-cluster-policy": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("cluster", "arn:aws:iam::aws:policy/AmazonEKSClusterPolicy"), - ), - ), - "iam-role-node": fnv1.Resource(resource=resource.dict_to_struct(_role("node", _ASSUME_NODE))), - "iam-attach-node-worker": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy"), - ), - ), - "iam-attach-node-cni": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy"), - ), - ), - "iam-attach-node-ecr": fnv1.Resource( +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """The function composes EKS cluster infrastructure.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) + + +def test_compose_capacity_block() -> None: + """A Capacity Block pool composes a launch template and a CAPACITY_BLOCK node group.""" + # The GPU node group must not set instanceTypes (EKS takes the type + # from the launch template), must set capacityType=CAPACITY_BLOCK, and + # must reference the launch template. The launch template targets the + # reservation via the capacity-block market type. + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly"), + _xr_capacity_block().model_dump(exclude_none=True, mode="json"), ), ), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_eks_cluster())), - "cluster-auth": fnv1.Resource(resource=resource.dict_to_struct(_cluster_auth())), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_system_node_group())), - "launch-template-gpu-h200": fnv1.Resource(resource=resource.dict_to_struct(_launch_template_efa())), - "efa-security-group": fnv1.Resource(resource=resource.dict_to_struct(_efa_security_group())), - "efa-security-group-ingress": fnv1.Resource( - resource=resource.dict_to_struct(_efa_security_group_ingress()), - ), - "efa-security-group-egress": fnv1.Resource( - resource=resource.dict_to_struct(_efa_security_group_egress()), - ), - "nodegroup-gpu-h200": fnv1.Resource(resource=resource.dict_to_struct(_gpu_node_group_efa())), - "addon-vpc-cni": fnv1.Resource(resource=resource.dict_to_struct(_addon("vpc-cni"))), - "addon-kube-proxy": fnv1.Resource(resource=resource.dict_to_struct(_addon("kube-proxy"))), - "addon-coredns": fnv1.Resource(resource=resource.dict_to_struct(_addon("coredns"))), - "efs-filesystem": fnv1.Resource(resource=resource.dict_to_struct(_efs_filesystem())), - "efs-security-group": fnv1.Resource(resource=resource.dict_to_struct(_efs_security_group())), - "efs-security-group-ingress": fnv1.Resource( - resource=resource.dict_to_struct(_efs_security_group_ingress()), - ), - "efs-mount-target-0": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_A))), - "efs-mount-target-1": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_B))), - "efs-mount-target-2": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_C))), - "iam-role-efs-csi": fnv1.Resource( - resource=resource.dict_to_struct(_role("efs-csi", _ASSUME_POD_IDENTITY)), - ), - "iam-attach-efs-csi": fnv1.Resource( - resource=resource.dict_to_struct(_role_policy_attachment("efs-csi", _POLICY_EFS_CSI)), - ), - "addon-eks-pod-identity-agent": fnv1.Resource( - resource=resource.dict_to_struct(_addon("eks-pod-identity-agent")), - ), - "pod-identity-efs-csi": fnv1.Resource(resource=resource.dict_to_struct(_pod_identity_association())), - "addon-aws-efs-csi-driver": fnv1.Resource( - resource=resource.dict_to_struct(_addon("aws-efs-csi-driver")), - ), - "iam-policy-cluster-autoscaler": fnv1.Resource(resource=resource.dict_to_struct(_autoscaler_policy())), - "iam-role-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_role("cluster-autoscaler", _ASSUME_POD_IDENTITY)), - ), - "iam-attach-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler_attachment()), + ), + ) + + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + resources = got.desired.resources + + # The launch template is composed and targets the reservation. + assert "launch-template-gpu-h200" in resources + assert resource.struct_to_dict(resources["launch-template-gpu-h200"].resource) == _launch_template() + + # The GPU node group uses CAPACITY_BLOCK + the launch template and + # carries no instanceTypes. + assert resource.struct_to_dict(resources["nodegroup-gpu-h200"].resource) == _gpu_node_group_capacity_block() + + +def test_compose_efa() -> None: + """An EFA GPU pool composes EFA infrastructure end to end.""" + # The node group's launch template carries one EFA interface per network + # card (card 0 keeps device index 0 for the node's IP traffic, the rest + # device index 1 for RDMA), the cluster gets an EFA security group with + # self-referencing all-traffic ingress and egress rules, and the node + # group references the launch template instead of setting instanceTypes. + want_resources = { + "vpc": fnv1.Resource(resource=resource.dict_to_struct(_vpc())), + "subnet-0": fnv1.Resource( + resource=resource.dict_to_struct(_subnet(_SUBNET_A, "us-west-2a", "10.0.0.0/20")), + ), + "subnet-1": fnv1.Resource( + resource=resource.dict_to_struct(_subnet(_SUBNET_B, "us-west-2b", "10.0.16.0/20")), + ), + "subnet-2": fnv1.Resource( + resource=resource.dict_to_struct(_subnet(_SUBNET_C, "us-west-2c", "10.0.32.0/20")), + ), + "private-subnet-0": fnv1.Resource( + resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_A, "us-west-2a", "10.0.48.0/20")), + ), + "private-subnet-1": fnv1.Resource( + resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_B, "us-west-2b", "10.0.64.0/20")), + ), + "private-subnet-2": fnv1.Resource( + resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_C, "us-west-2c", "10.0.80.0/20")), + ), + "internet-gateway": fnv1.Resource(resource=resource.dict_to_struct(_internet_gateway())), + "nat-eip": fnv1.Resource(resource=resource.dict_to_struct(_nat_eip())), + "nat-gateway": fnv1.Resource(resource=resource.dict_to_struct(_nat_gateway("us-west-2a"))), + "route-table": fnv1.Resource(resource=resource.dict_to_struct(_route_table())), + "route-default": fnv1.Resource(resource=resource.dict_to_struct(_route_default())), + "private-route-table": fnv1.Resource(resource=resource.dict_to_struct(_private_route_table())), + "private-route-default": fnv1.Resource(resource=resource.dict_to_struct(_private_route_default())), + "route-table-association-0": fnv1.Resource( + resource=resource.dict_to_struct(_route_table_association("us-west-2a")), + ), + "route-table-association-1": fnv1.Resource( + resource=resource.dict_to_struct(_route_table_association("us-west-2b")), + ), + "route-table-association-2": fnv1.Resource( + resource=resource.dict_to_struct(_route_table_association("us-west-2c")), + ), + "private-route-table-association-0": fnv1.Resource( + resource=resource.dict_to_struct(_private_route_table_association("us-west-2a")), + ), + "private-route-table-association-1": fnv1.Resource( + resource=resource.dict_to_struct(_private_route_table_association("us-west-2b")), + ), + "private-route-table-association-2": fnv1.Resource( + resource=resource.dict_to_struct(_private_route_table_association("us-west-2c")), + ), + "iam-role-cluster": fnv1.Resource( + resource=resource.dict_to_struct(_role("cluster", _ASSUME_CLUSTER)), + ), + "iam-attach-cluster-policy": fnv1.Resource( + resource=resource.dict_to_struct( + _role_policy_attachment("cluster", "arn:aws:iam::aws:policy/AmazonEKSClusterPolicy"), ), - "pod-identity-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler_pod_identity()), + ), + "iam-role-node": fnv1.Resource(resource=resource.dict_to_struct(_role("node", _ASSUME_NODE))), + "iam-attach-node-worker": fnv1.Resource( + resource=resource.dict_to_struct( + _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy"), ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config("kubernetes.m.crossplane.io/v1alpha1")), - ready=fnv1.READY_TRUE, + ), + "iam-attach-node-cni": fnv1.Resource( + resource=resource.dict_to_struct( + _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy"), ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config("helm.m.crossplane.io/v1beta1")), - ready=fnv1.READY_TRUE, + ), + "iam-attach-node-ecr": fnv1.Resource( + resource=resource.dict_to_struct( + _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly"), ), - } + ), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_eks_cluster())), + "cluster-auth": fnv1.Resource(resource=resource.dict_to_struct(_cluster_auth())), + "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_system_node_group())), + "launch-template-gpu-h200": fnv1.Resource(resource=resource.dict_to_struct(_launch_template_efa())), + "efa-security-group": fnv1.Resource(resource=resource.dict_to_struct(_efa_security_group())), + "efa-security-group-ingress": fnv1.Resource( + resource=resource.dict_to_struct(_efa_security_group_ingress()), + ), + "efa-security-group-egress": fnv1.Resource( + resource=resource.dict_to_struct(_efa_security_group_egress()), + ), + "nodegroup-gpu-h200": fnv1.Resource(resource=resource.dict_to_struct(_gpu_node_group_efa())), + "addon-vpc-cni": fnv1.Resource(resource=resource.dict_to_struct(_addon("vpc-cni"))), + "addon-kube-proxy": fnv1.Resource(resource=resource.dict_to_struct(_addon("kube-proxy"))), + "addon-coredns": fnv1.Resource(resource=resource.dict_to_struct(_addon("coredns"))), + "efs-filesystem": fnv1.Resource(resource=resource.dict_to_struct(_efs_filesystem())), + "efs-security-group": fnv1.Resource(resource=resource.dict_to_struct(_efs_security_group())), + "efs-security-group-ingress": fnv1.Resource( + resource=resource.dict_to_struct(_efs_security_group_ingress()), + ), + "efs-mount-target-0": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_A))), + "efs-mount-target-1": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_B))), + "efs-mount-target-2": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_C))), + "iam-role-efs-csi": fnv1.Resource( + resource=resource.dict_to_struct(_role("efs-csi", _ASSUME_POD_IDENTITY)), + ), + "iam-attach-efs-csi": fnv1.Resource( + resource=resource.dict_to_struct(_role_policy_attachment("efs-csi", _POLICY_EFS_CSI)), + ), + "addon-eks-pod-identity-agent": fnv1.Resource( + resource=resource.dict_to_struct(_addon("eks-pod-identity-agent")), + ), + "pod-identity-efs-csi": fnv1.Resource(resource=resource.dict_to_struct(_pod_identity_association())), + "addon-aws-efs-csi-driver": fnv1.Resource( + resource=resource.dict_to_struct(_addon("aws-efs-csi-driver")), + ), + "iam-policy-cluster-autoscaler": fnv1.Resource(resource=resource.dict_to_struct(_autoscaler_policy())), + "iam-role-cluster-autoscaler": fnv1.Resource( + resource=resource.dict_to_struct(_role("cluster-autoscaler", _ASSUME_POD_IDENTITY)), + ), + "iam-attach-cluster-autoscaler": fnv1.Resource( + resource=resource.dict_to_struct(_autoscaler_attachment()), + ), + "pod-identity-cluster-autoscaler": fnv1.Resource( + resource=resource.dict_to_struct(_autoscaler_pod_identity()), + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config("kubernetes.m.crossplane.io/v1alpha1")), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config("helm.m.crossplane.io/v1beta1")), + ready=fnv1.READY_TRUE, + ), + } - case = Case( - name="an EFA pool composes EFA launch template, security group, and rules", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr_efa().model_dump(exclude_none=True, mode="json"), - ), + case = Case( + name="an EFA pool composes EFA launch template, security group, and rules", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr_efa().model_dump(exclude_none=True, mode="json"), ), ), ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources=want_resources, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct(_expected_status()), ), - context=structpb.Struct(), + resources=want_resources, ), - ) - - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - ) + context=structpb.Struct(), + ), + ) - async def test_compose_efa_cluster_security_group(self) -> None: - """Once both security groups are observed, every interface carries them. - - A launch template with networkInterfaces makes its security groups - authoritative, so the interfaces must carry both the EFA security group - and the EKS cluster security group or the node never joins. Both are set - as raw IDs in securityGroups (not securityGroupRefs): the provider's - reference resolver no-ops once that field is populated, so a ref mixed - with a literal would be dropped. The EFA group's ID comes from its - observed external name, the cluster group's from the observed cluster's - status, so both appear only once their resources report them. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr_efa().model_dump(exclude_none=True, mode="json"), - ), + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) + + +def test_compose_efa_cluster_security_group() -> None: + """Once both security groups are observed, every interface carries them.""" + # A launch template with networkInterfaces makes its security groups + # authoritative, so the interfaces must carry both the EFA security group + # and the EKS cluster security group or the node never joins. Both are set + # as raw IDs in securityGroups (not securityGroupRefs): the provider's + # reference resolver no-ops once that field is populated, so a ref mixed + # with a literal would be dropped. The EFA group's ID comes from its + # observed external name, the cluster group's from the observed cluster's + # status, so both appear only once their resources report them. + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr_efa().model_dump(exclude_none=True, mode="json"), ), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_eks_cluster(), - "status": { - "atProvider": { - "vpcConfig": {"clusterSecurityGroupId": "sg-0cluster"}, - }, + ), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + **_eks_cluster(), + "status": { + "atProvider": { + "vpcConfig": {"clusterSecurityGroupId": "sg-0cluster"}, }, }, - ), + }, ), - "efa-security-group": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_efa_security_group(), - "metadata": { - **_efa_security_group()["metadata"], - "annotations": {"crossplane.io/external-name": "sg-0efa"}, - }, + ), + "efa-security-group": fnv1.Resource( + resource=resource.dict_to_struct( + { + **_efa_security_group(), + "metadata": { + **_efa_security_group()["metadata"], + "annotations": {"crossplane.io/external-name": "sg-0efa"}, }, - ), + }, ), - }, - ), - ) - - got = await self.runner.RunFunction(req, None) - lt = resource.struct_to_dict(got.desired.resources["launch-template-gpu-h200"].resource) - interfaces = lt["spec"]["forProvider"]["networkInterfaces"] - - # Every interface carries both SGs as raw IDs (EFA first, then cluster) - # and no securityGroupRefs; no interface requests a public IP (nodes are - # in private subnets). - self.assertEqual("efa", interfaces[0]["interfaceType"]) - for ni in interfaces: - self.assertNotIn("securityGroupRefs", ni) - self.assertEqual(["sg-0efa", "sg-0cluster"], ni["securityGroups"]) - self.assertNotIn("associatePublicIpAddress", ni) - for ni in interfaces[1:]: - self.assertEqual("efa-only", ni["interfaceType"]) - - async def test_compose_efa_dra_driver(self) -> None: - """An EFA pool installs the EFA DRA driver Helm release. - - Like the autoscaler, the release is gated on the cluster being observed - so provider-helm can reach it. A pool without the EFA fabric installs no - driver even once the cluster is observed. - """ - observed_cluster = { - "cluster": fnv1.Resource( - resource=resource.dict_to_struct( - {**_eks_cluster(), "status": {"conditions": [_ready_condition()]}}, ), + }, + ), + ) + + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + lt = resource.struct_to_dict(got.desired.resources["launch-template-gpu-h200"].resource) + interfaces = lt["spec"]["forProvider"]["networkInterfaces"] + + # Every interface carries both SGs as raw IDs (EFA first, then cluster) + # and no securityGroupRefs; no interface requests a public IP (nodes are + # in private subnets). + assert interfaces[0]["interfaceType"] == "efa" + for ni in interfaces: + assert "securityGroupRefs" not in ni + assert ni["securityGroups"] == ["sg-0efa", "sg-0cluster"] + assert "associatePublicIpAddress" not in ni + for ni in interfaces[1:]: + assert ni["interfaceType"] == "efa-only" + + +def test_compose_efa_dra_driver() -> None: + """An EFA pool installs the EFA DRA driver Helm release, and a pool without EFA doesn't.""" + # Like the autoscaler, the release is gated on the cluster being observed + # so provider-helm can reach it. A pool without the EFA fabric installs no + # driver even once the cluster is observed. + observed_cluster = { + "cluster": fnv1.Resource( + resource=resource.dict_to_struct( + {**_eks_cluster(), "status": {"conditions": [_ready_condition()]}}, ), - } + ), + } - got_efa = await self.runner.RunFunction( + got_efa = asyncio.run( + fn.FunctionRunner().RunFunction( fnv1.RunFunctionRequest( observed=fnv1.State( composite=fnv1.Resource( @@ -1546,13 +1527,15 @@ async def test_compose_efa_dra_driver(self) -> None: ), None, ) - self.assertIn("release-efa-dra-driver", got_efa.desired.resources) - self.assertEqual( - _efa_dra_driver_release(), - resource.struct_to_dict(got_efa.desired.resources["release-efa-dra-driver"].resource), - ) + ) + assert "release-efa-dra-driver" in got_efa.desired.resources + assert ( + resource.struct_to_dict(got_efa.desired.resources["release-efa-dra-driver"].resource) + == _efa_dra_driver_release() + ) - got_none = await self.runner.RunFunction( + got_none = asyncio.run( + fn.FunctionRunner().RunFunction( fnv1.RunFunctionRequest( observed=fnv1.State( composite=fnv1.Resource( @@ -1563,100 +1546,97 @@ async def test_compose_efa_dra_driver(self) -> None: ), None, ) - self.assertNotIn("release-efa-dra-driver", got_none.desired.resources) - - async def test_custom_credentials(self) -> None: - """Custom credentials flow through to all cloud MRs. - - When spec.credentials is set with a custom type and name, every cloud - provider MR (VPC, subnets, IAM roles, EKS cluster, node groups, addons, - EFS resources, autoscaler IAM resources) carries the corresponding - providerConfigRef. The kubeconfig-based resources (provider-config-kubernetes, - provider-config-helm, release-*, storage-class-*) are unaffected. - """ - ck = "ProviderConfig" - cn = "my-aws-account" - creds = v1alpha1.Credentials(type=ck, name=cn) - - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr(credentials=creds).model_dump(exclude_none=True, mode="json"), - ), + ) + assert "release-efa-dra-driver" not in got_none.desired.resources + + +def test_custom_credentials() -> None: + """Custom credentials flow through to all cloud MRs.""" + # When spec.credentials is set with a custom type and name, every cloud + # provider MR (VPC, subnets, IAM roles, EKS cluster, node groups, addons, + # EFS resources, autoscaler IAM resources) carries the corresponding + # providerConfigRef. The kubeconfig-based resources (provider-config-kubernetes, + # provider-config-helm, release-*, storage-class-*) are unaffected. + ck = "ProviderConfig" + cn = "my-aws-account" + creds = v1alpha1.Credentials(type=ck, name=cn) + + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr(credentials=creds).model_dump(exclude_none=True, mode="json"), ), ), - ) + ), + ) - got = await self.runner.RunFunction(req, None) - rs = got.desired.resources - - cloud_checks = { - "vpc": _vpc(ck, cn), - "subnet-0": _subnet(_SUBNET_A, "us-west-2a", "10.0.0.0/20", ck, cn), - "subnet-1": _subnet(_SUBNET_B, "us-west-2b", "10.0.16.0/20", ck, cn), - "subnet-2": _subnet(_SUBNET_C, "us-west-2c", "10.0.32.0/20", ck, cn), - "private-subnet-0": _private_subnet(_PRIVATE_SUBNET_A, "us-west-2a", "10.0.48.0/20", ck, cn), - "private-subnet-1": _private_subnet(_PRIVATE_SUBNET_B, "us-west-2b", "10.0.64.0/20", ck, cn), - "private-subnet-2": _private_subnet(_PRIVATE_SUBNET_C, "us-west-2c", "10.0.80.0/20", ck, cn), - "internet-gateway": _internet_gateway(ck, cn), - "nat-eip": _nat_eip(ck, cn), - "nat-gateway": _nat_gateway("us-west-2a", ck, cn), - "route-table": _route_table(ck, cn), - "route-default": _route_default(ck, cn), - "private-route-table": _private_route_table(ck, cn), - "private-route-default": _private_route_default(ck, cn), - "route-table-association-0": _route_table_association("us-west-2a", ck, cn), - "route-table-association-1": _route_table_association("us-west-2b", ck, cn), - "route-table-association-2": _route_table_association("us-west-2c", ck, cn), - "private-route-table-association-0": _private_route_table_association("us-west-2a", ck, cn), - "private-route-table-association-1": _private_route_table_association("us-west-2b", ck, cn), - "private-route-table-association-2": _private_route_table_association("us-west-2c", ck, cn), - "iam-role-cluster": _role("cluster", _ASSUME_CLUSTER, ck, cn), - "iam-attach-cluster-policy": _role_policy_attachment( - "cluster", "arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", ck, cn - ), - "iam-role-node": _role("node", _ASSUME_NODE, ck, cn), - "iam-attach-node-worker": _role_policy_attachment( - "node", "arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", ck, cn - ), - "iam-attach-node-cni": _role_policy_attachment( - "node", "arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", ck, cn - ), - "iam-attach-node-ecr": _role_policy_attachment( - "node", "arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", ck, cn - ), - "cluster": _eks_cluster(ck, cn), - "cluster-auth": _cluster_auth(ck, cn), - "nodegroup-system": _system_node_group(ck, cn), - "nodegroup-gpu-l4": _gpu_node_group(ck, cn), - "addon-vpc-cni": _addon("vpc-cni", ck, cn), - "addon-kube-proxy": _addon("kube-proxy", ck, cn), - "addon-coredns": _addon("coredns", ck, cn), - "efs-filesystem": _efs_filesystem(ck, cn), - "efs-security-group": _efs_security_group(ck, cn), - "efs-security-group-ingress": _efs_security_group_ingress(ck, cn), - "efs-mount-target-0": _efs_mount_target(_PRIVATE_SUBNET_A, ck, cn), - "efs-mount-target-1": _efs_mount_target(_PRIVATE_SUBNET_B, ck, cn), - "efs-mount-target-2": _efs_mount_target(_PRIVATE_SUBNET_C, ck, cn), - "iam-role-efs-csi": _role("efs-csi", _ASSUME_POD_IDENTITY, ck, cn), - "iam-attach-efs-csi": _role_policy_attachment("efs-csi", _POLICY_EFS_CSI, ck, cn), - "addon-eks-pod-identity-agent": _addon("eks-pod-identity-agent", ck, cn), - "pod-identity-efs-csi": _pod_identity_association(ck, cn), - "addon-aws-efs-csi-driver": _addon("aws-efs-csi-driver", ck, cn), - "iam-policy-cluster-autoscaler": _autoscaler_policy(ck, cn), - "iam-role-cluster-autoscaler": _role("cluster-autoscaler", _ASSUME_POD_IDENTITY, ck, cn), - "iam-attach-cluster-autoscaler": _autoscaler_attachment(ck, cn), - "pod-identity-cluster-autoscaler": _autoscaler_pod_identity(ck, cn), - } - - for key, want in cloud_checks.items(): - with self.subTest(resource=key): - self.assertIn(key, rs, f"resource {key!r} not found in desired") - got_dict = resource.struct_to_dict(rs[key].resource) - self.assertEqual(want, got_dict, f"resource {key!r} mismatch") - - # kubeconfig-based resources must NOT carry the cloud providerConfigRef - for key in ("provider-config-kubernetes", "provider-config-helm"): - got_dict = resource.struct_to_dict(rs[key].resource) - self.assertNotIn("providerConfigRef", got_dict.get("spec", {}), f"{key} should not have providerConfigRef") + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + rs = got.desired.resources + + cloud_checks = { + "vpc": _vpc(ck, cn), + "subnet-0": _subnet(_SUBNET_A, "us-west-2a", "10.0.0.0/20", ck, cn), + "subnet-1": _subnet(_SUBNET_B, "us-west-2b", "10.0.16.0/20", ck, cn), + "subnet-2": _subnet(_SUBNET_C, "us-west-2c", "10.0.32.0/20", ck, cn), + "private-subnet-0": _private_subnet(_PRIVATE_SUBNET_A, "us-west-2a", "10.0.48.0/20", ck, cn), + "private-subnet-1": _private_subnet(_PRIVATE_SUBNET_B, "us-west-2b", "10.0.64.0/20", ck, cn), + "private-subnet-2": _private_subnet(_PRIVATE_SUBNET_C, "us-west-2c", "10.0.80.0/20", ck, cn), + "internet-gateway": _internet_gateway(ck, cn), + "nat-eip": _nat_eip(ck, cn), + "nat-gateway": _nat_gateway("us-west-2a", ck, cn), + "route-table": _route_table(ck, cn), + "route-default": _route_default(ck, cn), + "private-route-table": _private_route_table(ck, cn), + "private-route-default": _private_route_default(ck, cn), + "route-table-association-0": _route_table_association("us-west-2a", ck, cn), + "route-table-association-1": _route_table_association("us-west-2b", ck, cn), + "route-table-association-2": _route_table_association("us-west-2c", ck, cn), + "private-route-table-association-0": _private_route_table_association("us-west-2a", ck, cn), + "private-route-table-association-1": _private_route_table_association("us-west-2b", ck, cn), + "private-route-table-association-2": _private_route_table_association("us-west-2c", ck, cn), + "iam-role-cluster": _role("cluster", _ASSUME_CLUSTER, ck, cn), + "iam-attach-cluster-policy": _role_policy_attachment( + "cluster", "arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", ck, cn + ), + "iam-role-node": _role("node", _ASSUME_NODE, ck, cn), + "iam-attach-node-worker": _role_policy_attachment( + "node", "arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", ck, cn + ), + "iam-attach-node-cni": _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", ck, cn), + "iam-attach-node-ecr": _role_policy_attachment( + "node", "arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", ck, cn + ), + "cluster": _eks_cluster(ck, cn), + "cluster-auth": _cluster_auth(ck, cn), + "nodegroup-system": _system_node_group(ck, cn), + "nodegroup-gpu-l4": _gpu_node_group(ck, cn), + "addon-vpc-cni": _addon("vpc-cni", ck, cn), + "addon-kube-proxy": _addon("kube-proxy", ck, cn), + "addon-coredns": _addon("coredns", ck, cn), + "efs-filesystem": _efs_filesystem(ck, cn), + "efs-security-group": _efs_security_group(ck, cn), + "efs-security-group-ingress": _efs_security_group_ingress(ck, cn), + "efs-mount-target-0": _efs_mount_target(_PRIVATE_SUBNET_A, ck, cn), + "efs-mount-target-1": _efs_mount_target(_PRIVATE_SUBNET_B, ck, cn), + "efs-mount-target-2": _efs_mount_target(_PRIVATE_SUBNET_C, ck, cn), + "iam-role-efs-csi": _role("efs-csi", _ASSUME_POD_IDENTITY, ck, cn), + "iam-attach-efs-csi": _role_policy_attachment("efs-csi", _POLICY_EFS_CSI, ck, cn), + "addon-eks-pod-identity-agent": _addon("eks-pod-identity-agent", ck, cn), + "pod-identity-efs-csi": _pod_identity_association(ck, cn), + "addon-aws-efs-csi-driver": _addon("aws-efs-csi-driver", ck, cn), + "iam-policy-cluster-autoscaler": _autoscaler_policy(ck, cn), + "iam-role-cluster-autoscaler": _role("cluster-autoscaler", _ASSUME_POD_IDENTITY, ck, cn), + "iam-attach-cluster-autoscaler": _autoscaler_attachment(ck, cn), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity(ck, cn), + } + + for key, want in cloud_checks.items(): + assert key in rs, f"resource {key!r} not found in desired" + got_dict = resource.struct_to_dict(rs[key].resource) + assert got_dict == want, f"resource {key!r} mismatch" + + # kubeconfig-based resources must NOT carry the cloud providerConfigRef + for key in ("provider-config-kubernetes", "provider-config-helm"): + got_dict = resource.struct_to_dict(rs[key].resource) + assert "providerConfigRef" not in got_dict.get("spec", {}), f"{key} should not have providerConfigRef" diff --git a/functions/compose-gke-cluster/tests/__init__.py b/functions/compose-gke-cluster/tests/__init__.py deleted file mode 100644 index 5d373016d..000000000 --- a/functions/compose-gke-cluster/tests/__init__.py +++ /dev/null @@ -1,15 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - - diff --git a/functions/compose-gke-cluster/tests/test_fn.py b/functions/compose-gke-cluster/tests/test_fn.py index 2f556d9de..07898ac6b 100644 --- a/functions/compose-gke-cluster/tests/test_fn.py +++ b/functions/compose-gke-cluster/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-gke-cluster function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.gkecluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -36,10 +38,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - _DEFAULT_CRED_KIND = "ClusterProviderConfig" _DEFAULT_CRED_NAME = "default" @@ -384,345 +382,330 @@ def _expected_status() -> dict: } -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function composes GKE cluster infrastructure.""" - req1 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr().model_dump(exclude_none=True, mode="json"), - ), +def _compose_cases() -> list[Case]: + """The cases for test_compose.""" + req1 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _gke_xr().model_dump(exclude_none=True, mode="json"), ), ), - ) - req1.required_resources["gcp-provider-config"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(_GCP_PROVIDER_CONFIG)) - ) - - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ), - "projectservice-filestore": fnv1.Resource( - resource=resource.dict_to_struct(_projectservice_filestore()), - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "nodepool-system": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_system()), - ), - "nodepool-gpu-pool": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu()), - ), - "service-account": fnv1.Resource( - resource=resource.dict_to_struct(_service_account()), - ), - "service-account-key": fnv1.Resource( - resource=resource.dict_to_struct(_service_account_key()), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_kubernetes()), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - }, + ), + ) + req1.required_resources["gcp-provider-config"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(_GCP_PROVIDER_CONFIG)) + ) + + want1 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct(_expected_status()), ), - context=structpb.Struct(), - ) - want1.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct(_network()), + ), + "projectservice-filestore": fnv1.Resource( + resource=resource.dict_to_struct(_projectservice_filestore()), + ), + "subnet": fnv1.Resource( + resource=resource.dict_to_struct(_subnet()), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ), + "nodepool-system": fnv1.Resource( + resource=resource.dict_to_struct(_nodepool_system()), + ), + "nodepool-gpu-pool": fnv1.Resource( + resource=resource.dict_to_struct(_nodepool_gpu()), + ), + "service-account": fnv1.Resource( + resource=resource.dict_to_struct(_service_account()), + ), + "service-account-key": fnv1.Resource( + resource=resource.dict_to_struct(_service_account_key()), + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config_kubernetes()), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config_helm()), + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + ) + want1.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) - req2 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr().model_dump(exclude_none=True, mode="json"), - ), + req2 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _gke_xr().model_dump(exclude_none=True, mode="json"), ), - resources={ - "service-account": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ServiceAccount", - "spec": { - "forProvider": {}, + ), + resources={ + "service-account": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ServiceAccount", + "spec": { + "forProvider": {}, + }, + "status": { + "atProvider": { + "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", }, - "status": { - "atProvider": { - "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", }, - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - }, - } - ), + ], + }, + } ), - "network": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "compute.gcp.m.upbound.io/v1beta1", - "kind": "Network", - # The external-name annotation carries the - # provider-generated VPC name, which the - # function pins the Filestore StorageClass to. - "metadata": { - "annotations": {"crossplane.io/external-name": "test-cluster-abc12"}, + ), + "network": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.gcp.m.upbound.io/v1beta1", + "kind": "Network", + # The external-name annotation carries the + # provider-generated VPC name, which the + # function pins the Filestore StorageClass to. + "metadata": { + "annotations": {"crossplane.io/external-name": "test-cluster-abc12"}, + }, + "spec": { + "forProvider": { + "autoCreateSubnetworks": False, }, - "spec": { - "forProvider": { - "autoCreateSubnetworks": False, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", }, - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - }, - } - ), + ], + }, + } ), - }, - ), - ) - req2.required_resources["gcp-provider-config"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(_GCP_PROVIDER_CONFIG)) - ) - - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), ), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ready=fnv1.READY_TRUE, - ), - "projectservice-filestore": fnv1.Resource( - resource=resource.dict_to_struct(_projectservice_filestore()), - ), - # With the network name known, the managed Filestore - # StorageClass is composed against the cluster's own - # provider-kubernetes ProviderConfig, pinned to the - # observed VPC. StorageClass has no Ready condition, - # so readiness is SuccessfulCreate. It's orphaned (no - # Delete policy) so it dies with the cluster instead of - # wedging on a deleted kubeconfig Secret during teardown. - "storage-class-rwx": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class_rwx("test-cluster-abc12")), - ready=fnv1.READY_TRUE, - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "nodepool-system": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_system()), - ), - "nodepool-gpu-pool": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu()), - ), - "service-account": fnv1.Resource( - resource=resource.dict_to_struct(_service_account()), - ready=fnv1.READY_TRUE, - ), - "service-account-key": fnv1.Resource( - resource=resource.dict_to_struct(_service_account_key()), - ), - "iam-binding": fnv1.Resource( - resource=resource.dict_to_struct(_iam_binding()), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_kubernetes()), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - }, + }, + ), + ) + req2.required_resources["gcp-provider-config"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(_GCP_PROVIDER_CONFIG)) + ) + + want2 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct(_expected_status()), ), - context=structpb.Struct(), - ) - want2.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) - - # The ProviderConfig resolved to nothing and no cluster is observed to - # take the project from, so nothing can be composed. The XR is marked - # not ready rather than left to aggregate to trivially ready. - req3 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr().model_dump(exclude_none=True, mode="json"), - ), + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct(_network()), + ready=fnv1.READY_TRUE, ), - ), - ) - req3.required_resources["gcp-provider-config"].SetInParent() - - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for GCP ClusterProviderConfig default", + "projectservice-filestore": fnv1.Resource( + resource=resource.dict_to_struct(_projectservice_filestore()), + ), + # With the network name known, the managed Filestore + # StorageClass is composed against the cluster's own + # provider-kubernetes ProviderConfig, pinned to the + # observed VPC. StorageClass has no Ready condition, + # so readiness is SuccessfulCreate. It's orphaned (no + # Delete policy) so it dies with the cluster instead of + # wedging on a deleted kubeconfig Secret during teardown. + "storage-class-rwx": fnv1.Resource( + resource=resource.dict_to_struct(_storage_class_rwx("test-cluster-abc12")), + ready=fnv1.READY_TRUE, + ), + "subnet": fnv1.Resource( + resource=resource.dict_to_struct(_subnet()), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ), + "nodepool-system": fnv1.Resource( + resource=resource.dict_to_struct(_nodepool_system()), + ), + "nodepool-gpu-pool": fnv1.Resource( + resource=resource.dict_to_struct(_nodepool_gpu()), + ), + "service-account": fnv1.Resource( + resource=resource.dict_to_struct(_service_account()), + ready=fnv1.READY_TRUE, + ), + "service-account-key": fnv1.Resource( + resource=resource.dict_to_struct(_service_account_key()), + ), + "iam-binding": fnv1.Resource( + resource=resource.dict_to_struct(_iam_binding()), + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config_kubernetes()), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config_helm()), + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + ) + want2.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) + + # The ProviderConfig resolved to nothing and no cluster is observed to + # take the project from, so nothing can be composed. The XR is marked + # not ready rather than left to aggregate to trivially ready. + req3 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _gke_xr().model_dump(exclude_none=True, mode="json"), ), - ], - context=structpb.Struct(), - ) - want3.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) - - cases = [ - Case(name="first pass composes infra resources; IAM binding gated", req=req1, want=want1), - Case(name="a missing ProviderConfig composes nothing and isn't ready", req=req3, want=want3), - Case( - name="second pass with observed SA email composes IAM binding and marks ready resources", - req=req2, - want=want2, ), - ] - - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) - - async def test_custom_credentials(self) -> None: - """Custom credentials flow through to all cloud MRs. - - When spec.credentials is set with a custom type and name, every cloud - provider MR (Network, ProjectService, Subnetwork, Cluster, NodePools, - ServiceAccount, ServiceAccountKey, IAM binding) carries the corresponding - providerConfigRef. The kubeconfig-based resources (provider-config-kubernetes, - provider-config-helm, storage-class-rwx) are unaffected. - """ - ck = "ProviderConfig" - cn = "my-gcp-account" - creds = v1alpha1.Credentials(type=ck, name=cn) - custom_pc = { - "apiVersion": "gcp.m.upbound.io/v1beta1", - "kind": "ProviderConfig", - "metadata": {"name": cn, "namespace": "crossplane-system"}, - "spec": { - "projectID": "my-gcp-project", - "credentials": { - "source": "Secret", - "secretRef": { - "name": "gcp-credentials", - "namespace": "crossplane-system", - "key": "credentials", - }, + ), + ) + req3.required_resources["gcp-provider-config"].SetInParent() + + want3 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for GCP ClusterProviderConfig default", + ), + ], + context=structpb.Struct(), + ) + want3.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) + + return [ + Case(name="first pass composes infra resources; IAM binding gated", req=req1, want=want1), + Case(name="a missing ProviderConfig composes nothing and isn't ready", req=req3, want=want3), + Case( + name="second pass with observed SA email composes IAM binding and marks ready resources", + req=req2, + want=want2, + ), + ] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes GKE cluster infrastructure.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) + + +def test_custom_credentials() -> None: + """Custom credentials flow through to all cloud MRs.""" + # When spec.credentials is set with a custom type and name, every cloud + # provider MR (Network, ProjectService, Subnetwork, Cluster, NodePools, + # ServiceAccount, ServiceAccountKey, IAM binding) carries the corresponding + # providerConfigRef. The kubeconfig-based resources (provider-config-kubernetes, + # provider-config-helm, storage-class-rwx) are unaffected. + ck = "ProviderConfig" + cn = "my-gcp-account" + creds = v1alpha1.Credentials(type=ck, name=cn) + custom_pc = { + "apiVersion": "gcp.m.upbound.io/v1beta1", + "kind": "ProviderConfig", + "metadata": {"name": cn, "namespace": "crossplane-system"}, + "spec": { + "projectID": "my-gcp-project", + "credentials": { + "source": "Secret", + "secretRef": { + "name": "gcp-credentials", + "namespace": "crossplane-system", + "key": "credentials", }, }, - } + }, + } - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr(credentials=creds).model_dump(exclude_none=True, mode="json"), - ), + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _gke_xr(credentials=creds).model_dump(exclude_none=True, mode="json"), ), - resources={ - "service-account": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ServiceAccount", - "spec": { - "forProvider": {}, - }, - "status": { - "atProvider": { - "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", - }, + ), + resources={ + "service-account": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ServiceAccount", + "spec": { + "forProvider": {}, + }, + "status": { + "atProvider": { + "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", }, - } - ), + }, + } ), - }, - ), - ) - req.required_resources["gcp-provider-config"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(custom_pc)) - ) - - got = await self.runner.RunFunction(req, None) - rs = got.desired.resources - - cloud_checks = { - "network": _network(ck, cn), - "projectservice-filestore": _projectservice_filestore(ck, cn), - "subnet": _subnet(ck, cn), - "cluster": _cluster(ck, cn), - "nodepool-system": _nodepool_system(ck, cn), - "nodepool-gpu-pool": _nodepool_gpu(ck, cn), - "service-account": _service_account(ck, cn), - "service-account-key": _service_account_key(ck, cn), - "iam-binding": _iam_binding("test-sa@my-gcp-project.iam.gserviceaccount.com", ck, cn), - } - - for key, want in cloud_checks.items(): - with self.subTest(resource=key): - self.assertIn(key, rs, f"resource {key!r} not found in desired") - got_dict = resource.struct_to_dict(rs[key].resource) - self.assertEqual(want, got_dict, f"resource {key!r} mismatch") - - # kubeconfig-based resources must NOT carry the cloud providerConfigRef - for key in ("provider-config-kubernetes", "provider-config-helm"): - got_dict = resource.struct_to_dict(rs[key].resource) - self.assertNotIn( - "providerConfigRef", - got_dict.get("spec", {}), - f"{key} should not have providerConfigRef", - ) - - custom_selector = fnv1.ResourceSelector( - api_version="gcp.m.upbound.io/v1beta1", - kind="ProviderConfig", - match_name=cn, - namespace="modelplane-system", - ) - self.assertEqual( - custom_selector, - got.requirements.resources["gcp-provider-config"], - ) + ), + }, + ), + ) + req.required_resources["gcp-provider-config"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(custom_pc)) + ) + + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + rs = got.desired.resources + + cloud_checks = { + "network": _network(ck, cn), + "projectservice-filestore": _projectservice_filestore(ck, cn), + "subnet": _subnet(ck, cn), + "cluster": _cluster(ck, cn), + "nodepool-system": _nodepool_system(ck, cn), + "nodepool-gpu-pool": _nodepool_gpu(ck, cn), + "service-account": _service_account(ck, cn), + "service-account-key": _service_account_key(ck, cn), + "iam-binding": _iam_binding("test-sa@my-gcp-project.iam.gserviceaccount.com", ck, cn), + } + + got_cloud = {key: resource.struct_to_dict(r.resource) for key, r in rs.items() if key in cloud_checks} + assert got_cloud == cloud_checks + + # kubeconfig-based resources must NOT carry the cloud providerConfigRef + for key in ("provider-config-kubernetes", "provider-config-helm"): + got_dict = resource.struct_to_dict(rs[key].resource) + assert "providerConfigRef" not in got_dict.get("spec", {}), f"{key} should not have providerConfigRef" + + custom_selector = fnv1.ResourceSelector( + api_version="gcp.m.upbound.io/v1beta1", + kind="ProviderConfig", + match_name=cn, + namespace="modelplane-system", + ) + assert _to_dict(got.requirements.resources["gcp-provider-config"]) == _to_dict(custom_selector) diff --git a/functions/compose-inference-class/tests/__init__.py b/functions/compose-inference-class/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-inference-class/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-inference-class/tests/test_fn.py b/functions/compose-inference-class/tests/test_fn.py index d4c5c99c4..327807286 100644 --- a/functions/compose-inference-class/tests/test_fn.py +++ b/functions/compose-inference-class/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-inference-class function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.inferenceclass import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -36,70 +38,60 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function marks the InferenceClass as ready.""" - cases = [ - Case( - name="marks XR ready with Accepted condition and empty status", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceClass( - metadata=metav1.ObjectMeta(name="gpu-l4"), - spec=v1alpha1.Spec( - devices=[ - v1alpha1.Device( - name="gpu", - claim="DRA", - driver="gpu.nvidia.com", - deviceClassName="gpu.nvidia.com", - count=1, - capacity={"memory": v1alpha1.Capacity(value="24Gi")}, - ), - ], +COMPOSE_CASES = [ + Case( + name="marks XR ready with Accepted condition and empty status", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceClass( + metadata=metav1.ObjectMeta(name="gpu-l4"), + spec=v1alpha1.Spec( + devices=[ + v1alpha1.Device( + name="gpu", + claim="DRA", + driver="gpu.nvidia.com", + deviceClassName="gpu.nvidia.com", + count=1, + capacity={"memory": v1alpha1.Capacity(value="24Gi")}, ), - ).model_dump(exclude_none=True, mode="json") + ], ), - ), + ).model_dump(exclude_none=True, mode="json") ), ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {}}), - ready=fnv1.READY_TRUE, - ), - ), - conditions=[ - fnv1.Condition( - type="Accepted", - status=fnv1.STATUS_CONDITION_TRUE, - reason="Available", - ), - ], - context=structpb.Struct(), + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {}}), + ready=fnv1.READY_TRUE, ), ), - ] + conditions=[ + fnv1.Condition( + type="Accepted", + status=fnv1.STATUS_CONDITION_TRUE, + reason="Available", + ), + ], + context=structpb.Struct(), + ), + ), +] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction marks the InferenceClass ready.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-inference-cluster/tests/__init__.py b/functions/compose-inference-cluster/tests/__init__.py deleted file mode 100644 index 5d373016d..000000000 --- a/functions/compose-inference-cluster/tests/__init__.py +++ /dev/null @@ -1,15 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - - diff --git a/functions/compose-inference-cluster/tests/test_fn.py b/functions/compose-inference-cluster/tests/test_fn.py index b35de1639..9b7625902 100644 --- a/functions/compose-inference-cluster/tests/test_fn.py +++ b/functions/compose-inference-cluster/tests/test_fn.py @@ -14,15 +14,17 @@ """Tests for the compose-inference-cluster function.""" +import asyncio import copy import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.inferencecluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -42,10 +44,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - _ACTIVATION_API_VERSION = "apiextensions.crossplane.io/v1alpha1" @@ -420,2114 +418,2391 @@ def _early_return_guard_case() -> tuple[fnv1.RunFunctionRequest, fnv1.RunFunctio return req, want -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() +def _compose_cases() -> list[Case]: # noqa: PLR0915 + """The RunFunction cases, built from shared bases. - async def test_compose(self) -> None: # noqa: PLR0915 - """The function composes an InferenceCluster. - - Many table entries, each exercising a distinct compose path across the - GKE, EKS, and Existing sources, push this over the statement limit. - """ - # Shared InferenceClass resource for required_resources. - inference_class_l4 = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-l4"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", + Many table entries, each exercising a distinct compose path across the + GKE, EKS, and Existing sources, push this over the statement limit. + """ + # Shared InferenceClass resource for required_resources. + inference_class_l4 = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceClass", + "metadata": {"name": "gpu-l4"}, + "spec": { + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + "provisioning": { + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": { + "type": "nvidia-l4", "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], - "provisioning": { - "provider": "GKE", - "gke": { - "machineType": "g2-standard-48", - "diskSizeGb": 100, - "accelerator": { - "type": "nvidia-l4", - "count": 1, - }, }, }, }, - } - - # Shared resource selector for class requirement. - class_selector = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-l4", - ) + }, + } + + # Shared resource selector for class requirement. + class_selector = fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceClass", + match_name="gpu-l4", + ) - # --- Case 1: Existing cluster with secrets composes backend and CPC. --- - req1 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing( - secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), - ), + # --- Case 1: Existing cluster with secrets composes backend and CPC. --- + req1 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing( + secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - ), - ], ), - ).model_dump(exclude_none=True, mode="json") - ), + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + ), + ], + ), + ).model_dump(exclude_none=True, mode="json") ), ), - ) - req1.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) + ), + ) + req1.required_resources["class-gpu-l4"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) + ) - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + want1 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + }, + } + ), + ), + resources={ + "cluster-provider-config-kubernetes": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "my-kubeconfig", + "key": "kubeconfig", + }, }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "serving-stack": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "cloud": "Existing", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "type": "Kubeconfig", + "name": "my-kubeconfig", + "key": "kubeconfig", }, ], }, } ), ), - resources={ - "cluster-provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "my-kubeconfig", - "key": "kubeconfig", - }, - }, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Existing", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "my-kubeconfig", - "key": "kubeconfig", - }, - ], - }, - } - ), - ), - }, + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ], - context=structpb.Struct(), - ) - want1.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Installing", + ), + ], + context=structpb.Struct(), + ) + want1.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - # --- Case 1b: Existing cluster with a non-GCP identity threads the - # declared identity type into the CPC and the ServingStack. --- - req1b = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing( - secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), - identitySecretRef=v1alpha1.IdentitySecretRef( - name="nebius-creds", - key="credentials.json", - type="NebiusServiceAccountCredentials", - ), + # --- Case 1b: Existing cluster with a non-GCP identity threads the + # declared identity type into the CPC and the ServingStack. --- + req1b = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing( + secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), + identitySecretRef=v1alpha1.IdentitySecretRef( + name="nebius-creds", + key="credentials.json", + type="NebiusServiceAccountCredentials", ), ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - ), - ], ), - ).model_dump(exclude_none=True, mode="json") - ), + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + ), + ], + ), + ).model_dump(exclude_none=True, mode="json") ), ), - ) - req1b.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) + ), + ) + req1b.required_resources["class-gpu-l4"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) + ) - # want1b mirrors want1 but with the Nebius identity on the CPC and an - # extra ServingStack identity secret of the same type. - want1b = fnv1.RunFunctionResponse() - want1b.CopyFrom(want1) - cpc1b = want1b.desired.resources["cluster-provider-config-kubernetes"] - cpc1b_dict = resource.struct_to_dict(cpc1b.resource) - cpc1b_dict["spec"]["identity"] = { + # want1b mirrors want1 but with the Nebius identity on the CPC and an + # extra ServingStack identity secret of the same type. + want1b = fnv1.RunFunctionResponse() + want1b.CopyFrom(want1) + cpc1b = want1b.desired.resources["cluster-provider-config-kubernetes"] + cpc1b_dict = resource.struct_to_dict(cpc1b.resource) + cpc1b_dict["spec"]["identity"] = { + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "nebius-creds", + "key": "credentials.json", + }, + } + cpc1b.resource.CopyFrom(resource.dict_to_struct(cpc1b_dict)) + backend1b = want1b.desired.resources["serving-stack"] + backend1b_dict = resource.struct_to_dict(backend1b.resource) + backend1b_dict["spec"]["secrets"].append( + { "type": "NebiusServiceAccountCredentials", - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "nebius-creds", - "key": "credentials.json", - }, + "name": "nebius-creds", + "key": "credentials.json", } - cpc1b.resource.CopyFrom(resource.dict_to_struct(cpc1b_dict)) - backend1b = want1b.desired.resources["serving-stack"] - backend1b_dict = resource.struct_to_dict(backend1b.resource) - backend1b_dict["spec"]["secrets"].append( - { - "type": "NebiusServiceAccountCredentials", - "name": "nebius-creds", - "key": "credentials.json", - } - ) - backend1b.resource.CopyFrom(resource.dict_to_struct(backend1b_dict)) - # want1 gains the replica, route, cache and gateway requirements in place - # from the guard cases below, after this snapshot; add them here so want1b - # matches on its own. - want1b.requirements.resources["gateways"].CopyFrom(_gateways_selector()) - want1b.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) - want1b.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) - want1b.requirements.resources["model-caches"].CopyFrom(_caches_selector()) - - # --- Case 2: GKE cluster first pass - no observed GKE, classes resolved. --- - req2 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="GKE", - gke=v1alpha1.Gke( - region="us-central1", - ), + ) + backend1b.resource.CopyFrom(resource.dict_to_struct(backend1b_dict)) + # want1 gains the replica, route, cache and gateway requirements in place + # from the guard cases below, after this snapshot; add them here so want1b + # matches on its own. + want1b.requirements.resources["gateways"].CopyFrom(_gateways_selector()) + want1b.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) + want1b.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) + want1b.requirements.resources["model-caches"].CopyFrom(_caches_selector()) + + # --- Case 2: GKE cluster first pass - no observed GKE, classes resolved. --- + req2 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="GKE", + gke=v1alpha1.Gke( + region="us-central1", ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - zones=["us-central1-a"], - ), - ], ), - ).model_dump(exclude_none=True, mode="json") - ), + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], + ), + ], + ), + ).model_dump(exclude_none=True, mode="json") ), ), - ) - req2.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) + ), + ) + req2.required_resources["class-gpu-l4"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) + ) - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + want2 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + }, + } + ), + ), + resources={ + "gke-cluster": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "metadata": { + "name": "test-cluster", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "region": "us-central1", + "kubernetesVersion": "1.35", + "nodePools": [ { "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "role": "GPU", + "machineType": "g2-standard-48", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": { + "acceleratorType": "nvidia-l4", + "acceleratorCount": 1, + }, + "zones": ["us-central1-a"], }, ], }, } ), ), - resources={ - "gke-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-central1", - "kubernetesVersion": "1.35", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "machineType": "g2-standard-48", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - "acceleratorCount": 1, - }, - "zones": ["us-central1-a"], - }, - ], - }, - } - ), - ), - }, + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want2.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want2.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - # --- Case 3: Existing cluster second pass - backend observed ready. --- - req3 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing( - secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), - ), + # --- Case 3: Existing cluster second pass - backend observed ready. --- + req3 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing( + secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - ), - ], ), - ).model_dump(exclude_none=True, mode="json") - ), - ), - resources={ - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": {"name": "test-cluster-serving-stack-fd00b"}, - "status": { - "conditions": [{"type": "Ready", "status": "True"}], - "gateway": {"address": "34.55.100.10"}, - }, - } + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + ), + ], ), - ), - }, + ).model_dump(exclude_none=True, mode="json") + ), ), - ) - req3.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + resources={ + "serving-stack": fnv1.Resource( resource=resource.dict_to_struct( { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": {"name": "test-cluster-serving-stack-fd00b"}, "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ - { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], - }, - ], + "conditions": [{"type": "Ready", "status": "True"}], "gateway": {"address": "34.55.100.10"}, }, } ), ), - resources={ - "cluster-provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "my-kubeconfig", - "key": "kubeconfig", - }, - }, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Existing", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ + }, + ), + ) + req3.required_resources["class-gpu-l4"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) + ) + + want3 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "type": "Kubeconfig", - "name": "my-kubeconfig", - "key": "kubeconfig", + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - } - ), - ready=fnv1.READY_TRUE, - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="BackendHealthy", + ], + "gateway": {"address": "34.55.100.10"}, + }, + } ), - ], - context=structpb.Struct(), - ) - want3.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - - # --- Case 4: EKS cluster first pass - no observed EKS, classes resolved. --- - inference_class_l4_eks = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-l4-eks"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], - "provisioning": { - "provider": "EKS", - "eks": { - "instanceType": "g6.xlarge", - "diskSizeGb": 100, - "accelerator": {"type": "nvidia-l4", "count": 1}, - }, - }, - }, - } - class_selector_eks = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-l4-eks", - ) - - req4 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( + ), + resources={ + "cluster-provider-config-kubernetes": fnv1.Resource( resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="EKS", - eks=v1alpha1.Eks(region="us-west-2"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4-eks", - nodeCount=2, - maxNodeCount=4, - zones=["us-west-2a", "us-west-2b"], - ), - ], - ), - ).model_dump(exclude_none=True, mode="json"), + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "my-kubeconfig", + "key": "kubeconfig", + }, + }, + }, + } ), + ready=fnv1.READY_TRUE, ), - ), - ) - req4.required_resources["class-gpu-l4-eks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), - ) - - want4 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + "serving-stack": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "cloud": "Existing", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "type": "Kubeconfig", + "name": "my-kubeconfig", + "key": "kubeconfig", }, ], }, - }, + } ), + ready=fnv1.READY_TRUE, ), - resources={ - "eks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-west-2", - "kubernetesVersion": "1.36", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - }, - "zones": ["us-west-2a", "us-west-2b"], - }, - ], - }, - }, - ), - ), - }, + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="BackendHealthy", + ), + ], + context=structpb.Struct(), + ) + want3.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) + + # --- Case 4: EKS cluster first pass - no observed EKS, classes resolved. --- + inference_class_l4_eks = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceClass", + "metadata": {"name": "gpu-l4-eks"}, + "spec": { + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, ], - context=structpb.Struct(), - ) - want4.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) - - # --- Case 8: EKS first pass with a node pool backed by a Capacity - # Block. The reservation ID flows through to the EKSCluster node pool's - # capacityBlock, which compose-eks-cluster turns into a CAPACITY_BLOCK - # node group. --- - req8 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", + "provisioning": { + "provider": "EKS", + "eks": { + "instanceType": "g6.xlarge", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + }, + } + class_selector_eks = fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceClass", + match_name="gpu-l4-eks", + ) + + req4 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="EKS", + eks=v1alpha1.Eks(region="us-west-2"), ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="EKS", - eks=v1alpha1.Eks(region="us-west-2"), + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a", "us-west-2b"], ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4-eks", - nodeCount=2, - maxNodeCount=4, - zones=["us-west-2a"], - capacityBlock=v1alpha1.CapacityBlock( - capacityReservationId="cr-0123456789abcdef0", - ), - ), - ], - ), - ).model_dump(exclude_none=True, mode="json"), - ), + ], + ), + ).model_dump(exclude_none=True, mode="json"), ), ), - ) - req8.required_resources["class-gpu-l4-eks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), - ) + ), + ) + req4.required_resources["class-gpu-l4-eks"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), + ) - want8 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + want4 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + }, + }, + ), + ), + resources={ + "eks-cluster": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "EKSCluster", + "metadata": { + "name": "test-cluster", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "region": "us-west-2", + "kubernetesVersion": "1.36", + "nodePools": [ { "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "role": "GPU", + "instanceType": "g6.xlarge", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": { + "acceleratorType": "nvidia-l4", + }, + "zones": ["us-west-2a", "us-west-2b"], }, ], }, }, ), ), - resources={ - "eks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-west-2", - "kubernetesVersion": "1.36", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - }, - "zones": ["us-west-2a"], - "capacityBlock": { - "capacityReservationId": "cr-0123456789abcdef0", - }, - }, - ], - }, - }, - ), - ), - }, + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want8.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) - - # --- Case 9: EKS first pass with a node pool that opts into the EFA - # fabric. fabric.type flows through to the EKSCluster node pool, which - # compose-eks-cluster turns into EFA launch-template interfaces. --- - req9 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want4.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) + + # --- Case 8: EKS first pass with a node pool backed by a Capacity + # Block. The reservation ID flows through to the EKSCluster node pool's + # capacityBlock, which compose-eks-cluster turns into a CAPACITY_BLOCK + # node group. --- + req8 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="EKS", + eks=v1alpha1.Eks(region="us-west-2"), ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="EKS", - eks=v1alpha1.Eks(region="us-west-2"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4-eks", - nodeCount=2, - maxNodeCount=4, - zones=["us-west-2a"], - fabric=v1alpha1.Fabric(type="EFA"), + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a"], + capacityBlock=v1alpha1.CapacityBlock( + capacityReservationId="cr-0123456789abcdef0", ), - ], - ), - ).model_dump(exclude_none=True, mode="json"), - ), + ), + ], + ), + ).model_dump(exclude_none=True, mode="json"), ), ), - ) - req9.required_resources["class-gpu-l4-eks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), - ) + ), + ) + req8.required_resources["class-gpu-l4-eks"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), + ) - want9 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + want8 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + }, + }, + ), + ), + resources={ + "eks-cluster": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "EKSCluster", + "metadata": { + "name": "test-cluster", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "region": "us-west-2", + "kubernetesVersion": "1.36", + "nodePools": [ { "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "role": "GPU", + "instanceType": "g6.xlarge", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": { + "acceleratorType": "nvidia-l4", + }, + "zones": ["us-west-2a"], + "capacityBlock": { + "capacityReservationId": "cr-0123456789abcdef0", + }, }, ], }, }, ), ), - resources={ - "eks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-west-2", - "kubernetesVersion": "1.36", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - }, - "zones": ["us-west-2a"], - "fabric": "EFA", - }, - ], - }, - }, - ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want9.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) - - # --- Case 5: EKS cluster not yet ready (no kubeconfig observed) but a - # ClusterProviderConfig already exists from a prior reconcile. The CPC - # is built only from the kubeconfig, so without one it's simply omitted - # from desired state this reconcile (and recreated once the kubeconfig - # is observed again) - it is never emitted with an empty secretRef. - observed_cpc = { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, - }, - } - req5 = fnv1.RunFunctionRequest() - req5.CopyFrom(req4) - req5.observed.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource(resource=resource.dict_to_struct(observed_cpc)), - ) - - # Desired state is identical to case 4: no ClusterProviderConfig. - want5 = fnv1.RunFunctionResponse() - want5.CopyFrom(want4) - - # --- Case 6: GKE cluster ready - composes CPC, backend, usage, and the - # VPC-pinned modelplane-rwx Filestore StorageClass on the workload - # cluster (default cache storage class). --- - observed_gke_ready = { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "region": "us-central1", - "nodePools": [{"name": "system", "role": "System", "machineType": "e2-standard-4"}], }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2026-06-08T00:00:00Z", - }, - ], - # The backing GKECluster reports its effective RWX StorageClass; - # the InferenceCluster relays it up to its own status.cache. - "cache": {"storageClassName": "modelplane-rwx"}, - "secrets": [ - {"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}, - { - "type": "GoogleApplicationCredentials", - "name": "test-cluster-sa-key-fghij", - "key": "credentials.json", - }, - ], - }, - } - req6 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want8.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) + + # --- Case 9: EKS first pass with a node pool that opts into the EFA + # fabric. fabric.type flows through to the EKSCluster node pool, which + # compose-eks-cluster turns into EFA launch-template interfaces. --- + req9 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="EKS", + eks=v1alpha1.Eks(region="us-west-2"), ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="GKE", - gke=v1alpha1.Gke( - region="us-central1", - ), + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a"], + fabric=v1alpha1.Fabric(type="EFA"), ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - zones=["us-central1-a"], - ), - ], - ), - ).model_dump(exclude_none=True, mode="json") - ), + ], + ), + ).model_dump(exclude_none=True, mode="json"), ), - resources={ - "gke-cluster": fnv1.Resource(resource=resource.dict_to_struct(observed_gke_ready)), - }, ), - ) - req6.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) + ), + ) + req9.required_resources["class-gpu-l4-eks"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), + ) - want6 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + want9 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + }, + }, + ), + ), + resources={ + "eks-cluster": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "EKSCluster", + "metadata": { + "name": "test-cluster", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "region": "us-west-2", + "kubernetesVersion": "1.36", + "nodePools": [ { "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - } - ], + "role": "GPU", + "instanceType": "g6.xlarge", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": { + "acceleratorType": "nvidia-l4", + }, + "zones": ["us-west-2a"], + "fabric": "EFA", }, ], - # Relayed from the backing GKECluster's status.cache. - "cache": {"storageClassName": "modelplane-rwx"}, }, - } + }, ), ), - resources={ - "gke-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-central1", - "kubernetesVersion": "1.35", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "machineType": "g2-standard-48", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - "acceleratorCount": 1, - }, - "zones": ["us-central1-a"], - }, - ], - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "cluster-provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, - "identity": { - "type": "GoogleApplicationCredentials", - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-sa-key-fghij", - "key": "credentials.json", - }, - }, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "GKE", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - { - "type": "GoogleApplicationCredentials", - "name": "test-cluster-sa-key-fghij", - "key": "credentials.json", - }, - ], - }, - } - ), - ), - "usage-gke-by-backend": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - }, + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want9.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) + + # --- Case 5: EKS cluster not yet ready (no kubeconfig observed) but a + # ClusterProviderConfig already exists from a prior reconcile. The CPC + # is built only from the kubeconfig, so without one it's simply omitted + # from desired state this reconcile (and recreated once the kubeconfig + # is observed again) - it is never emitted with an empty secretRef. + observed_cpc = { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + }, + }, + } + req5 = fnv1.RunFunctionRequest() + req5.CopyFrom(req4) + req5.observed.resources["cluster-provider-config-kubernetes"].CopyFrom( + fnv1.Resource(resource=resource.dict_to_struct(observed_cpc)), + ) + + # Desired state is identical to case 4: no ClusterProviderConfig. + want5 = fnv1.RunFunctionResponse() + want5.CopyFrom(want4) + + # --- Case 6: GKE cluster ready - composes CPC, backend, usage, and the + # VPC-pinned modelplane-rwx Filestore StorageClass on the workload + # cluster (default cache storage class). --- + observed_gke_ready = { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "us-central1", + "nodePools": [{"name": "system", "role": "System", "machineType": "e2-standard-4"}], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + }, ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="GKE cluster ready, composing backend", - ), + # The backing GKECluster reports its effective RWX StorageClass; + # the InferenceCluster relays it up to its own status.cache. + "cache": {"storageClassName": "modelplane-rwx"}, + "secrets": [ + {"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}, + { + "type": "GoogleApplicationCredentials", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", + }, ], - context=structpb.Struct(), - ) - want6.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - - # --- Case 7: EKS cluster ready - kubeconfig observed on the EKSCluster - # status. The function wires the ClusterProviderConfig, composes the - # ServingStack backend, and emits the Usage that blocks EKSCluster - # deletion until the ServingStack is gone. --- - req7 = fnv1.RunFunctionRequest() - req7.CopyFrom(req4) - req7.observed.resources["eks-cluster"].CopyFrom( - fnv1.Resource( + }, + } + req6 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "region": "us-west-2", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - }, - ], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="GKE", + gke=v1alpha1.Gke( + region="us-central1", + ), + ), + nodePools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], + ), ], - # The backing EKSCluster reports its effective RWX - # StorageClass; the InferenceCluster relays it up. - "cache": {"storageClassName": "modelplane-rwx-efs"}, - }, - } + ), + ).model_dump(exclude_none=True, mode="json") ), ), - ) + resources={ + "gke-cluster": fnv1.Resource(resource=resource.dict_to_struct(observed_gke_ready)), + }, + ), + ) + req6.required_resources["class-gpu-l4"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) + ) - want7 = fnv1.RunFunctionResponse() - want7.CopyFrom(want4) - # Mark the EKSCluster ready and relay its status.cache up to status.cache. - _eks_ready_extras(want7, "modelplane-rwx-efs") - want7.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( + want6 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", }, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - ) - want7.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", "namespace": "modelplane-system", - }, - "spec": { - "cloud": "EKS", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ + "gpuPools": [ { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + } + ], }, ], + # Relayed from the backing GKECluster's status.cache. + "cache": {"storageClassName": "modelplane-rwx"}, }, } ), ), - ) - want7.desired.resources["usage-eks-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - ) - del want7.conditions[:] - want7.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want7.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="EKS cluster ready, composing backend", - ) - ) - - # --- Case 10: Nebius first pass composes the NebiusCluster XR only. - # The pool's InfiniBand fabric flows through to the NebiusCluster - # pool's fabric, and minNodeCount stays unset so the pool's - # autoscaling floor defaults to its node count downstream. --- - inference_class_h100_nebius = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-h100-nebius"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, - }, - ], - "provisioning": { - "provider": "Nebius", - "nebius": { - "platform": "gpu-h100-sxm", - "preset": "8gpu-128vcpu-1600gb", - "diskSizeGb": 200, - "accelerator": {"type": "nvidia-h100", "count": 8}, - }, - }, - }, - } - class_selector_nebius = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-h100-nebius", - ) - - req10 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Nebius", - nebius=v1alpha1.Nebius(), - ), - nodePools=[ - v1alpha1.NodePool( - name="h100-pool", - className="gpu-h100-nebius", - nodeCount=2, - maxNodeCount=4, - fabric=v1alpha1.Fabric( - type="InfiniBand", - infiniband=v1alpha1.Infiniband(fabric="fabric-2"), - ), - ), - ], - ), - ).model_dump(exclude_none=True, mode="json"), - ), - ), - ), - ) - req10.required_resources["class-gpu-h100-nebius"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_nebius)), - ) - - want10 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + resources={ + "gke-cluster": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "metadata": { + "name": "test-cluster", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "region": "us-central1", + "kubernetesVersion": "1.35", + "nodePools": [ { - "name": "h100-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, - }, - ], + "name": "l4-pool", + "role": "GPU", + "machineType": "g2-standard-48", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": { + "acceleratorType": "nvidia-l4", + "acceleratorCount": 1, + }, + "zones": ["us-central1-a"], }, ], }, - }, + } ), + ready=fnv1.READY_TRUE, ), - resources={ - "nebius-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "NebiusCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", + "cluster-provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, }, - "spec": { - "kubernetesVersion": "1.34", - "nodePools": [ - { - "name": "h100-pool", - "role": "GPU", - "platform": "gpu-h100-sxm", - "preset": "8gpu-128vcpu-1600gb", - "diskSizeGb": 200, - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-h100", - "driversPreset": "cuda13.0", - }, - "fabric": { - "type": "InfiniBand", - "infiniband": {"fabric": "fabric-2"}, - }, - }, - ], + "identity": { + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", + }, }, }, - ), + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", + ready=fnv1.READY_TRUE, ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want10.requirements.resources["class-gpu-h100-nebius"].CopyFrom(class_selector_nebius) - - # --- Case 11: Nebius cluster ready - kubeconfig and service account - # credentials observed on the NebiusCluster status. The function wires - # the ClusterProviderConfig with the Nebius identity (the mk8s - # kubeconfig has no embedded credentials), composes the ServingStack - # backend with both secrets, and emits the Usage that blocks - # NebiusCluster deletion until the ServingStack is gone. --- - req11 = fnv1.RunFunctionRequest() - req11.CopyFrom(req10) - req11.observed.resources["nebius-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "NebiusCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "nodePools": [ - { - "name": "h100-pool", - "role": "GPU", - "platform": "gpu-h100-sxm", - "preset": "8gpu-128vcpu-1600gb", - "nodeCount": 2, - }, - ], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - # The credential entry carries a namespace: it - # is the Nebius ClusterProviderConfig's Secret, - # which lives outside modelplane-system. - { - "type": "NebiusServiceAccountCredentials", - "name": "nebius-credentials", - "key": "credentials.json", - "namespace": "crossplane-system", - }, - ], - }, - } + "serving-stack": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", + "namespace": "modelplane-system", + }, + "spec": { + "cloud": "GKE", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + { + "type": "GoogleApplicationCredentials", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", + }, + ], + }, + } + ), + ), + "usage-gke-by-backend": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "of": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "resourceSelector": {"matchControllerRef": True}, + }, + "by": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "resourceSelector": {"matchControllerRef": True}, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, ), + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", ), - ) + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Installing", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="GKE cluster ready, composing backend", + ), + ], + context=structpb.Struct(), + ) + want6.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) + + # --- Case 7: EKS cluster ready - kubeconfig observed on the EKSCluster + # status. The function wires the ClusterProviderConfig, composes the + # ServingStack backend, and emits the Usage that blocks EKSCluster + # deletion until the ServingStack is gone. --- + req7 = fnv1.RunFunctionRequest() + req7.CopyFrom(req4) + req7.observed.resources["eks-cluster"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "EKSCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "us-west-2", + "nodePools": [ + { + "name": "l4-pool", + "role": "GPU", + "instanceType": "g6.xlarge", + "nodeCount": 2, + }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + # The backing EKSCluster reports its effective RWX + # StorageClass; the InferenceCluster relays it up. + "cache": {"storageClassName": "modelplane-rwx-efs"}, + }, + } + ), + ), + ) - want11 = fnv1.RunFunctionResponse() - want11.CopyFrom(want10) - want11.desired.resources["nebius-cluster"].ready = fnv1.READY_TRUE - want11.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, + want7 = fnv1.RunFunctionResponse() + want7.CopyFrom(want4) + # Mark the EKSCluster ready and relay its status.cache up to status.cache. + _eks_ready_extras(want7, "modelplane-rwx-efs") + want7.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", }, - "identity": { - "type": "NebiusServiceAccountCredentials", - "source": "Secret", - "secretRef": { - "namespace": "crossplane-system", - "name": "nebius-credentials", - "key": "credentials.json", - }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + ) + want7.desired.resources["serving-stack"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", + "namespace": "modelplane-system", + }, + "spec": { + "cloud": "EKS", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", }, + ], + }, + } + ), + ), + ) + want7.desired.resources["usage-eks-by-backend"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "of": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "EKSCluster", + "resourceSelector": {"matchControllerRef": True}, }, - } - ), - ready=fnv1.READY_TRUE, + "by": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "resourceSelector": {"matchControllerRef": True}, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + ) + del want7.conditions[:] + want7.conditions.extend( + [ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Installing", + ), + ] + ) + want7.results.append( + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="EKS cluster ready, composing backend", ) - want11.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( + ) + + # --- Case 10: Nebius first pass composes the NebiusCluster XR only. + # The pool's InfiniBand fabric flows through to the NebiusCluster + # pool's fabric, and minNodeCount stays unset so the pool's + # autoscaling floor defaults to its node count downstream. --- + inference_class_h100_nebius = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceClass", + "metadata": {"name": "gpu-h100-nebius"}, + "spec": { + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], + "provisioning": { + "provider": "Nebius", + "nebius": { + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "diskSizeGb": 200, + "accelerator": {"type": "nvidia-h100", "count": 8}, + }, + }, + }, + } + class_selector_nebius = fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceClass", + match_name="gpu-h100-nebius", + ) + + req10 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Nebius", + nebius=v1alpha1.Nebius(), + ), + nodePools=[ + v1alpha1.NodePool( + name="h100-pool", + className="gpu-h100-nebius", + nodeCount=2, + maxNodeCount=4, + fabric=v1alpha1.Fabric( + type="InfiniBand", + infiniband=v1alpha1.Infiniband(fabric="fabric-2"), + ), + ), + ], + ), + ).model_dump(exclude_none=True, mode="json"), + ), + ), + ), + ) + req10.required_resources["class-gpu-h100-nebius"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_nebius)), + ) + + want10 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Nebius", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, + "gpuPools": [ { - "type": "NebiusServiceAccountCredentials", - "name": "nebius-credentials", - "key": "credentials.json", - "namespace": "crossplane-system", + "name": "h100-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], }, ], }, - } + }, ), ), - ) - want11.desired.resources["usage-nebius-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "NebiusCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, + resources={ + "nebius-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "NebiusCluster", + "metadata": { + "name": "test-cluster", + "namespace": "modelplane-system", + }, + "spec": { + "kubernetesVersion": "1.34", + "nodePools": [ + { + "name": "h100-pool", + "role": "GPU", + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "diskSizeGb": 200, + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": { + "acceleratorType": "nvidia-h100", + "driversPreset": "cuda13.0", + }, + "fabric": { + "type": "InfiniBand", + "infiniband": {"fabric": "fabric-2"}, + }, + }, + ], + }, }, - } + ), ), - ready=fnv1.READY_TRUE, + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want10.requirements.resources["class-gpu-h100-nebius"].CopyFrom(class_selector_nebius) + + # --- Case 11: Nebius cluster ready - kubeconfig and service account + # credentials observed on the NebiusCluster status. The function wires + # the ClusterProviderConfig with the Nebius identity (the mk8s + # kubeconfig has no embedded credentials), composes the ServingStack + # backend with both secrets, and emits the Usage that blocks + # NebiusCluster deletion until the ServingStack is gone. --- + req11 = fnv1.RunFunctionRequest() + req11.CopyFrom(req10) + req11.observed.resources["nebius-cluster"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "NebiusCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "nodePools": [ + { + "name": "h100-pool", + "role": "GPU", + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "nodeCount": 2, + }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + # The credential entry carries a namespace: it + # is the Nebius ClusterProviderConfig's Secret, + # which lives outside modelplane-system. + { + "type": "NebiusServiceAccountCredentials", + "name": "nebius-credentials", + "key": "credentials.json", + "namespace": "crossplane-system", + }, + ], + }, + } + ), + ), + ) + + want11 = fnv1.RunFunctionResponse() + want11.CopyFrom(want10) + want11.desired.resources["nebius-cluster"].ready = fnv1.READY_TRUE + want11.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + }, + "identity": { + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": { + "namespace": "crossplane-system", + "name": "nebius-credentials", + "key": "credentials.json", + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + ) + want11.desired.resources["serving-stack"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", + "namespace": "modelplane-system", + }, + "spec": { + "cloud": "Nebius", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + { + "type": "NebiusServiceAccountCredentials", + "name": "nebius-credentials", + "key": "credentials.json", + "namespace": "crossplane-system", + }, + ], + }, + } + ), + ), + ) + want11.desired.resources["usage-nebius-by-backend"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "of": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "NebiusCluster", + "resourceSelector": {"matchControllerRef": True}, + }, + "by": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "resourceSelector": {"matchControllerRef": True}, + }, + "replayDeletion": True, + }, + } ), + ready=fnv1.READY_TRUE, + ), + ) + del want11.conditions[:] + want11.conditions.extend( + [ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Installing", + ), + ] + ) + want11.results.append( + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Nebius cluster ready, composing backend", ) - del want11.conditions[:] - want11.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", + ) + + # --- Case 12: AKS first pass composes the AKSCluster XR only. The + # pool's InfiniBand fabric flows through to the AKSCluster pool as the + # plain fabric string - Azure has no user-selectable fabric ID. --- + inference_class_h100_aks = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceClass", + "metadata": {"name": "gpu-h100-aks"}, + "spec": { + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], + "provisioning": { + "provider": "AKS", + "aks": { + "vmSize": "Standard_ND96isr_H100_v5", + "diskSizeGb": 200, + "accelerator": {"type": "nvidia-h100", "count": 8}, + }, + }, + }, + } + class_selector_aks = fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceClass", + match_name="gpu-h100-aks", + ) + + req12 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="AKS", + aks=v1alpha1.Aks(location="westeurope"), + ), + nodePools=[ + v1alpha1.NodePool( + name="h100pool", + className="gpu-h100-aks", + nodeCount=2, + # AKS GPU pools keep minNodeCount at + # 1: the AKS autoscaler can't scale a + # DRA pool up from zero nodes. + minNodeCount=1, + maxNodeCount=4, + fabric=v1alpha1.Fabric(type="InfiniBand"), + ), + ], + ), + ).model_dump(exclude_none=True, mode="json"), ), - ] - ) - want11.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Nebius cluster ready, composing backend", - ) - ) + ), + ), + ) + req12.required_resources["class-gpu-h100-aks"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_aks)), + ) - # --- Case 12: AKS first pass composes the AKSCluster XR only. The - # pool's InfiniBand fabric flows through to the AKSCluster pool as the - # plain fabric string - Azure has no user-selectable fabric ID. --- - inference_class_h100_aks = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-h100-aks"}, - "spec": { - "devices": [ + want12 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, - }, - ], - "provisioning": { - "provider": "AKS", - "aks": { - "vmSize": "Standard_ND96isr_H100_v5", - "diskSizeGb": 200, - "accelerator": {"type": "nvidia-h100", "count": 8}, + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "h100pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], + }, + ], + }, }, - }, - }, - } - class_selector_aks = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-h100-aks", - ) - - req12 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="AKS", - aks=v1alpha1.Aks(location="westeurope"), - ), - nodePools=[ - v1alpha1.NodePool( - name="h100pool", - className="gpu-h100-aks", - nodeCount=2, - # AKS GPU pools keep minNodeCount at - # 1: the AKS autoscaler can't scale a - # DRA pool up from zero nodes. - minNodeCount=1, - maxNodeCount=4, - fabric=v1alpha1.Fabric(type="InfiniBand"), - ), - ], - ), - ).model_dump(exclude_none=True, mode="json"), - ), ), ), - ) - req12.required_resources["class-gpu-h100-aks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_aks)), - ) - - want12 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + resources={ + "aks-cluster": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "AKSCluster", + "metadata": { + "name": "test-cluster", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "location": "westeurope", + "kubernetesVersion": "1.34", + "nodePools": [ { "name": "h100pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, - }, - ], + "role": "GPU", + "vmSize": "Standard_ND96isr_H100_v5", + "diskSizeGb": 200, + "nodeCount": 2, + "minNodeCount": 1, + "maxNodeCount": 4, + "gpu": { + "acceleratorType": "nvidia-h100", + }, + "fabric": "InfiniBand", }, ], }, }, ), ), - resources={ - "aks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want12.requirements.resources["class-gpu-h100-aks"].CopyFrom(class_selector_aks) + + # --- Case 13: AKS cluster ready - kubeconfig observed on the + # AKSCluster status. The kubeconfig embeds a client certificate, so + # the ClusterProviderConfig carries no identity (unlike GKE/Nebius). + # The function composes the ServingStack backend and the Usage that + # blocks AKSCluster deletion until the ServingStack is gone, and + # relays the AKSCluster's status.cache up to status.cache. --- + req13 = fnv1.RunFunctionRequest() + req13.CopyFrom(req12) + req13.observed.resources["aks-cluster"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "AKSCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "location": "westeurope", + "nodePools": [ { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "AKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "location": "westeurope", - "kubernetesVersion": "1.34", - "nodePools": [ - { - "name": "h100pool", - "role": "GPU", - "vmSize": "Standard_ND96isr_H100_v5", - "diskSizeGb": 200, - "nodeCount": 2, - "minNodeCount": 1, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-h100", - }, - "fabric": "InfiniBand", - }, - ], - }, + "name": "h100pool", + "role": "GPU", + "vmSize": "Standard_ND96isr_H100_v5", + "nodeCount": 2, }, - ), - ), - }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + # The backing AKSCluster reports its effective RWX + # StorageClass; the InferenceCluster relays it up. + "cache": {"storageClassName": "modelplane-rwx-fs"}, + }, + } ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want12.requirements.resources["class-gpu-h100-aks"].CopyFrom(class_selector_aks) - - # --- Case 13: AKS cluster ready - kubeconfig observed on the - # AKSCluster status. The kubeconfig embeds a client certificate, so - # the ClusterProviderConfig carries no identity (unlike GKE/Nebius). - # The function composes the ServingStack backend and the Usage that - # blocks AKSCluster deletion until the ServingStack is gone, and - # relays the AKSCluster's status.cache up to status.cache. --- - req13 = fnv1.RunFunctionRequest() - req13.CopyFrom(req12) - req13.observed.resources["aks-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "AKSCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "location": "westeurope", - "nodePools": [ - { - "name": "h100pool", - "role": "GPU", - "vmSize": "Standard_ND96isr_H100_v5", - "nodeCount": 2, - }, - ], + ), + ) + + want13 = fnv1.RunFunctionResponse() + want13.CopyFrom(want12) + want13.desired.resources["aks-cluster"].ready = fnv1.READY_TRUE + status13 = want13.desired.composite.resource.fields["status"].struct_value + status13.fields["cache"].struct_value.fields["storageClassName"].string_value = "modelplane-rwx-fs" + want13.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - ], - # The backing AKSCluster reports its effective RWX - # StorageClass; the InferenceCluster relays it up. - "cache": {"storageClassName": "modelplane-rwx-fs"}, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + ) + want13.desired.resources["serving-stack"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", + "namespace": "modelplane-system", + }, + "spec": { + "cloud": "AKS", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + }, + } + ), + ), + ) + want13.desired.resources["usage-aks-by-backend"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "of": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "AKSCluster", + "resourceSelector": {"matchControllerRef": True}, }, - } - ), + "by": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "resourceSelector": {"matchControllerRef": True}, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + ) + del want13.conditions[:] + want13.conditions.extend( + [ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Installing", + ), + ] + ) + want13.results.append( + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="AKS cluster ready, composing backend", ) + ) - want13 = fnv1.RunFunctionResponse() - want13.CopyFrom(want12) - want13.desired.resources["aks-cluster"].ready = fnv1.READY_TRUE - status13 = want13.desired.composite.resource.fields["status"].struct_value - status13.fields["cache"].struct_value.fields["storageClassName"].string_value = "modelplane-rwx-fs" - want13.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( + # --- Case 14: Vultr first pass composes the VultrCluster XR only. + # minNodeCount stays unset so the pool's autoscaling floor defaults + # to its node count downstream. --- + inference_class_l40s_vultr = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceClass", + "metadata": {"name": "gpu-l40s-vultr"}, + "spec": { + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, + }, + ], + "provisioning": { + "provider": "Vultr", + "vultr": { + "plan": "vcg-l40s-16c-180g-48vram", + "accelerator": {"type": "nvidia-l40s", "count": 1}, + }, + }, + }, + } + class_selector_vultr = fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceClass", + match_name="gpu-l40s-vultr", + ) + + req14 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, - }, - } + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Vultr", + vultr=v1alpha1.Vultr(region="ewr"), + ), + nodePools=[ + v1alpha1.NodePool( + name="l40s-pool", + className="gpu-l40s-vultr", + nodeCount=2, + maxNodeCount=4, + ), + ], + ), + ).model_dump(exclude_none=True, mode="json"), ), - ready=fnv1.READY_TRUE, ), - ) - want13.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( + ), + ) + req14.required_resources["class-gpu-l40s-vultr"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l40s_vultr)), + ) + + want14 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, "namespace": "modelplane-system", - }, - "spec": { - "cloud": "AKS", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ + "gpuPools": [ { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + "name": "l40s-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, + }, + ], }, ], }, - } + }, ), ), - ) - want13.desired.resources["usage-aks-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "AKSCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, + resources={ + "vultr-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "VultrCluster", + "metadata": { + "name": "test-cluster", + "namespace": "modelplane-system", + }, + "spec": { + "region": "ewr", + "kubernetesVersion": "v1.36.2+1", + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "plan": "vcg-l40s-16c-180g-48vram", + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": { + "acceleratorType": "nvidia-l40s", + }, + }, + ], + }, }, - } - ), - ready=fnv1.READY_TRUE, - ), - ) - del want13.conditions[:] - want13.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", + ), ), - ] - ) - want13.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="AKS cluster ready, composing backend", - ) - ) - - # --- Case 14: Vultr first pass composes the VultrCluster XR only. - # minNodeCount stays unset so the pool's autoscaling floor defaults - # to its node count downstream. --- - inference_class_l40s_vultr = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-l40s-vultr"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, - }, - ], - "provisioning": { - "provider": "Vultr", - "vultr": { - "plan": "vcg-l40s-16c-180g-48vram", - "accelerator": {"type": "nvidia-l40s", "count": 1}, - }, - }, }, - } - class_selector_vultr = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-l40s-vultr", - ) + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + ], + context=structpb.Struct(), + ) + want14.requirements.resources["class-gpu-l40s-vultr"].CopyFrom(class_selector_vultr) - req14 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Vultr", - vultr=v1alpha1.Vultr(region="ewr"), + # --- Case 14b: Vultr credentials pass through to the VultrCluster + # spec, mirroring the GKE/EKS/AKS passthrough. --- + req_creds_vultr = fnv1.RunFunctionRequest() + req_creds_vultr.CopyFrom(req14) + req_creds_vultr.observed.composite.CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta( + name="test-cluster", + namespace="modelplane-system", + ), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Vultr", + vultr=v1alpha1.Vultr( + region="ewr", + credentials=v1alpha1.Credentials( + type="ProviderConfig", + name="my-vultr-account", ), - nodePools=[ - v1alpha1.NodePool( - name="l40s-pool", - className="gpu-l40s-vultr", - nodeCount=2, - maxNodeCount=4, - ), - ], ), - ).model_dump(exclude_none=True, mode="json"), + ), + nodePools=[ + v1alpha1.NodePool( + name="l40s-pool", + className="gpu-l40s-vultr", + nodeCount=2, + maxNodeCount=4, + ), + ], ), - ), + ).model_dump(exclude_none=True, mode="json"), ), - ) - req14.required_resources["class-gpu-l40s-vultr"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l40s_vultr)), - ) + ), + ) - want14 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + want_creds_vultr = fnv1.RunFunctionResponse() + want_creds_vultr.CopyFrom(want14) + want_creds_vultr.desired.resources["vultr-cluster"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "VultrCluster", + "metadata": { + "name": "test-cluster", + "namespace": "modelplane-system", + }, + "spec": { + "region": "ewr", + "kubernetesVersion": "v1.36.2+1", + "credentials": { + "type": "ProviderConfig", + "name": "my-vultr-account", + }, + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "plan": "vcg-l40s-16c-180g-48vram", + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": { + "acceleratorType": "nvidia-l40s", }, - "namespace": "modelplane-system", - "gpuPools": [ + }, + ], + }, + }, + ), + ), + ) + + # --- Case 15: Vultr cluster ready - kubeconfig observed on the + # VultrCluster status. The VKE kubeconfig embeds static client + # certificates, so the ClusterProviderConfig carries no identity + # (unlike Nebius). The function composes the ServingStack backend + # with the kubeconfig and emits the Usage that blocks VultrCluster + # deletion until the ServingStack is gone. VultrCluster reports no + # cache StorageClass, so status.cache stays unset. --- + req15 = fnv1.RunFunctionRequest() + req15.CopyFrom(req14) + req15.observed.resources["vultr-cluster"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "VultrCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "ewr", + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "plan": "vcg-l40s-16c-180g-48vram", + "nodeCount": 2, + }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + }, + } + ), + ), + ) + + want15 = fnv1.RunFunctionResponse() + want15.CopyFrom(want14) + want15.desired.resources["vultr-cluster"].ready = fnv1.READY_TRUE + want15.desired.composite.CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "providerConfigRef": { + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "namespace": "modelplane-system", + "gpuPools": [ + { + "name": "l40s-pool", + "nodes": 4, + "devices": [ { - "name": "l40s-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, }, ], }, + ], + }, + }, + ), + ), + ) + want15.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, }, - ), - ), - resources={ - "vultr-cluster": fnv1.Resource( - resource=resource.dict_to_struct( + }, + } + ), + ready=fnv1.READY_TRUE, + ), + ) + want15.desired.resources["serving-stack"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": { + "name": "test-cluster-serving-stack-fd00b", + "namespace": "modelplane-system", + }, + "spec": { + "cloud": "Vultr", + "gateway": {"hostname": _GATEWAY_HOSTNAME}, + "stack": "Standard", + "secrets": [ { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "ewr", - "kubernetesVersion": "v1.36.2+1", - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "plan": "vcg-l40s-16c-180g-48vram", - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-l40s", - }, - }, - ], - }, + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", }, - ), - ), - }, + ], + }, + } ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), + ), + ) + want15.desired.resources["usage-vultr-by-backend"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "of": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "VultrCluster", + "resourceSelector": {"matchControllerRef": True}, + }, + "by": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "resourceSelector": {"matchControllerRef": True}, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + ) + del want15.conditions[:] + want15.conditions.extend( + [ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ClusterRunning", + ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Installing", + ), + ] + ) + want15.results.append( + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Vultr cluster ready, composing backend", ) - want14.requirements.resources["class-gpu-l40s-vultr"].CopyFrom(class_selector_vultr) + ) - # --- Case 14b: Vultr credentials pass through to the VultrCluster - # spec, mirroring the GKE/EKS/AKS passthrough. --- - req_creds_vultr = fnv1.RunFunctionRequest() - req_creds_vultr.CopyFrom(req14) - req_creds_vultr.observed.composite.CopyFrom( - fnv1.Resource( + # Every compose path emits the ModelReplica guard requirement. + for want in ( + want1, + want2, + want3, + want4, + want5, + want6, + want7, + want8, + want9, + want10, + want11, + want12, + want13, + want14, + want_creds_vultr, + want15, + ): + want.requirements.resources["gateways"].CopyFrom(_gateways_selector()) + want.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) + want.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) + want.requirements.resources["model-caches"].CopyFrom(_caches_selector()) + + # The guard cases reuse case 1's request and response. + guard_cases = [ + Case( + "ModelReplicas, ModelRoutes and ModelCaches compose the guard and the mirrored namespaces", + *_guard_case(req1, want1), + ), + Case("a ModelRoute on the cluster composes the guard", *_route_guard_case(req1, want1)), + Case("a ModelCache staging onto the cluster composes the guard", *_cache_guard_case(req1, want1)), + Case("an InferenceGateway on the cluster composes the guard", *_gateway_guard_case(req1, want1)), + Case("nothing on the cluster leaves it deletable", *_unused_case(req1, want1)), + Case("guard is composed even when compose returns early", *_early_return_guard_case()), + ] + + # --- Case credentials: GKE with custom credentials passes them through to GKECluster. --- + req_creds = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( v1alpha1.InferenceCluster( metadata=metav1.ObjectMeta( @@ -2536,120 +2811,38 @@ async def test_compose(self) -> None: # noqa: PLR0915 ), spec=v1alpha1.Spec( cluster=v1alpha1.Cluster( - source="Vultr", - vultr=v1alpha1.Vultr( - region="ewr", + source="GKE", + gke=v1alpha1.Gke( + region="us-central1", credentials=v1alpha1.Credentials( type="ProviderConfig", - name="my-vultr-account", + name="my-gcp-account", ), ), ), nodePools=[ v1alpha1.NodePool( - name="l40s-pool", - className="gpu-l40s-vultr", + name="l4-pool", + className="gpu-l4", nodeCount=2, maxNodeCount=4, + zones=["us-central1-a"], ), ], ), - ).model_dump(exclude_none=True, mode="json"), - ), - ), - ) - - want_creds_vultr = fnv1.RunFunctionResponse() - want_creds_vultr.CopyFrom(want14) - want_creds_vultr.desired.resources["vultr-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "ewr", - "kubernetesVersion": "v1.36.2+1", - "credentials": { - "type": "ProviderConfig", - "name": "my-vultr-account", - }, - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "plan": "vcg-l40s-16c-180g-48vram", - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-l40s", - }, - }, - ], - }, - }, - ), - ), - ) - - # --- Case 15: Vultr cluster ready - kubeconfig observed on the - # VultrCluster status. The VKE kubeconfig embeds static client - # certificates, so the ClusterProviderConfig carries no identity - # (unlike Nebius). The function composes the ServingStack backend - # with the kubeconfig and emits the Usage that blocks VultrCluster - # deletion until the ServingStack is gone. VultrCluster reports no - # cache StorageClass, so status.cache stays unset. --- - req15 = fnv1.RunFunctionRequest() - req15.CopyFrom(req14) - req15.observed.resources["vultr-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "region": "ewr", - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "plan": "vcg-l40s-16c-180g-48vram", - "nodeCount": 2, - }, - ], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - ], - }, - } + ).model_dump(exclude_none=True, mode="json") ), ), - ) + ), + ) + req_creds.required_resources["class-gpu-l4"].items.append( + fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) + ) - want15 = fnv1.RunFunctionResponse() - want15.CopyFrom(want14) - want15.desired.resources["vultr-cluster"].ready = fnv1.READY_TRUE - want15.desired.composite.CopyFrom( - fnv1.Resource( + want_creds = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( { "status": { @@ -2659,7 +2852,7 @@ async def test_compose(self) -> None: # noqa: PLR0915 "namespace": "modelplane-system", "gpuPools": [ { - "name": "l40s-pool", + "name": "l4-pool", "nodes": 4, "devices": [ { @@ -2668,535 +2861,332 @@ async def test_compose(self) -> None: # noqa: PLR0915 "driver": "gpu.nvidia.com", "deviceClassName": "gpu.nvidia.com", "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, ], }, - }, - ), - ), - ) - want15.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - ) - want15.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Vultr", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - ], - }, - } - ), - ), - ) - want15.desired.resources["usage-vultr-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, } ), - ready=fnv1.READY_TRUE, - ), - ) - del want15.conditions[:] - want15.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want15.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Vultr cluster ready, composing backend", - ) - ) - - # Every compose path emits the ModelReplica guard requirement. - for want in ( - want1, - want2, - want3, - want4, - want5, - want6, - want7, - want8, - want9, - want10, - want11, - want12, - want13, - want14, - want_creds_vultr, - want15, - ): - want.requirements.resources["gateways"].CopyFrom(_gateways_selector()) - want.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) - want.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) - want.requirements.resources["model-caches"].CopyFrom(_caches_selector()) - - # The guard cases reuse case 1's request and response. - guard_cases = [ - Case( - "ModelReplicas, ModelRoutes and ModelCaches compose the guard and the mirrored namespaces", - *_guard_case(req1, want1), - ), - Case("a ModelRoute on the cluster composes the guard", *_route_guard_case(req1, want1)), - Case("a ModelCache staging onto the cluster composes the guard", *_cache_guard_case(req1, want1)), - Case("an InferenceGateway on the cluster composes the guard", *_gateway_guard_case(req1, want1)), - Case("nothing on the cluster leaves it deletable", *_unused_case(req1, want1)), - Case("guard is composed even when compose returns early", *_early_return_guard_case()), - ] - - # --- Case credentials: GKE with custom credentials passes them through to GKECluster. --- - req_creds = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="GKE", - gke=v1alpha1.Gke( - region="us-central1", - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="my-gcp-account", - ), - ), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - zones=["us-central1-a"], - ), - ], - ), - ).model_dump(exclude_none=True, mode="json") - ), - ), ), - ) - req_creds.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - want_creds = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + resources={ + "gke-cluster": fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "metadata": { + "name": "test-cluster", "namespace": "modelplane-system", - "gpuPools": [ + }, + "spec": { + "region": "us-central1", + "kubernetesVersion": "1.35", + "credentials": { + "type": "ProviderConfig", + "name": "my-gcp-account", + }, + "nodePools": [ { "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "role": "GPU", + "machineType": "g2-standard-48", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": { + "acceleratorType": "nvidia-l4", + "acceleratorCount": 1, + }, + "zones": ["us-central1-a"], }, ], }, } ), ), - resources={ - "gke-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-central1", - "kubernetesVersion": "1.35", - "credentials": { - "type": "ProviderConfig", - "name": "my-gcp-account", - }, - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "machineType": "g2-standard-48", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - "acceleratorCount": 1, - }, - "zones": ["us-central1-a"], - }, - ], - }, - } - ), - ), - }, + }, + ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Provisioning", ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want_creds.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - want_creds.requirements.resources["gateways"].CopyFrom(_gateways_selector()) - want_creds.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) - want_creds.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) - want_creds.requirements.resources["model-caches"].CopyFrom(_caches_selector()) - - # Every cloud cluster composes an activation policy; with the policy - # observed Healthy the cluster XR is composed. - for req, want, kinds in [ - (req2, want2, fn._ACTIVATE_GCP), - (req_creds, want_creds, fn._ACTIVATE_GCP), - (req4, want4, fn._ACTIVATE_AWS), - (req5, want5, fn._ACTIVATE_AWS), - (req6, want6, fn._ACTIVATE_GCP), - (req7, want7, fn._ACTIVATE_AWS), - (req8, want8, fn._ACTIVATE_AWS), - (req9, want9, fn._ACTIVATE_AWS), - (req10, want10, fn._ACTIVATE_NEBIUS), - (req11, want11, fn._ACTIVATE_NEBIUS), - (req12, want12, fn._ACTIVATE_AZURE), - (req13, want13, fn._ACTIVATE_AZURE), - (req14, want14, fn._ACTIVATE_VULTR), - (req_creds_vultr, want_creds_vultr, fn._ACTIVATE_VULTR), - (req15, want15, fn._ACTIVATE_VULTR), - ]: - _observe_activated(req, kinds) - _want_activation(want, kinds) - - # While the policy is missing even one of the kinds from status.activated - # (e.g. a provider still installing), and with no cluster observed, the - # function composes only the activation policy (not marked ready, so the - # composite doesn't report ready), not the cluster XR. Observing the - # policy with a kind missing exercises the all-kinds check rather than - # the policy-absent branch. - req_unactivated = copy.deepcopy(req2) - _observe_activated(req_unactivated, fn._ACTIVATE_GCP[:-1]) - want_unactivated = copy.deepcopy(want2) - del want_unactivated.desired.resources["gke-cluster"] - want_unactivated.desired.resources["activation"].ClearField("ready") - - # Once the cluster is observed, the function keeps composing it even - # when the policy momentarily stops reporting the kinds active, so an - # activation blip never drops a provisioned cluster from desired state. - req_blip = copy.deepcopy(req6) - del req_blip.observed.resources["activation"] - want_blip = copy.deepcopy(want6) - - cases = [ - Case(name="existing cluster with secrets composes backend and CPC", req=req1, want=want1), - Case(name="existing cluster with a non-GCP identity threads the identity type", req=req1b, want=want1b), - Case(name="GKE cluster first pass composes GKECluster XR only", req=req2, want=want2), - Case(name="GKE credentials pass through to GKECluster spec", req=req_creds, want=want_creds), - Case(name="existing cluster second pass with backend ready", req=req3, want=want3), - Case(name="EKS cluster first pass composes EKSCluster XR only", req=req4, want=want4), - Case(name="EKS cluster not ready re-emits existing CPC unchanged", req=req5, want=want5), - Case(name="GKE cluster ready composes CPC, backend, usage, and RWX StorageClass", req=req6, want=want6), - Case(name="EKS cluster ready composes ServingStack and Usage", req=req7, want=want7), - Case( - name="EKS node pool with a Capacity Block sets capacityBlock on the EKSCluster pool", - req=req8, - want=want8, - ), - Case( - name="EKS node pool with fabric EFA sets fabric on the EKSCluster pool", - req=req9, - want=want9, - ), - Case(name="Nebius cluster first pass composes NebiusCluster XR only", req=req10, want=want10), - Case( - name="Nebius cluster ready composes CPC with Nebius identity, ServingStack, and Usage", - req=req11, - want=want11, - ), - Case(name="AKS cluster first pass composes AKSCluster XR only", req=req12, want=want12), - Case( - name="AKS cluster ready composes CPC without identity, ServingStack, and Usage", - req=req13, - want=want13, - ), - Case( - name="cloud cluster not activated composes only the policy", req=req_unactivated, want=want_unactivated - ), - Case(name="observed cluster keeps composing through an activation blip", req=req_blip, want=want_blip), - Case(name="Vultr cluster first pass composes VultrCluster XR only", req=req14, want=want14), - Case( - name="Vultr credentials pass through to VultrCluster spec", - req=req_creds_vultr, - want=want_creds_vultr, - ), - Case( - name="Vultr cluster ready composes CPC without identity, ServingStack, and Usage", - req=req15, - want=want15, - ), - *guard_cases, - ] - - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) - - -class TestGatewayStatus(unittest.IsolatedAsyncioTestCase): - """The hostname gate, which is what keeps a cluster off the schedule until - traffic to it is mutually authenticated in both directions.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - @staticmethod - def _request(*, address: str | None, ca: str | None, gateway_cas: list[str]) -> fnv1.RunFunctionRequest: - """A cluster and whatever its serving stack and the fleet's gateways have - published so far. The gateway name is Modelplane's own, so nothing - configures it.""" - xr = v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), - ), + fnv1.Condition( + type="BackendReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", ), - ) - stack_status: dict = {"conditions": [{"type": "Ready", "status": "True"}]} - gateway: dict = {} - if address: - gateway["address"] = address - if ca: - gateway["caCertificate"] = ca - if gateway: - stack_status["gateway"] = gateway - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json")) - ), - resources={ - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": {"name": "test-cluster-serving-stack-fd00b"}, - "status": stack_status, - } - ), - ), - }, + ], + context=structpb.Struct(), + ) + want_creds.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) + want_creds.requirements.resources["gateways"].CopyFrom(_gateways_selector()) + want_creds.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) + want_creds.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) + want_creds.requirements.resources["model-caches"].CopyFrom(_caches_selector()) + + # Every cloud cluster composes an activation policy; with the policy + # observed Healthy the cluster XR is composed. + for req, want, kinds in [ + (req2, want2, fn._ACTIVATE_GCP), + (req_creds, want_creds, fn._ACTIVATE_GCP), + (req4, want4, fn._ACTIVATE_AWS), + (req5, want5, fn._ACTIVATE_AWS), + (req6, want6, fn._ACTIVATE_GCP), + (req7, want7, fn._ACTIVATE_AWS), + (req8, want8, fn._ACTIVATE_AWS), + (req9, want9, fn._ACTIVATE_AWS), + (req10, want10, fn._ACTIVATE_NEBIUS), + (req11, want11, fn._ACTIVATE_NEBIUS), + (req12, want12, fn._ACTIVATE_AZURE), + (req13, want13, fn._ACTIVATE_AZURE), + (req14, want14, fn._ACTIVATE_VULTR), + (req_creds_vultr, want_creds_vultr, fn._ACTIVATE_VULTR), + (req15, want15, fn._ACTIVATE_VULTR), + ]: + _observe_activated(req, kinds) + _want_activation(want, kinds) + + # While the policy is missing even one of the kinds from status.activated + # (e.g. a provider still installing), and with no cluster observed, the + # function composes only the activation policy (not marked ready, so the + # composite doesn't report ready), not the cluster XR. Observing the + # policy with a kind missing exercises the all-kinds check rather than + # the policy-absent branch. + req_unactivated = copy.deepcopy(req2) + _observe_activated(req_unactivated, fn._ACTIVATE_GCP[:-1]) + want_unactivated = copy.deepcopy(want2) + del want_unactivated.desired.resources["gke-cluster"] + want_unactivated.desired.resources["activation"].ClearField("ready") + + # Once the cluster is observed, the function keeps composing it even + # when the policy momentarily stops reporting the kinds active, so an + # activation blip never drops a provisioned cluster from desired state. + req_blip = copy.deepcopy(req6) + del req_blip.observed.resources["activation"] + want_blip = copy.deepcopy(want6) + + return [ + Case(name="existing cluster with secrets composes backend and CPC", req=req1, want=want1), + Case(name="existing cluster with a non-GCP identity threads the identity type", req=req1b, want=want1b), + Case(name="GKE cluster first pass composes GKECluster XR only", req=req2, want=want2), + Case(name="GKE credentials pass through to GKECluster spec", req=req_creds, want=want_creds), + Case(name="existing cluster second pass with backend ready", req=req3, want=want3), + Case(name="EKS cluster first pass composes EKSCluster XR only", req=req4, want=want4), + Case(name="EKS cluster not ready re-emits existing CPC unchanged", req=req5, want=want5), + Case(name="GKE cluster ready composes CPC, backend, usage, and RWX StorageClass", req=req6, want=want6), + Case(name="EKS cluster ready composes ServingStack and Usage", req=req7, want=want7), + Case( + name="EKS node pool with a Capacity Block sets capacityBlock on the EKSCluster pool", + req=req8, + want=want8, + ), + Case( + name="EKS node pool with fabric EFA sets fabric on the EKSCluster pool", + req=req9, + want=want9, + ), + Case(name="Nebius cluster first pass composes NebiusCluster XR only", req=req10, want=want10), + Case( + name="Nebius cluster ready composes CPC with Nebius identity, ServingStack, and Usage", + req=req11, + want=want11, + ), + Case(name="AKS cluster first pass composes AKSCluster XR only", req=req12, want=want12), + Case( + name="AKS cluster ready composes CPC without identity, ServingStack, and Usage", + req=req13, + want=want13, + ), + Case(name="cloud cluster not activated composes only the policy", req=req_unactivated, want=want_unactivated), + Case(name="observed cluster keeps composing through an activation blip", req=req_blip, want=want_blip), + Case(name="Vultr cluster first pass composes VultrCluster XR only", req=req14, want=want14), + Case( + name="Vultr credentials pass through to VultrCluster spec", + req=req_creds_vultr, + want=want_creds_vultr, + ), + Case( + name="Vultr cluster ready composes CPC without identity, ServingStack, and Usage", + req=req15, + want=want15, + ), + *guard_cases, + ] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes the resources an InferenceCluster needs.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) + + +# The hostname gate, which is what keeps a cluster off the schedule until +# traffic to it is mutually authenticated in both directions. + + +def _gateway_status_request(*, address: str | None, ca: str | None, gateway_cas: list[str]) -> fnv1.RunFunctionRequest: + """A cluster and whatever its serving stack and the fleet's gateways have + published so far. The gateway name is Modelplane's own, so nothing + configures it.""" + xr = v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), + spec=v1alpha1.Spec( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), ), - ) - for i, cert in enumerate(gateway_cas): - req.required_resources["gateways"].items.append( - fnv1.Resource( + ), + ) + stack_status: dict = {"conditions": [{"type": "Ready", "status": "True"}]} + gateway: dict = {} + if address: + gateway["address"] = address + if ca: + gateway["caCertificate"] = ca + if gateway: + stack_status["gateway"] = gateway + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json"))), + resources={ + "serving-stack": fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceGateway", - "metadata": {"name": f"fleet-{i}"}, - "spec": {"clusterName": "test-cluster"}, - "status": {"clientCACertificate": cert}, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": {"name": "test-cluster-serving-stack-fd00b"}, + "status": stack_status, } ), - ) - ) - return req - - async def _gateway_status(self, req: fnv1.RunFunctionRequest) -> dict: - got = await self.runner.RunFunction(req, None) - return resource.struct_to_dict(got.desired.composite.resource).get("status", {}).get("gateway", {}) - - async def test_hostname_published_once_both_directions_are_authenticated(self) -> None: - """An address to reach, this cluster's CA so an InferenceGateway can tell - it reached the right cluster, and an InferenceGateway CA so the cluster - gateway demands a client certificate.""" - status = await self._gateway_status( - self._request(address="34.55.100.10", ca="cluster-ca", gateway_cas=["fleet-ca"]) - ) - self.assertEqual( - status, - { - "address": "34.55.100.10", - "caCertificate": "cluster-ca", - "hostname": _GATEWAY_HOSTNAME, + ), }, + ), + ) + for i, cert in enumerate(gateway_cas): + req.required_resources["gateways"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceGateway", + "metadata": {"name": f"fleet-{i}"}, + "spec": {"clusterName": "test-cluster"}, + "status": {"clientCACertificate": cert}, + } + ), + ) ) - - async def test_no_hostname_without_an_inference_gateway_ca(self) -> None: - """The case that matters: the cluster gateway only demands a client - certificate when it has a CA to check against, and with none it serves no - Gateway at all. Publishing the hostname anyway would make the cluster - schedulable when nothing is listening on it, so every request routed - there would be stranded.""" - status = await self._gateway_status(self._request(address="34.55.100.10", ca="cluster-ca", gateway_cas=[])) - self.assertEqual(status, {"address": "34.55.100.10", "caCertificate": "cluster-ca"}) - - async def test_no_hostname_without_this_clusters_ca(self) -> None: - """Without it an InferenceGateway can't validate the cluster gateway it - reaches, so it would have to fall back to the public trust store.""" - status = await self._gateway_status(self._request(address="34.55.100.10", ca=None, gateway_cas=["fleet-ca"])) - self.assertEqual(status, {"address": "34.55.100.10"}) - - async def test_no_gateway_status_before_an_address(self) -> None: - """A hostname that resolves to nothing strands every request routed to - it, and the CA is republished from the same status.""" - status = await self._gateway_status(self._request(address=None, ca="cluster-ca", gateway_cas=["fleet-ca"])) - self.assertEqual(status, {}) - - async def test_serving_stack_accepts_every_inference_gateway_ca(self) -> None: - """Any InferenceGateway may forward to this cluster, so its gateway - accepts every published CA, whichever cluster the InferenceGateway runs - on. These CAs are what switches the cluster gateway's mTLS listener on. - One that hasn't published a CA yet is left out rather than holding the - others back, and the list is sorted so it doesn't churn.""" - req = self._request(address="34.55.100.10", ca="cluster-ca", gateway_cas=["fleet-0-ca", "fleet-1-ca"]) - for name, status in (("aaa", {"clientCACertificate": "aaa-ca"}), ("pending", {})): - req.required_resources["gateways"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceGateway", - "metadata": {"name": name}, - "spec": {"clusterName": "elsewhere"}, - "status": status, - } - ), - ) + return req + + +def _gateway_status(req: fnv1.RunFunctionRequest) -> dict: + """The status.gateway RunFunction writes to the InferenceCluster for req.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + return resource.struct_to_dict(got.desired.composite.resource).get("status", {}).get("gateway", {}) + + +def test_gateway_status_hostname_published_once_both_directions_are_authenticated() -> None: + """The hostname is published once the address and both directions' CAs are.""" + # An address to reach, this cluster's CA so an InferenceGateway can tell it + # reached the right cluster, and an InferenceGateway CA so the cluster + # gateway demands a client certificate. + status = _gateway_status(_gateway_status_request(address="34.55.100.10", ca="cluster-ca", gateway_cas=["fleet-ca"])) + assert status == { + "address": "34.55.100.10", + "caCertificate": "cluster-ca", + "hostname": _GATEWAY_HOSTNAME, + } + + +def test_gateway_status_no_hostname_without_an_inference_gateway_ca() -> None: + """No hostname is published until an InferenceGateway has published a CA.""" + # The case that matters: the cluster gateway only demands a client + # certificate when it has a CA to check against, and with none it serves no + # Gateway at all. Publishing the hostname anyway would make the cluster + # schedulable when nothing is listening on it, so every request routed + # there would be stranded. + status = _gateway_status(_gateway_status_request(address="34.55.100.10", ca="cluster-ca", gateway_cas=[])) + assert status == {"address": "34.55.100.10", "caCertificate": "cluster-ca"} + + +def test_gateway_status_no_hostname_without_this_clusters_ca() -> None: + """No hostname is published until the cluster's own gateway CA is.""" + # Without it an InferenceGateway can't validate the cluster gateway it + # reaches, so it would have to fall back to the public trust store. + status = _gateway_status(_gateway_status_request(address="34.55.100.10", ca=None, gateway_cas=["fleet-ca"])) + assert status == {"address": "34.55.100.10"} + + +def test_gateway_status_no_gateway_status_before_an_address() -> None: + """No gateway status at all is published before the gateway has an address.""" + # A hostname that resolves to nothing strands every request routed to it, + # and the CA is republished from the same status. + status = _gateway_status(_gateway_status_request(address=None, ca="cluster-ca", gateway_cas=["fleet-ca"])) + assert status == {} + + +def test_gateway_status_serving_stack_accepts_every_inference_gateway_ca() -> None: + """The ServingStack's gateway accepts every published InferenceGateway CA, sorted.""" + # Any InferenceGateway may forward to this cluster, so its gateway accepts + # every published CA, whichever cluster the InferenceGateway runs on. These + # CAs are what switches the cluster gateway's mTLS listener on. One that + # hasn't published a CA yet is left out rather than holding the others + # back, and the list is sorted so it doesn't churn. + req = _gateway_status_request(address="34.55.100.10", ca="cluster-ca", gateway_cas=["fleet-0-ca", "fleet-1-ca"]) + for name, status in (("aaa", {"clientCACertificate": "aaa-ca"}), ("pending", {})): + req.required_resources["gateways"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceGateway", + "metadata": {"name": name}, + "spec": {"clusterName": "elsewhere"}, + "status": status, + } + ), ) - got = await self.runner.RunFunction(req, None) - stack = resource.struct_to_dict(got.desired.resources[fn.BACKEND_RESOURCE_KEY].resource) - self.assertEqual( - stack["spec"]["gateway"], - { - "hostname": _GATEWAY_HOSTNAME, - "clientCAs": [ - {"name": "aaa", "certificate": "aaa-ca"}, - {"name": "fleet-0", "certificate": "fleet-0-ca"}, - {"name": "fleet-1", "certificate": "fleet-1-ca"}, - ], - }, ) - - -class TestGatewayHostname(unittest.TestCase): - """The derived gateway hostname doubles as an SNI and a certificate SAN, so - two clusters must never derive the same one.""" - - def test_dots_become_a_single_dns_label(self) -> None: - """A dotted cluster name is a DNS-1123 subdomain, but the first segment - of the hostname has to be one DNS-1035 label.""" - hostname = fn._gateway_hostname("eu.example") - label = hostname.split(".")[0] - self.assertNotIn(".", label) - self.assertTrue(label.startswith("gateway-")) - - def test_names_differing_only_in_dots_do_not_collide(self) -> None: - """The hash covers the raw cluster name, so 'eu.example' and - 'eu-example' get different hostnames. Sharing one, a cluster's Service - would shadow the other's under a certificate it accepts.""" - self.assertNotEqual(fn._gateway_hostname("eu.example"), fn._gateway_hostname("eu-example")) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + stack = resource.struct_to_dict(got.desired.resources[fn.BACKEND_RESOURCE_KEY].resource) + assert stack["spec"]["gateway"] == { + "hostname": _GATEWAY_HOSTNAME, + "clientCAs": [ + {"name": "aaa", "certificate": "aaa-ca"}, + {"name": "fleet-0", "certificate": "fleet-0-ca"}, + {"name": "fleet-1", "certificate": "fleet-1-ca"}, + ], + } + + +# The derived gateway hostname doubles as an SNI and a certificate SAN, so two +# clusters must never derive the same one. + + +def test_gateway_hostname_dots_become_a_single_dns_label() -> None: + """A dotted cluster name still gives a hostname whose first label has no dots.""" + # A dotted cluster name is a DNS-1123 subdomain, but the first segment of + # the hostname has to be one DNS-1035 label. + hostname = fn._gateway_hostname("eu.example") + label = hostname.split(".")[0] + assert "." not in label + assert label.startswith("gateway-") + + +def test_gateway_hostname_names_differing_only_in_dots_do_not_collide() -> None: + """Cluster names that differ only in dots get different hostnames.""" + # The hash covers the raw cluster name, so 'eu.example' and 'eu-example' + # get different hostnames. Sharing one, a cluster's Service would shadow + # the other's under a certificate it accepts. + assert fn._gateway_hostname("eu.example") != fn._gateway_hostname("eu-example") diff --git a/functions/compose-inference-gateway/tests/__init__.py b/functions/compose-inference-gateway/tests/__init__.py deleted file mode 100644 index 5d373016d..000000000 --- a/functions/compose-inference-gateway/tests/__init__.py +++ /dev/null @@ -1,15 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - - diff --git a/functions/compose-inference-gateway/tests/test_fn.py b/functions/compose-inference-gateway/tests/test_fn.py index c19ce6379..9f4379031 100644 --- a/functions/compose-inference-gateway/tests/test_fn.py +++ b/functions/compose-inference-gateway/tests/test_fn.py @@ -14,15 +14,17 @@ """Tests for the compose-inference-gateway function.""" +import asyncio import base64 import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.inferencegateway import v1alpha1 @@ -230,1040 +232,934 @@ def _not_ready(reason: str, message: str, requirements: fnv1.Requirements) -> fn ) -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) +GATES_CASES = [ + Case( + name="unresolved requirements compose nothing", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + ), + want=_not_ready( + fn.CONDITION_REASON_WAITING_FOR_CLUSTER, + "Waiting for the gateway's cluster and the other gateways to resolve", + _requirements(), + ), + ), + Case( + name="a named cluster that does not exist", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + required_resources=_required(clusters=[], gateways=[_gateway_xr("eu", _CLUSTER)]), + ), + want=_not_ready( + fn.CONDITION_REASON_WAITING_FOR_CLUSTER, + f"InferenceCluster {_CLUSTER} does not exist", + _requirements(), + ), + ), + Case( + name="a cluster that already hosts a lower-named gateway", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + required_resources=_required( + clusters=[_cluster()], + gateways=[_gateway_xr("eu", _CLUSTER), _gateway_xr("aaa", _CLUSTER)], + ), + ), + want=_not_ready( + fn.CONDITION_REASON_CLUSTER_TAKEN, + f"InferenceCluster {_CLUSTER} already hosts InferenceGateway aaa", + _requirements(), + ), + ), + Case( + name="a cluster with no providerConfigRef yet", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + required_resources=_required( + clusters=[_cluster(provider_config=None)], gateways=[_gateway_xr("eu", _CLUSTER)] + ), + ), + want=_not_ready( + fn.CONDITION_REASON_WAITING_FOR_CLUSTER, + f"InferenceCluster {_CLUSTER} has not published a providerConfigRef", + _requirements(), + ), + ), +] -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - maxDiff = None +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - async def test_gates(self) -> None: - """Passes where the gateway can't be composed compose nothing, and say - why. Asserting the whole response proves nothing is composed against a - cluster we can't reach, rather than a subset being applied.""" - cases = [ - Case( - name="unresolved requirements compose nothing", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - ), - want=_not_ready( - fn.CONDITION_REASON_WAITING_FOR_CLUSTER, - "Waiting for the gateway's cluster and the other gateways to resolve", - _requirements(), - ), - ), - Case( - name="a named cluster that does not exist", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required(clusters=[], gateways=[_gateway_xr("eu", _CLUSTER)]), - ), - want=_not_ready( - fn.CONDITION_REASON_WAITING_FOR_CLUSTER, - f"InferenceCluster {_CLUSTER} does not exist", - _requirements(), - ), - ), - Case( - name="a cluster that already hosts a lower-named gateway", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER), _gateway_xr("aaa", _CLUSTER)], - ), - ), - want=_not_ready( - fn.CONDITION_REASON_CLUSTER_TAKEN, - f"InferenceCluster {_CLUSTER} already hosts InferenceGateway aaa", - _requirements(), - ), - ), - Case( - name="a cluster with no providerConfigRef yet", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required( - clusters=[_cluster(provider_config=None)], gateways=[_gateway_xr("eu", _CLUSTER)] - ), - ), - want=_not_ready( - fn.CONDITION_REASON_WAITING_FOR_CLUSTER, - f"InferenceCluster {_CLUSTER} has not published a providerConfigRef", - _requirements(), - ), - ), - ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) +@pytest.mark.parametrize("case", GATES_CASES, ids=lambda case: case.name) +def test_gates(case: Case) -> None: + """Passes where the gateway can't be composed compose nothing, and say + why. Asserting the whole response proves nothing is composed against a + cluster we can't reach, rather than a subset being applied.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) - async def test_minimal_gateway(self) -> None: - """A gateway with no TLS or auth: the getting-started shape. - Composes the gateway objects and no auth policies, and reports no - endpoints until the Gateway has an address. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = await self.runner.RunFunction(req, None) - - self.assertEqual( - sorted(got.desired.resources), - sorted( - [ - # The CA whose client certificates a cluster gateway trusts, - # published as a ClusterIssuer for compose-model-route to issue - # per-namespace client certificates from. - "client-ca-certificate", - "client-ca-issuer", - "client-ca-bundle", - "client-ca-configmap", - "client-selfsigned-issuer", - "client-traffic-policy", - "envoy-proxy", - "failover-policy", - "gateway", - "healthz-filter", - "healthz-route", - ] - ), - "composes the gateway objects and its client PKI, and no caller auth", - ) - for key, res in got.desired.resources.items(): - d = resource.struct_to_dict(res.resource) - self.assertEqual(d["kind"], "Object", f"{key} targets the gateway's cluster") - self.assertEqual( - d["spec"]["providerConfigRef"], - {"kind": "ClusterProviderConfig", "name": _PC}, - f"{key} uses the cluster's ClusterProviderConfig", - ) - # An InferenceGateway is cluster-scoped, and Crossplane only - # defaults a composed namespaced resource's namespace from a - # namespaced composite. Without this every reconcile fails with - # "an empty namespace may not be set when a resource name is - # provided" and nothing is composed at all. - self.assertEqual( - d["metadata"]["namespace"], - fn.CONTROL_PLANE_NAMESPACE, - f"{key} sets its own namespace, which a cluster-scoped XR must", - ) - manifest = d["spec"]["forProvider"]["manifest"] - if manifest["kind"] == "ClusterIssuer": - # Cluster-scoped: compose-model-route issues client certs from it - # into team namespaces, so it has no namespace of its own. - self.assertNotIn( - "namespace", - manifest["metadata"], - f"{key} is cluster-scoped, so it sets no namespace", - ) - continue - if manifest["kind"] == "Bundle": - # A Bundle is cluster-scoped, so it has no namespace of its own. - # It picks the namespace it syncs its ConfigMap to by selector. - self.assertNotIn( - "namespace", - manifest["metadata"], - f"{key} is cluster-scoped, so it sets no namespace", - ) - self.assertEqual( - manifest["spec"]["target"]["namespaceSelector"], - {"matchLabels": {"kubernetes.io/metadata.name": fn.REMOTE_NAMESPACE}}, - f"{key} syncs only to the remote namespace", - ) - continue - self.assertEqual( - manifest["metadata"]["namespace"], - fn.REMOTE_NAMESPACE, - f"{key} lands in the remote namespace", - ) +def test_minimal_gateway() -> None: + """A gateway with no TLS or auth: the getting-started shape. - # Two attempts per priority, so a retry tries another endpoint at the - # same priority before moving down. At one, a single transient failure - # on one replica would send the request to the next priority, which may - # be a paid provider. - failover = resource.struct_to_dict(got.desired.resources["failover-policy"].resource) - self.assertEqual( - failover["spec"]["forProvider"]["manifest"]["spec"]["retry"], - { - "numAttemptsPerPriority": 2, - "numRetries": 3, - "retryOn": { - # retriable-status-codes has to be present for the status - # codes below to do anything: Envoy Gateway replaces retry_on - # wholesale with this list, and Envoy only consults - # retriable_status_codes when retry_on names it. Without it a - # provider answering 503 or 429 is never retried, which is - # the case failover exists for. - "triggers": [ - "connect-failure", - "refused-stream", - "reset", - "retriable-status-codes", - ], - # 429 so a rate-limited provider's traffic overflows to - # another endpoint rather than failing back to the caller. - "httpStatusCodes": [429, 503], - }, - }, - ) - # Panic mode defaults to 50%, above which Envoy ignores health and - # spreads traffic over every endpoint including the ejected ones. Every - # endpoint of a ModelService shares one cluster, so ejecting a whole - # priority tier usually crosses it and failover stops working. - # - # Asserted on the whole healthCheck, because panicThreshold is a sibling - # of passive rather than a field inside it, and nested wrongly the API - # server prunes it while the policy still applies. - self.assertEqual( - failover["spec"]["forProvider"]["manifest"]["spec"]["healthCheck"], - { - "passive": { - "baseEjectionTime": "30s", - "consecutive5XxErrors": 5, - "interval": "5s", - "maxEjectionPercent": 100, - }, - "panicThreshold": 0, - }, + Composes the gateway objects and no auth policies, and reports no + endpoints until the Gateway has an address. + """ + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + assert sorted(got.desired.resources) == sorted( + [ + # The CA whose client certificates a cluster gateway trusts, + # published as a ClusterIssuer for compose-model-route to issue + # per-namespace client certificates from. + "client-ca-certificate", + "client-ca-issuer", + "client-ca-bundle", + "client-ca-configmap", + "client-selfsigned-issuer", + "client-traffic-policy", + "envoy-proxy", + "failover-policy", + "gateway", + "healthz-filter", + "healthz-route", + ] + ), "composes the gateway objects and its client PKI, and no caller auth" + for key, res in got.desired.resources.items(): + d = resource.struct_to_dict(res.resource) + assert d["kind"] == "Object", f"{key} targets the gateway's cluster" + assert d["spec"]["providerConfigRef"] == {"kind": "ClusterProviderConfig", "name": _PC}, ( + f"{key} uses the cluster's ClusterProviderConfig" ) - self.assertEqual( - failover["spec"]["forProvider"]["manifest"]["spec"]["targetRefs"], - [{"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": fn._GATEWAY_NAME}], - "targets the Gateway, so it covers every ModelService's route", + # An InferenceGateway is cluster-scoped, and Crossplane only + # defaults a composed namespaced resource's namespace from a + # namespaced composite. Without this every reconcile fails with + # "an empty namespace may not be set when a resource name is + # provided" and nothing is composed at all. + assert d["metadata"]["namespace"] == fn.CONTROL_PLANE_NAMESPACE, ( + f"{key} sets its own namespace, which a cluster-scoped XR must" ) - - # AI Gateway buffers whole bodies, and Envoy Gateway's 32KiB default - # buffer limit answers 413 to a long prompt or non-streamed completion. - self.assertEqual( - resource.struct_to_dict(got.desired.resources["client-traffic-policy"].resource)["spec"]["forProvider"][ - "manifest" + manifest = d["spec"]["forProvider"]["manifest"] + if manifest["kind"] == "ClusterIssuer": + # Cluster-scoped: compose-model-route issues client certs from it + # into team namespaces, so it has no namespace of its own. + assert "namespace" not in manifest["metadata"], f"{key} is cluster-scoped, so it sets no namespace" + continue + if manifest["kind"] == "Bundle": + # A Bundle is cluster-scoped, so it has no namespace of its own. + # It picks the namespace it syncs its ConfigMap to by selector. + assert "namespace" not in manifest["metadata"], f"{key} is cluster-scoped, so it sets no namespace" + assert manifest["spec"]["target"]["namespaceSelector"] == { + "matchLabels": {"kubernetes.io/metadata.name": fn.REMOTE_NAMESPACE} + }, f"{key} syncs only to the remote namespace" + continue + assert manifest["metadata"]["namespace"] == fn.REMOTE_NAMESPACE, f"{key} lands in the remote namespace" + + # Two attempts per priority, so a retry tries another endpoint at the + # same priority before moving down. At one, a single transient failure + # on one replica would send the request to the next priority, which may + # be a paid provider. + failover = resource.struct_to_dict(got.desired.resources["failover-policy"].resource) + assert failover["spec"]["forProvider"]["manifest"]["spec"]["retry"] == { + "numAttemptsPerPriority": 2, + "numRetries": 3, + "retryOn": { + # retriable-status-codes has to be present for the status + # codes below to do anything: Envoy Gateway replaces retry_on + # wholesale with this list, and Envoy only consults + # retriable_status_codes when retry_on names it. Without it a + # provider answering 503 or 429 is never retried, which is + # the case failover exists for. + "triggers": [ + "connect-failure", + "refused-stream", + "reset", + "retriable-status-codes", ], - { - "apiVersion": "gateway.envoyproxy.io/v1alpha1", - "kind": "ClientTrafficPolicy", - "metadata": {"name": "inference-gateway-client-traffic", "namespace": "modelplane-system"}, - "spec": { - "targetRefs": [ - {"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": "inference-gateway"} - ], - "connection": {"bufferLimit": "50Mi"}, - "http2": {"initialStreamWindowSize": "16Mi", "initialConnectionWindowSize": "24Mi"}, - }, - }, - ) + # 429 so a rate-limited provider's traffic overflows to + # another endpoint rather than failing back to the caller. + "httpStatusCodes": [429, 503], + }, + } + # Panic mode defaults to 50%, above which Envoy ignores health and + # spreads traffic over every endpoint including the ejected ones. Every + # endpoint of a ModelService shares one cluster, so ejecting a whole + # priority tier usually crosses it and failover stops working. + # + # Asserted on the whole healthCheck, because panicThreshold is a sibling + # of passive rather than a field inside it, and nested wrongly the API + # server prunes it while the policy still applies. + assert failover["spec"]["forProvider"]["manifest"]["spec"]["healthCheck"] == { + "passive": { + "baseEjectionTime": "30s", + "consecutive5XxErrors": 5, + "interval": "5s", + "maxEjectionPercent": 100, + }, + "panicThreshold": 0, + } + assert failover["spec"]["forProvider"]["manifest"]["spec"]["targetRefs"] == [ + {"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": fn._GATEWAY_NAME} + ], "targets the Gateway, so it covers every ModelService's route" + + # AI Gateway buffers whole bodies, and Envoy Gateway's 32KiB default + # buffer limit answers 413 to a long prompt or non-streamed completion. + assert resource.struct_to_dict(got.desired.resources["client-traffic-policy"].resource)["spec"]["forProvider"][ + "manifest" + ] == { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "ClientTrafficPolicy", + "metadata": {"name": "inference-gateway-client-traffic", "namespace": "modelplane-system"}, + "spec": { + "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": "inference-gateway"}], + "connection": {"bufferLimit": "50Mi"}, + "http2": {"initialStreamWindowSize": "16Mi", "initialConnectionWindowSize": "24Mi"}, + }, + } - # The token fields must read request metadata, not the response body or - # a header. The caller header is stripped before a third-party backend - # sees it, so a log reading the header loses the caller on exactly the - # records that attribute provider spend. - log = resource.struct_to_dict(got.desired.resources["envoy-proxy"].resource) - fields = log["spec"]["forProvider"]["manifest"]["spec"]["telemetry"]["accessLog"]["settings"][0]["format"][ - "json" - ] - # Two proxy pods spread softly across nodes and zones, a disruption - # budget so a drain can't evict both, and ndots:1. Without ndots:1 every - # backend hostname is resolved against each of the pod's search domains - # first, since they all have fewer than five dots. A cluster whose - # upstream resolver is slow then stalls resolution, and Envoy answers 503 - # with nothing but DNS timeouts to show for it. - proxy_labels = { - "gateway.envoyproxy.io/owning-gateway-name": "inference-gateway", - "gateway.envoyproxy.io/owning-gateway-namespace": "modelplane-system", - } - self.assertEqual( - log["spec"]["forProvider"]["manifest"]["spec"]["provider"], - { - "type": "Kubernetes", - "kubernetes": { - "envoyService": {"externalTrafficPolicy": "Cluster"}, - "envoyDeployment": { - "replicas": 2, - "patch": {"type": "StrategicMerge", "value": fn._NDOTS_PATCH}, - "pod": { - "topologySpreadConstraints": [ - { - "maxSkew": 1, - "topologyKey": "kubernetes.io/hostname", - "whenUnsatisfiable": "ScheduleAnyway", - "labelSelector": {"matchLabels": proxy_labels}, - }, - { - "maxSkew": 1, - "topologyKey": "topology.kubernetes.io/zone", - "whenUnsatisfiable": "ScheduleAnyway", - "labelSelector": {"matchLabels": proxy_labels}, - }, - ] + # The token fields must read request metadata, not the response body or + # a header. The caller header is stripped before a third-party backend + # sees it, so a log reading the header loses the caller on exactly the + # records that attribute provider spend. + log = resource.struct_to_dict(got.desired.resources["envoy-proxy"].resource) + fields = log["spec"]["forProvider"]["manifest"]["spec"]["telemetry"]["accessLog"]["settings"][0]["format"]["json"] + # Two proxy pods spread softly across nodes and zones, a disruption + # budget so a drain can't evict both, and ndots:1. Without ndots:1 every + # backend hostname is resolved against each of the pod's search domains + # first, since they all have fewer than five dots. A cluster whose + # upstream resolver is slow then stalls resolution, and Envoy answers 503 + # with nothing but DNS timeouts to show for it. + proxy_labels = { + "gateway.envoyproxy.io/owning-gateway-name": "inference-gateway", + "gateway.envoyproxy.io/owning-gateway-namespace": "modelplane-system", + } + assert log["spec"]["forProvider"]["manifest"]["spec"]["provider"] == { + "type": "Kubernetes", + "kubernetes": { + "envoyService": {"externalTrafficPolicy": "Cluster"}, + "envoyDeployment": { + "replicas": 2, + "patch": {"type": "StrategicMerge", "value": fn._NDOTS_PATCH}, + "pod": { + "topologySpreadConstraints": [ + { + "maxSkew": 1, + "topologyKey": "kubernetes.io/hostname", + "whenUnsatisfiable": "ScheduleAnyway", + "labelSelector": {"matchLabels": proxy_labels}, }, - }, - "envoyPDB": {"maxUnavailable": 1}, + { + "maxSkew": 1, + "topologyKey": "topology.kubernetes.io/zone", + "whenUnsatisfiable": "ScheduleAnyway", + "labelSelector": {"matchLabels": proxy_labels}, + }, + ] }, }, - ) - # A stopping pod drains for as long as a request may run by default, so - # a restart doesn't cut off streams in flight. - self.assertEqual(log["spec"]["forProvider"]["manifest"]["spec"]["shutdown"], {"drainTimeout": "300s"}) - self.assertEqual( - fn._NDOTS_PATCH["spec"]["template"]["spec"]["dnsConfig"]["options"], - [{"name": "ndots", "value": "1"}], - ) - - self.assertEqual(fields["caller"], "%DYNAMIC_METADATA(io.envoy.ai_gateway:caller)%") - self.assertEqual(fields["input_tokens"], "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_input_token)%") - self.assertEqual(fields["output_tokens"], "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_output_token)%") - - gw = resource.struct_to_dict(got.desired.resources["gateway"].resource) - manifest = gw["spec"]["forProvider"]["manifest"] - self.assertEqual( - manifest["spec"]["listeners"], - [ - { - "name": "http", - "protocol": "HTTP", - "port": 80, - "allowedRoutes": { - "namespaces": { - "from": "Selector", - "selector": { - "matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}] - }, - } - }, + "envoyPDB": {"maxUnavailable": 1}, + }, + } + # A stopping pod drains for as long as a request may run by default, so + # a restart doesn't cut off streams in flight. + assert log["spec"]["forProvider"]["manifest"]["spec"]["shutdown"] == {"drainTimeout": "300s"} + assert fn._NDOTS_PATCH["spec"]["template"]["spec"]["dnsConfig"]["options"] == [{"name": "ndots", "value": "1"}] + + assert fields["caller"] == "%DYNAMIC_METADATA(io.envoy.ai_gateway:caller)%" + assert fields["input_tokens"] == "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_input_token)%" + assert fields["output_tokens"] == "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_output_token)%" + + gw = resource.struct_to_dict(got.desired.resources["gateway"].resource) + manifest = gw["spec"]["forProvider"]["manifest"] + assert manifest["spec"]["listeners"] == [ + { + "name": "http", + "protocol": "HTTP", + "port": 80, + "allowedRoutes": { + "namespaces": { + "from": "Selector", + "selector": {"matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}]}, } - ], - "one HTTP listener, no hostname, accepting routes from the mirrored namespaces", - ) - self.assertEqual( - manifest["spec"]["infrastructure"]["parametersRef"], - {"group": "gateway.envoyproxy.io", "kind": "EnvoyProxy", "name": fn._GATEWAY_NAME}, - "its own EnvoyProxy, not the GatewayClass's", - ) - self.assertEqual( - resource.struct_to_dict(got.desired.composite.resource).get("status"), - {}, - "nothing to report until the Gateway has an address", - ) + }, + } + ], "one HTTP listener, no hostname, accepting routes from the mirrored namespaces" + assert manifest["spec"]["infrastructure"]["parametersRef"] == { + "group": "gateway.envoyproxy.io", + "kind": "EnvoyProxy", + "name": fn._GATEWAY_NAME, + }, "its own EnvoyProxy, not the GatewayClass's" + assert resource.struct_to_dict(got.desired.composite.resource).get("status") == {}, ( + "nothing to report until the Gateway has an address" + ) - async def test_full_gateway(self) -> None: - """A gateway with TLS and auth, whose Gateway has an address. - - Checks the things a caller depends on: the HTTPS listener, the Secrets - copied to the cluster, the caller policy naming them, /healthz exempted - from that policy, and a status publishing no URLs, since a caller - reaches a TLS gateway on a DNS name only its owner knows. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr( - tls=v1alpha1.Tls(certificateRefs=[v1alpha1.CertificateRef(name="eu-tls-0")]), - auth=_api_key_auth(), - ) - ) - ), - resources={ - "gateway": _observed_gateway(_ADDRESS, ready=True), - "caller-auth": _observed_accepted(), - }, - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{ - "caller-secrets": [_secret("ml-team-keys", {"ml-team-assistant": "sk-mp-a1b2c3"})], - "tls-secret-0": [_secret("eu-tls-0", {"tls.crt": "cert", "tls.key": "key"})], - }, - ), - ) - got = await self.runner.RunFunction(req, None) - - self.assertEqual( - sorted(got.desired.resources), - [ - "caller-auth", - "caller-secret-ml-team-keys", - "client-ca-bundle", - "client-ca-certificate", - "client-ca-configmap", - "client-ca-issuer", - "client-selfsigned-issuer", - "client-traffic-policy", - "envoy-proxy", - "failover-policy", - "gateway", - "healthz-auth", - "healthz-filter", - "healthz-route", - "redirect-auth", - "redirect-route", - "tls-secret-eu-tls-0", - ], - ) - def manifest(key: str) -> dict: - return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] +def test_full_gateway() -> None: + """A gateway with TLS and auth, whose Gateway has an address. - self.assertEqual( - manifest("gateway")["spec"]["listeners"][1], - { - "name": "https", - "protocol": "HTTPS", - "port": 443, - "tls": {"mode": "Terminate", "certificateRefs": [{"name": "eu-tls-0"}]}, - "allowedRoutes": { - "namespaces": { - "from": "Selector", - "selector": {"matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}]}, - } - }, - }, - ) - self.assertEqual(got.requirements, _requirements(auth=True, tls=1)) - self.assertEqual( - manifest("tls-secret-eu-tls-0"), - { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": "eu-tls-0", "namespace": fn.REMOTE_NAMESPACE}, - "type": "kubernetes.io/tls", - "data": { - "tls.crt": base64.b64encode(b"cert").decode(), - "tls.key": base64.b64encode(b"key").decode(), - }, - }, - "the certificate is copied verbatim, keeping the name the Gateway refers to it by", - ) - self.assertEqual( - manifest("caller-auth")["spec"]["apiKeyAuth"], - { - "credentialRefs": [{"name": "callers-ml-team-keys"}], - # Authorization for OpenAI clients, x-api-key for Anthropic ones. - "extractFrom": [{"headers": ["Authorization", "x-api-key"]}], - "forwardClientIDHeader": fn._CALLER_HEADER, - "sanitize": True, + Checks the things a caller depends on: the HTTPS listener, the Secrets + copied to the cluster, the caller policy naming them, /healthz exempted + from that policy, and a status publishing no URLs, since a caller + reaches a TLS gateway on a DNS name only its owner knows. + """ + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr( + tls=v1alpha1.Tls(certificateRefs=[v1alpha1.CertificateRef(name="eu-tls-0")]), + auth=_api_key_auth(), + ) + ) + ), + resources={ + "gateway": _observed_gateway(_ADDRESS, ready=True), + "caller-auth": _observed_accepted(), }, - ) - self.assertEqual( - manifest("healthz-auth")["spec"], - { - "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "HTTPRoute", "name": fn._HEALTHZ_NAME}], - "authorization": {"defaultAction": "Allow"}, + ), + required_resources=_required( + clusters=[_cluster()], + gateways=[_gateway_xr("eu", _CLUSTER)], + **{ + "caller-secrets": [_secret("ml-team-keys", {"ml-team-assistant": "sk-mp-a1b2c3"})], + "tls-secret-0": [_secret("eu-tls-0", {"tls.crt": "cert", "tls.key": "key"})], }, - "/healthz overrides the Gateway-level policy so a health check needs no credential", - ) - # Inference binds to the HTTPS listener alone, so :80 carries only - # /healthz and this catch-all redirect to it. /healthz is an Exact match, - # so it still answers a plain-HTTP health check. - self.assertEqual( - manifest("healthz-route")["spec"]["parentRefs"], - [ - { - "group": "gateway.networking.k8s.io", - "kind": "Gateway", - "name": fn._GATEWAY_NAME, - "sectionName": "http", - } - ], - ) - self.assertEqual( - manifest("redirect-route")["spec"], + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + assert sorted(got.desired.resources) == [ + "caller-auth", + "caller-secret-ml-team-keys", + "client-ca-bundle", + "client-ca-certificate", + "client-ca-configmap", + "client-ca-issuer", + "client-selfsigned-issuer", + "client-traffic-policy", + "envoy-proxy", + "failover-policy", + "gateway", + "healthz-auth", + "healthz-filter", + "healthz-route", + "redirect-auth", + "redirect-route", + "tls-secret-eu-tls-0", + ] + + def manifest(key: str) -> dict: + return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] + + assert manifest("gateway")["spec"]["listeners"][1] == { + "name": "https", + "protocol": "HTTPS", + "port": 443, + "tls": {"mode": "Terminate", "certificateRefs": [{"name": "eu-tls-0"}]}, + "allowedRoutes": { + "namespaces": { + "from": "Selector", + "selector": {"matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}]}, + } + }, + } + assert _to_dict(got.requirements) == _to_dict(_requirements(auth=True, tls=1)) + assert manifest("tls-secret-eu-tls-0") == { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "eu-tls-0", "namespace": fn.REMOTE_NAMESPACE}, + "type": "kubernetes.io/tls", + "data": { + "tls.crt": base64.b64encode(b"cert").decode(), + "tls.key": base64.b64encode(b"key").decode(), + }, + }, "the certificate is copied verbatim, keeping the name the Gateway refers to it by" + assert manifest("caller-auth")["spec"]["apiKeyAuth"] == { + "credentialRefs": [{"name": "callers-ml-team-keys"}], + # Authorization for OpenAI clients, x-api-key for Anthropic ones. + "extractFrom": [{"headers": ["Authorization", "x-api-key"]}], + "forwardClientIDHeader": fn._CALLER_HEADER, + "sanitize": True, + } + assert manifest("healthz-auth")["spec"] == { + "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "HTTPRoute", "name": fn._HEALTHZ_NAME}], + "authorization": {"defaultAction": "Allow"}, + }, "/healthz overrides the Gateway-level policy so a health check needs no credential" + # Inference binds to the HTTPS listener alone, so :80 carries only + # /healthz and this catch-all redirect to it. /healthz is an Exact match, + # so it still answers a plain-HTTP health check. + assert manifest("healthz-route")["spec"]["parentRefs"] == [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": fn._GATEWAY_NAME, + "sectionName": "http", + } + ] + assert manifest("redirect-route")["spec"] == { + "parentRefs": [ { - "parentRefs": [ - { - "group": "gateway.networking.k8s.io", - "kind": "Gateway", - "name": fn._GATEWAY_NAME, - "sectionName": "http", - } - ], - "rules": [ - { - "matches": [{"path": {"type": "PathPrefix", "value": "/"}}], - "filters": [ - {"type": "RequestRedirect", "requestRedirect": {"scheme": "https", "statusCode": 301}} - ], - } - ], - }, - ) - self.assertEqual( - manifest("redirect-auth")["spec"], + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": fn._GATEWAY_NAME, + "sectionName": "http", + } + ], + "rules": [ { - "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "HTTPRoute", "name": fn._REDIRECT_NAME}], - "authorization": {"defaultAction": "Allow"}, - }, - "the redirect must happen before auth, or an unauthenticated caller gets 401 instead of being sent to HTTPS", - ) - self.assertEqual(resource.struct_to_dict(got.desired.composite.resource)["status"], {"address": _ADDRESS}) - self.assertEqual( - list(got.conditions), - [ - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_TRUE, - reason=fn.CONDITION_REASON_GATEWAY_PROGRAMMED, - ) - ], + "matches": [{"path": {"type": "PathPrefix", "value": "/"}}], + "filters": [{"type": "RequestRedirect", "requestRedirect": {"scheme": "https", "statusCode": 301}}], + } + ], + } + assert manifest("redirect-auth")["spec"] == { + "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "HTTPRoute", "name": fn._REDIRECT_NAME}], + "authorization": {"defaultAction": "Allow"}, + }, "the redirect must happen before auth, or an unauthenticated caller gets 401 instead of being sent to HTTPS" + assert resource.struct_to_dict(got.desired.composite.resource)["status"] == {"address": _ADDRESS} + assert [_to_dict(c) for c in got.conditions] == [ + _to_dict( + fnv1.Condition( + type=fn.CONDITION_TYPE_GATEWAY_READY, + status=fnv1.STATUS_CONDITION_TRUE, + reason=fn.CONDITION_REASON_GATEWAY_PROGRAMMED, + ) ) + ] - async def test_endpoints_are_built_from_the_address(self) -> None: - """A gateway serving plain HTTP publishes URLs on its address, which is - something a caller can actually put in an SDK's base_url.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={"gateway": _observed_gateway(_ADDRESS, ready=False)}, - ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual( - resource.struct_to_dict(got.desired.composite.resource)["status"], - { - "address": _ADDRESS, - "endpoints": { - "openAI": f"http://{_ADDRESS}/v1", - "anthropic": f"http://{_ADDRESS}/anthropic/v1", - }, - }, - ) - self.assertEqual( - next(iter(got.conditions)).reason, - fn.CONDITION_REASON_WAITING_FOR_GATEWAY, - "an address alone isn't readiness; the Gateway must be programmed", - ) - async def test_an_ipv6_address_is_bracketed_in_the_endpoints(self) -> None: - """A bare IPv6 literal collides with the port separator in a URL, so an - SDK given http://2001:db8::1/v1 as a base_url can't use it.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={"gateway": _observed_gateway("2001:db8::1", ready=False)}, - ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual( - resource.struct_to_dict(got.desired.composite.resource)["status"]["endpoints"], - { - "openAI": "http://[2001:db8::1]/v1", - "anthropic": "http://[2001:db8::1]/anthropic/v1", - }, - ) +def test_endpoints_are_built_from_the_address() -> None: + """A gateway serving plain HTTP publishes URLs on its address, which is + something a caller can actually put in an SDK's base_url.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), + resources={"gateway": _observed_gateway(_ADDRESS, ready=False)}, + ), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert resource.struct_to_dict(got.desired.composite.resource)["status"] == { + "address": _ADDRESS, + "endpoints": { + "openAI": f"http://{_ADDRESS}/v1", + "anthropic": f"http://{_ADDRESS}/anthropic/v1", + }, + } + assert next(iter(got.conditions)).reason == fn.CONDITION_REASON_WAITING_FOR_GATEWAY, ( + "an address alone isn't readiness; the Gateway must be programmed" + ) - async def test_resolves_each_cluster_gateway_name(self) -> None: - """A Service per cluster gateway, resolving its name to its address here. - - A ModelService's backends address a cluster gateway by the name - compose-inference-cluster derived, and this gateway's Envoy resolves it, - so its cluster needs a Service of that name. An IP is served by a - headless Service and an EndpointSlice; a hostname, which is how a cloud - load balancer names itself, by an ExternalName Service. A cluster that - hasn't published both an address and a name gets neither. - """ - ipv4 = "prod-ipv4-gateway-aaaaa.modelplane-system.svc.cluster.local" - ipv6 = "prod-ipv6-gateway-bbbbb.modelplane-system.svc.cluster.local" - dns = "prod-dns-gateway-ccccc.modelplane-system.svc.cluster.local" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required( - gateways=[_gateway_xr("eu", _CLUSTER)], - clusters=[ - _cluster(), # this gateway's own cluster, no gateway published yet - _cluster_with_gateway("prod-ipv4", address="203.0.113.7", hostname=ipv4), - _cluster_with_gateway("prod-ipv6", address="2001:db8::1", hostname=ipv6), - _cluster_with_gateway("prod-dns", address="lb-x.elb.amazonaws.com", hostname=dns), - ], - ), - ) - got = await self.runner.RunFunction(req, None) - resolvers = { - key: resource.struct_to_dict(res.resource) - for key, res in got.desired.resources.items() - if key.startswith("cluster-name") - } - for key, obj in resolvers.items(): - self.assertEqual( - obj["spec"]["providerConfigRef"], - {"kind": "ClusterProviderConfig", "name": _PC}, - f"{key} is composed against this gateway's own cluster", - ) - manifests = {key: obj["spec"]["forProvider"]["manifest"] for key, obj in resolvers.items()} - self.assertEqual( - manifests, - { - "cluster-name-prod-ipv4-gateway-aaaaa": { - "apiVersion": "v1", - "kind": "Service", - "metadata": {"name": "prod-ipv4-gateway-aaaaa", "namespace": fn.REMOTE_NAMESPACE}, - "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, - }, - "cluster-name-slice-prod-ipv4-gateway-aaaaa": { - "apiVersion": "discovery.k8s.io/v1", - "kind": "EndpointSlice", - "metadata": { - "name": "prod-ipv4-gateway-aaaaa", - "namespace": fn.REMOTE_NAMESPACE, - "labels": {"kubernetes.io/service-name": "prod-ipv4-gateway-aaaaa"}, - }, - "addressType": "IPv4", - "ports": [{"name": "https", "port": 443}], - "endpoints": [{"addresses": ["203.0.113.7"], "conditions": {"ready": True}}], - }, - "cluster-name-prod-ipv6-gateway-bbbbb": { - "apiVersion": "v1", - "kind": "Service", - "metadata": {"name": "prod-ipv6-gateway-bbbbb", "namespace": fn.REMOTE_NAMESPACE}, - "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, - }, - "cluster-name-slice-prod-ipv6-gateway-bbbbb": { - "apiVersion": "discovery.k8s.io/v1", - "kind": "EndpointSlice", - "metadata": { - "name": "prod-ipv6-gateway-bbbbb", - "namespace": fn.REMOTE_NAMESPACE, - "labels": {"kubernetes.io/service-name": "prod-ipv6-gateway-bbbbb"}, - }, - "addressType": "IPv6", - "ports": [{"name": "https", "port": 443}], - "endpoints": [{"addresses": ["2001:db8::1"], "conditions": {"ready": True}}], - }, - "cluster-name-prod-dns-gateway-ccccc": { - "apiVersion": "v1", - "kind": "Service", - "metadata": {"name": "prod-dns-gateway-ccccc", "namespace": fn.REMOTE_NAMESPACE}, - "spec": {"type": "ExternalName", "externalName": "lb-x.elb.amazonaws.com"}, - }, - }, - "IP clusters get a headless Service + EndpointSlice, the hostname cluster an ExternalName, " - "and the own cluster with nothing published gets neither", - ) +def test_an_ipv6_address_is_bracketed_in_the_endpoints() -> None: + """A bare IPv6 literal collides with the port separator in a URL, so an + SDK given http://2001:db8::1/v1 as a base_url can't use it.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), + resources={"gateway": _observed_gateway("2001:db8::1", ready=False)}, + ), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert resource.struct_to_dict(got.desired.composite.resource)["status"]["endpoints"] == { + "openAI": "http://[2001:db8::1]/v1", + "anthropic": "http://[2001:db8::1]/anthropic/v1", + } - async def test_certificate_common_names_fit_the_x509_limit(self) -> None: - """A long gateway name must not push a certificate commonName past the - 64-byte X.509 limit, which cert-manager's webhook rejects. A gateway name - is a cluster-scoped resource name, so it can be up to 253 characters.""" - long_name = "g" + "a" * 62 - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(name=long_name)))), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr(long_name, _CLUSTER)]), - ) - got = await self.runner.RunFunction(req, None) - manifest = resource.struct_to_dict(got.desired.resources["client-ca-certificate"].resource)["spec"][ - "forProvider" - ]["manifest"] - cn = manifest["spec"]["commonName"] - self.assertLessEqual(len(cn.encode()), 64, "client-ca-certificate commonName exceeds the 64-byte X.509 limit") - - async def test_a_rejected_caller_policy_is_not_ready(self) -> None: - """A gateway whose caller policy was rejected refuses every request with - a 500 while its Gateway still has an address. Envoy Gateway rejects the - policy when two selected Secrets share a key value, so this is reachable - by writing two Secrets.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth()))), - # The Gateway is programmed; the policy is not accepted. - resources={"gateway": _observed_gateway(_ADDRESS, ready=True)}, - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{"caller-secrets": [_secret("ml-team-keys", {"a": "sk-1"})]}, - ), - ) - got = await self.runner.RunFunction(req, None) - cond = next(iter(got.conditions)) - self.assertEqual(cond.status, fnv1.STATUS_CONDITION_FALSE) - self.assertEqual(cond.reason, fn.CONDITION_REASON_AUTH_NOT_ACCEPTED) - - async def test_a_shared_caller_key_is_not_ready_before_the_policy_is_observed(self) -> None: - """Envoy Gateway rejects the caller policy when two callers share a key, - but the policy's Object still reads as accepted until provider-kubernetes - next observes it. The gateway reports the outage from the Secrets - themselves, and skips a repeated caller name before comparing its key, - as Envoy Gateway does.""" - cases = [ - ( - "two Secrets share a key", - [_secret("team-a-keys", {"a": "sk-1"}), _secret("team-b-keys", {"b": "sk-1"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, - message="Caller team-b-keys/b has the same key as team-a-keys/a, so Envoy Gateway rejects the " - "caller authentication policy and every request is refused", - ), - ), - ( - "one Secret shares a key", - [_secret("team-a-keys", {"a": "sk-1", "z": "sk-1"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, - message="Caller team-a-keys/z has the same key as team-a-keys/a, so Envoy Gateway rejects the " - "caller authentication policy and every request is refused", - ), - ), - ( - # Listed out of order. Walked in name order, team-a's x is seen - # first, so team-b's x is skipped and its y repeats x's key. - # Walked as listed, team-a's x would be the skipped one. - "Secrets are walked in name order", - [_secret("team-b-keys", {"x": "sk-2", "y": "sk-1"}), _secret("team-a-keys", {"x": "sk-1"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, - message="Caller team-b-keys/y has the same key as team-a-keys/x, so Envoy Gateway rejects the " - "caller authentication policy and every request is refused", - ), - ), - ( - "a repeated caller name is skipped, whatever its key", - [_secret("team-a-keys", {"a": "sk-1"}), _secret("team-b-keys", {"a": "sk-1", "b": "sk-2"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_TRUE, - reason=fn.CONDITION_REASON_GATEWAY_PROGRAMMED, - ), - ), - ] - for name, secrets, want in cases: - with self.subTest(name): - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth()))), - resources={ - "gateway": _observed_gateway(_ADDRESS, ready=True), - "caller-auth": _observed_accepted(), - }, - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{"caller-secrets": secrets}, - ), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual( - [json_format.MessageToDict(c) for c in got.conditions], - [json_format.MessageToDict(want)], - ) - async def test_caller_secrets_are_listed_in_name_order(self) -> None: - """Envoy Gateway keeps the first Secret listed when two hold the same - caller name, so the policy lists them by name rather than in the order - they resolved in, and the winner doesn't change between reconciles.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth())))), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{ - "caller-secrets": [ - _secret("team-b-keys", {"b": "sk-2"}), - _secret("team-a-keys", {"a": "sk-1"}), - ] - }, - ), - ) - got = await self.runner.RunFunction(req, None) - policy = resource.struct_to_dict(got.desired.resources["caller-auth"].resource) - self.assertEqual( - policy["spec"]["forProvider"]["manifest"]["spec"]["apiKeyAuth"]["credentialRefs"], - [{"name": "callers-team-a-keys"}, {"name": "callers-team-b-keys"}], +def test_resolves_each_cluster_gateway_name() -> None: + """A Service per cluster gateway, resolving its name to its address here. + + A ModelService's backends address a cluster gateway by the name + compose-inference-cluster derived, and this gateway's Envoy resolves it, + so its cluster needs a Service of that name. An IP is served by a + headless Service and an EndpointSlice; a hostname, which is how a cloud + load balancer names itself, by an ExternalName Service. A cluster that + hasn't published both an address and a name gets neither. + """ + ipv4 = "prod-ipv4-gateway-aaaaa.modelplane-system.svc.cluster.local" + ipv6 = "prod-ipv6-gateway-bbbbb.modelplane-system.svc.cluster.local" + dns = "prod-dns-gateway-ccccc.modelplane-system.svc.cluster.local" + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + required_resources=_required( + gateways=[_gateway_xr("eu", _CLUSTER)], + clusters=[ + _cluster(), # this gateway's own cluster, no gateway published yet + _cluster_with_gateway("prod-ipv4", address="203.0.113.7", hostname=ipv4), + _cluster_with_gateway("prod-ipv6", address="2001:db8::1", hostname=ipv6), + _cluster_with_gateway("prod-dns", address="lb-x.elb.amazonaws.com", hostname=dns), + ], + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + resolvers = { + key: resource.struct_to_dict(res.resource) + for key, res in got.desired.resources.items() + if key.startswith("cluster-name") + } + for key, obj in resolvers.items(): + assert obj["spec"]["providerConfigRef"] == {"kind": "ClusterProviderConfig", "name": _PC}, ( + f"{key} is composed against this gateway's own cluster" ) + manifests = {key: obj["spec"]["forProvider"]["manifest"] for key, obj in resolvers.items()} + assert manifests == { + "cluster-name-prod-ipv4-gateway-aaaaa": { + "apiVersion": "v1", + "kind": "Service", + "metadata": {"name": "prod-ipv4-gateway-aaaaa", "namespace": fn.REMOTE_NAMESPACE}, + "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, + }, + "cluster-name-slice-prod-ipv4-gateway-aaaaa": { + "apiVersion": "discovery.k8s.io/v1", + "kind": "EndpointSlice", + "metadata": { + "name": "prod-ipv4-gateway-aaaaa", + "namespace": fn.REMOTE_NAMESPACE, + "labels": {"kubernetes.io/service-name": "prod-ipv4-gateway-aaaaa"}, + }, + "addressType": "IPv4", + "ports": [{"name": "https", "port": 443}], + "endpoints": [{"addresses": ["203.0.113.7"], "conditions": {"ready": True}}], + }, + "cluster-name-prod-ipv6-gateway-bbbbb": { + "apiVersion": "v1", + "kind": "Service", + "metadata": {"name": "prod-ipv6-gateway-bbbbb", "namespace": fn.REMOTE_NAMESPACE}, + "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, + }, + "cluster-name-slice-prod-ipv6-gateway-bbbbb": { + "apiVersion": "discovery.k8s.io/v1", + "kind": "EndpointSlice", + "metadata": { + "name": "prod-ipv6-gateway-bbbbb", + "namespace": fn.REMOTE_NAMESPACE, + "labels": {"kubernetes.io/service-name": "prod-ipv6-gateway-bbbbb"}, + }, + "addressType": "IPv6", + "ports": [{"name": "https", "port": 443}], + "endpoints": [{"addresses": ["2001:db8::1"], "conditions": {"ready": True}}], + }, + "cluster-name-prod-dns-gateway-ccccc": { + "apiVersion": "v1", + "kind": "Service", + "metadata": {"name": "prod-dns-gateway-ccccc", "namespace": fn.REMOTE_NAMESPACE}, + "spec": {"type": "ExternalName", "externalName": "lb-x.elb.amazonaws.com"}, + }, + }, ( + "IP clusters get a headless Service + EndpointSlice, the hostname cluster an ExternalName, " + "and the own cluster with nothing published gets neither" + ) - async def test_a_missing_caller_secret_denies_but_keeps_the_gateway(self) -> None: - """Auth is asked for but no caller Secret has resolved. The Gateway is - still composed, so its load balancer and address survive, and its caller - policy denies every request rather than leaving the door open. Two states - reach this, the selector matching no Secret and the requirement not having - resolved yet, differing only in the reason reported.""" - for name, extra, message in [ - ( - "the selector matches no Secret", - {"caller-secrets": []}, - "spec.auth.apiKey.secretSelector matches no Secret, so no caller could authenticate", - ), - ( - "the caller Secrets have not resolved yet", - {}, - "Waiting for caller key Secrets to resolve", - ), - ]: - with self.subTest(name): - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth()))) - ), - required_resources=_required( - clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)], **extra - ), - ) - got = await self.runner.RunFunction(req, None) - self.assertIn("gateway", got.desired.resources, "the Gateway is kept, so its address survives") - self.assertFalse( - any(key.startswith("caller-secret-") for key in got.desired.resources), - "no caller Secret resolved, so none is copied to the cluster", +def test_certificate_common_names_fit_the_x509_limit() -> None: + """A long gateway name must not push a certificate commonName past the + 64-byte X.509 limit, which cert-manager's webhook rejects. A gateway name + is a cluster-scoped resource name, so it can be up to 253 characters.""" + long_name = "g" + "a" * 62 + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(name=long_name)))), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr(long_name, _CLUSTER)]), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + manifest = resource.struct_to_dict(got.desired.resources["client-ca-certificate"].resource)["spec"]["forProvider"][ + "manifest" + ] + cn = manifest["spec"]["commonName"] + assert len(cn.encode()) <= 64, "client-ca-certificate commonName exceeds the 64-byte X.509 limit" + + +def test_a_rejected_caller_policy_is_not_ready() -> None: + """A gateway whose caller policy was rejected refuses every request with + a 500 while its Gateway still has an address. Envoy Gateway rejects the + policy when two selected Secrets share a key value, so this is reachable + by writing two Secrets.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth()))), + # The Gateway is programmed; the policy is not accepted. + resources={"gateway": _observed_gateway(_ADDRESS, ready=True)}, + ), + required_resources=_required( + clusters=[_cluster()], + gateways=[_gateway_xr("eu", _CLUSTER)], + **{"caller-secrets": [_secret("ml-team-keys", {"a": "sk-1"})]}, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + cond = next(iter(got.conditions)) + assert cond.status == fnv1.STATUS_CONDITION_FALSE + assert cond.reason == fn.CONDITION_REASON_AUTH_NOT_ACCEPTED + + +SHARED_CALLER_KEY_CASES = [ + ( + "two Secrets share a key", + [_secret("team-a-keys", {"a": "sk-1"}), _secret("team-b-keys", {"b": "sk-1"})], + fnv1.Condition( + type=fn.CONDITION_TYPE_GATEWAY_READY, + status=fnv1.STATUS_CONDITION_FALSE, + reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, + message="Caller team-b-keys/b has the same key as team-a-keys/a, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ), + ), + ( + "one Secret shares a key", + [_secret("team-a-keys", {"a": "sk-1", "z": "sk-1"})], + fnv1.Condition( + type=fn.CONDITION_TYPE_GATEWAY_READY, + status=fnv1.STATUS_CONDITION_FALSE, + reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, + message="Caller team-a-keys/z has the same key as team-a-keys/a, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ), + ), + ( + # Listed out of order. Walked in name order, team-a's x is seen + # first, so team-b's x is skipped and its y repeats x's key. + # Walked as listed, team-a's x would be the skipped one. + "Secrets are walked in name order", + [_secret("team-b-keys", {"x": "sk-2", "y": "sk-1"}), _secret("team-a-keys", {"x": "sk-1"})], + fnv1.Condition( + type=fn.CONDITION_TYPE_GATEWAY_READY, + status=fnv1.STATUS_CONDITION_FALSE, + reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, + message="Caller team-b-keys/y has the same key as team-a-keys/x, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ), + ), + ( + "a repeated caller name is skipped, whatever its key", + [_secret("team-a-keys", {"a": "sk-1"}), _secret("team-b-keys", {"a": "sk-1", "b": "sk-2"})], + fnv1.Condition( + type=fn.CONDITION_TYPE_GATEWAY_READY, + status=fnv1.STATUS_CONDITION_TRUE, + reason=fn.CONDITION_REASON_GATEWAY_PROGRAMMED, + ), + ), +] + + +@pytest.mark.parametrize("case", SHARED_CALLER_KEY_CASES, ids=lambda case: case[0]) +def test_a_shared_caller_key_is_not_ready_before_the_policy_is_observed( + case: tuple[str, list[dict], fnv1.Condition], +) -> None: + """Envoy Gateway rejects the caller policy when two callers share a key, + but the policy's Object still reads as accepted until provider-kubernetes + next observes it. The gateway reports the outage from the Secrets + themselves, and skips a repeated caller name before comparing its key, + as Envoy Gateway does.""" + _, secrets, want = case + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth()))), + resources={ + "gateway": _observed_gateway(_ADDRESS, ready=True), + "caller-auth": _observed_accepted(), + }, + ), + required_resources=_required( + clusters=[_cluster()], + gateways=[_gateway_xr("eu", _CLUSTER)], + **{"caller-secrets": secrets}, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert [_to_dict(c) for c in got.conditions] == [_to_dict(want)] + + +def test_caller_secrets_are_listed_in_name_order() -> None: + """Envoy Gateway keeps the first Secret listed when two hold the same + caller name, so the policy lists them by name rather than in the order + they resolved in, and the winner doesn't change between reconciles.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth())))), + required_resources=_required( + clusters=[_cluster()], + gateways=[_gateway_xr("eu", _CLUSTER)], + **{ + "caller-secrets": [ + _secret("team-b-keys", {"b": "sk-2"}), + _secret("team-a-keys", {"a": "sk-1"}), + ] + }, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + policy = resource.struct_to_dict(got.desired.resources["caller-auth"].resource) + assert policy["spec"]["forProvider"]["manifest"]["spec"]["apiKeyAuth"]["credentialRefs"] == [ + {"name": "callers-team-a-keys"}, + {"name": "callers-team-b-keys"}, + ] + + +MISSING_CALLER_SECRET_CASES = [ + ( + "the selector matches no Secret", + {"caller-secrets": []}, + "spec.auth.apiKey.secretSelector matches no Secret, so no caller could authenticate", + ), + ( + "the caller Secrets have not resolved yet", + {}, + "Waiting for caller key Secrets to resolve", + ), +] + + +@pytest.mark.parametrize("case", MISSING_CALLER_SECRET_CASES, ids=lambda case: case[0]) +def test_a_missing_caller_secret_denies_but_keeps_the_gateway(case: tuple[str, dict, str]) -> None: + """Auth is asked for but no caller Secret has resolved. The Gateway is + still composed, so its load balancer and address survive, and its caller + policy denies every request rather than leaving the door open. Two states + reach this, the selector matching no Secret and the requirement not having + resolved yet, differing only in the reason reported.""" + _, extra, want_message = case + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth())))), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)], **extra), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + assert "gateway" in got.desired.resources, "the Gateway is kept, so its address survives" + assert not any(key.startswith("caller-secret-") for key in got.desired.resources), ( + "no caller Secret resolved, so none is copied to the cluster" + ) + spec = resource.struct_to_dict(got.desired.resources["caller-auth"].resource)["spec"]["forProvider"]["manifest"][ + "spec" + ] + assert spec == { + "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": fn._GATEWAY_NAME}], + "authorization": {"defaultAction": "Deny"}, + }, "with no caller key the policy denies every request rather than authenticating nobody by omission" + cond = next(iter(got.conditions)) + assert cond.status == fnv1.STATUS_CONDITION_FALSE + assert cond.reason == fn.CONDITION_REASON_SECRETS_MISSING + assert cond.message == want_message + + +def test_a_missing_tls_secret_keeps_the_gateway() -> None: + """A referenced TLS Secret hasn't resolved. The Gateway is still composed, + so its address survives; the HTTPS listener is left without a certificate + on the cluster until the Secret appears, rather than the whole Gateway + withdrawn and its load balancer moved.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr(tls=v1alpha1.Tls(certificateRefs=[v1alpha1.CertificateRef(name="eu-tls-0")])) ) - spec = resource.struct_to_dict(got.desired.resources["caller-auth"].resource)["spec"]["forProvider"][ - "manifest" - ]["spec"] - self.assertEqual( - spec, + ) + ), + required_resources=_required( + clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)], **{"tls-secret-0": []} + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + assert "gateway" in got.desired.resources, "the Gateway is kept, so its address survives" + assert "tls-secret-eu-tls-0" not in got.desired.resources, "the missing Secret isn't copied to the cluster" + listeners = resource.struct_to_dict(got.desired.resources["gateway"].resource)["spec"]["forProvider"]["manifest"][ + "spec" + ]["listeners"] + assert [ln["name"] for ln in listeners] == ["http", "https"], "the HTTPS listener is still declared" + cond = next(iter(got.conditions)) + assert cond.status == fnv1.STATUS_CONDITION_FALSE + assert cond.reason == fn.CONDITION_REASON_SECRETS_MISSING + assert cond.message == "Waiting for TLS Secrets: eu-tls-0" + + +def test_the_incumbent_keeps_its_cluster() -> None: + """A gateway created later must not take a cluster off one already + serving traffic. Doing so would delete the incumbent's Gateway and bring + its load balancer back on a different address.""" + # "aaa" sorts before "zzz" but "zzz" already has an address. + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( { - "targetRefs": [ - {"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": fn._GATEWAY_NAME} - ], - "authorization": {"defaultAction": "Deny"}, - }, - "with no caller key the policy denies every request rather than authenticating nobody by omission", - ) - cond = next(iter(got.conditions)) - self.assertEqual(cond.status, fnv1.STATUS_CONDITION_FALSE) - self.assertEqual(cond.reason, fn.CONDITION_REASON_SECRETS_MISSING) - self.assertEqual(cond.message, message) - - async def test_a_missing_tls_secret_keeps_the_gateway(self) -> None: - """A referenced TLS Secret hasn't resolved. The Gateway is still composed, - so its address survives; the HTTPS listener is left without a certificate - on the cluster until the Secret appears, rather than the whole Gateway - withdrawn and its load balancer moved.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr(tls=v1alpha1.Tls(certificateRefs=[v1alpha1.CertificateRef(name="eu-tls-0")])) - ) + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceGateway", + "metadata": {"name": "aaa"}, + "spec": {"clusterName": _CLUSTER}, + } ) - ), - required_resources=_required( - clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)], **{"tls-secret-0": []} - ), - ) - got = await self.runner.RunFunction(req, None) - - self.assertIn("gateway", got.desired.resources, "the Gateway is kept, so its address survives") - self.assertNotIn("tls-secret-eu-tls-0", got.desired.resources, "the missing Secret isn't copied to the cluster") - listeners = resource.struct_to_dict(got.desired.resources["gateway"].resource)["spec"]["forProvider"][ - "manifest" - ]["spec"]["listeners"] - self.assertEqual([ln["name"] for ln in listeners], ["http", "https"], "the HTTPS listener is still declared") - cond = next(iter(got.conditions)) - self.assertEqual(cond.status, fnv1.STATUS_CONDITION_FALSE) - self.assertEqual(cond.reason, fn.CONDITION_REASON_SECRETS_MISSING) - self.assertEqual(cond.message, "Waiting for TLS Secrets: eu-tls-0") - - async def test_the_incumbent_keeps_its_cluster(self) -> None: - """A gateway created later must not take a cluster off one already - serving traffic. Doing so would delete the incumbent's Gateway and bring - its load balancer back on a different address.""" - # "aaa" sorts before "zzz" but "zzz" already has an address. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceGateway", - "metadata": {"name": "aaa"}, - "spec": {"clusterName": _CLUSTER}, - } + ) + ), + required_resources=_required( + clusters=[_cluster()], + gateways=[ + _gateway_xr("aaa", _CLUSTER), + {**_gateway_xr("zzz", _CLUSTER), "status": {"address": _ADDRESS}}, + ], + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert len(got.desired.resources) == 0, "the newcomer composes nothing" + cond = next(iter(got.conditions)) + assert cond.reason == fn.CONDITION_REASON_CLUSTER_TAKEN + assert "zzz" in cond.message + + +def test_no_composed_object_observes_a_secret() -> None: + """No composed Object reads a Secret, which is what keeps this gateway's + client CA private key off the control plane. + + provider-kubernetes copies an observed object's whole manifest into the + Object's status, and its --sanitize-secrets flag defaults to false, so + observing a Secret publishes every key in it to anyone who can get + objects. This CA signs the certificate every cluster gateway in the fleet + accepts, so leaking its key means anyone can reach any engine. + + Asserted over everything composed rather than over the PKI, because the + cost of reintroducing this anywhere is the same. + + Observing is the case that matters here. The Secrets this function + *writes* also end up in status, because provider-kubernetes reports what + it observes of what it manages, so this alone doesn't keep their contents + off the control plane. Those hold caller keys and serving certificates + that came from control-plane Secrets to begin with, so the exposure is a + wider audience for data already present rather than data that would + otherwise never be there, and prerequisites.yaml runs + provider-kubernetes with --sanitize-secrets to redact it. A CA private + key is different in kind: it is generated on the workload cluster and + observing it is the only way it could ever reach the control plane. + """ + # Auth and TLS both on, so the Secret-copying path is exercised: without + # them this function composes no Secret at all and the assertion holds + # vacuously. + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr( + tls={"certificateRefs": [{"name": "eu-tls-0"}]}, + auth={"method": "APIKey", "apiKey": {"secretSelector": {"matchLabels": {"team": "ml"}}}}, ) ) ), - required_resources=_required( - clusters=[_cluster()], - gateways=[ - _gateway_xr("aaa", _CLUSTER), - {**_gateway_xr("zzz", _CLUSTER), "status": {"address": _ADDRESS}}, - ], - ), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual(len(got.desired.resources), 0, "the newcomer composes nothing") - cond = next(iter(got.conditions)) - self.assertEqual(cond.reason, fn.CONDITION_REASON_CLUSTER_TAKEN) - self.assertIn("zzz", cond.message) - - async def test_no_composed_object_observes_a_secret(self) -> None: - """No composed Object reads a Secret, which is what keeps this gateway's - client CA private key off the control plane. - - provider-kubernetes copies an observed object's whole manifest into the - Object's status, and its --sanitize-secrets flag defaults to false, so - observing a Secret publishes every key in it to anyone who can get - objects. This CA signs the certificate every cluster gateway in the fleet - accepts, so leaking its key means anyone can reach any engine. - - Asserted over everything composed rather than over the PKI, because the - cost of reintroducing this anywhere is the same. - - Observing is the case that matters here. The Secrets this function - *writes* also end up in status, because provider-kubernetes reports what - it observes of what it manages, so this alone doesn't keep their contents - off the control plane. Those hold caller keys and serving certificates - that came from control-plane Secrets to begin with, so the exposure is a - wider audience for data already present rather than data that would - otherwise never be there, and prerequisites.yaml runs - provider-kubernetes with --sanitize-secrets to redact it. A CA private - key is different in kind: it is generated on the workload cluster and - observing it is the only way it could ever reach the control plane. - """ - # Auth and TLS both on, so the Secret-copying path is exercised: without - # them this function composes no Secret at all and the assertion holds - # vacuously. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr( - tls={"certificateRefs": [{"name": "eu-tls-0"}]}, - auth={"method": "APIKey", "apiKey": {"secretSelector": {"matchLabels": {"team": "ml"}}}}, - ) - ) - ), - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{ - "caller-secrets": [_secret("ml-team-keys", {"alice": "key"})], - "tls-secret-0": [_secret("eu-tls-0", {"tls.crt": "cert", "tls.key": "key"})], - }, - ), - ) - got = await self.runner.RunFunction(req, None) - - composed_secrets = [] - observed_secrets = [] - for key, res in got.desired.resources.items(): - d = resource.struct_to_dict(res.resource) - manifest = d["spec"]["forProvider"]["manifest"] - if manifest["kind"] != "Secret": - continue - composed_secrets.append(key) - if "Observe" in d["spec"].get("managementPolicies", []): - observed_secrets.append(key) - self.assertEqual(observed_secrets, [], "these observe a Secret, so its private keys reach the control plane") - self.assertNotEqual(composed_secrets, [], "no Secret composed, so the assertion above proves nothing") - - async def test_client_pki_publishes_the_ca_without_its_key(self) -> None: - """The client CA's certificate reaches the control plane through a - trust-manager Bundle, which copies one named key into a ConfigMap, rather - than through the Secret that also holds the private key.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = await self.runner.RunFunction(req, None) + ), + required_resources=_required( + clusters=[_cluster()], + gateways=[_gateway_xr("eu", _CLUSTER)], + **{ + "caller-secrets": [_secret("ml-team-keys", {"alice": "key"})], + "tls-secret-0": [_secret("eu-tls-0", {"tls.crt": "cert", "tls.key": "key"})], + }, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + composed_secrets = [] + observed_secrets = [] + for key, res in got.desired.resources.items(): + d = resource.struct_to_dict(res.resource) + manifest = d["spec"]["forProvider"]["manifest"] + if manifest["kind"] != "Secret": + continue + composed_secrets.append(key) + if "Observe" in d["spec"].get("managementPolicies", []): + observed_secrets.append(key) + assert observed_secrets == [], "these observe a Secret, so its private keys reach the control plane" + assert composed_secrets != [], "no Secret composed, so the assertion above proves nothing" + + +def test_client_pki_publishes_the_ca_without_its_key() -> None: + """The client CA's certificate reaches the control plane through a + trust-manager Bundle, which copies one named key into a ConfigMap, rather + than through the Secret that also holds the private key.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - def manifest(key: str) -> dict: - return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] + def manifest(key: str) -> dict: + return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] - self.assertEqual( - manifest("client-ca-bundle"), - { - "apiVersion": "trust.cert-manager.io/v1alpha1", - "kind": "Bundle", - "metadata": {"name": "inference-gateway-ca"}, - "spec": { - "sources": [{"secret": {"name": "inference-gateway-ca", "key": "ca.crt"}}], - "target": { - "configMap": {"key": "ca.crt"}, - "namespaceSelector": {"matchLabels": {"kubernetes.io/metadata.name": "modelplane-system"}}, - }, - }, - }, - ) - # Named after the Bundle, because that's the ConfigMap a Bundle syncs. - self.assertEqual( - manifest("client-ca-configmap"), - { - "apiVersion": "v1", - "kind": "ConfigMap", - "metadata": {"name": "inference-gateway-ca", "namespace": "modelplane-system"}, + assert manifest("client-ca-bundle") == { + "apiVersion": "trust.cert-manager.io/v1alpha1", + "kind": "Bundle", + "metadata": {"name": "inference-gateway-ca"}, + "spec": { + "sources": [{"secret": {"name": "inference-gateway-ca", "key": "ca.crt"}}], + "target": { + "configMap": {"key": "ca.crt"}, + "namespaceSelector": {"matchLabels": {"kubernetes.io/metadata.name": "modelplane-system"}}, }, - ) - self.assertEqual( - resource.struct_to_dict(got.desired.resources["client-ca-configmap"].resource)["spec"][ - "managementPolicies" - ], - ["Observe"], - "trust-manager owns this ConfigMap; Crossplane must not write it", - ) + }, + } + # Named after the Bundle, because that's the ConfigMap a Bundle syncs. + assert manifest("client-ca-configmap") == { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "inference-gateway-ca", "namespace": "modelplane-system"}, + } + assert resource.struct_to_dict(got.desired.resources["client-ca-configmap"].resource)["spec"][ + "managementPolicies" + ] == ["Observe"], "trust-manager owns this ConfigMap; Crossplane must not write it" + - async def test_client_ca_published_from_the_observed_configmap(self) -> None: - """status.clientCACertificate comes from the ConfigMap trust-manager - syncs, as plain text rather than base64. A cluster only trusts this - gateway once it has it, so nothing reaches an engine before it appears. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={ - "gateway": _observed_gateway("gw.example.org", ready=True), - "client-ca-configmap": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "status": { - "atProvider": { - "manifest": { - "apiVersion": "v1", - "kind": "ConfigMap", - "data": {"ca.crt": "-----BEGIN CERTIFICATE-----\nclient\n"}, - } +def test_client_ca_published_from_the_observed_configmap() -> None: + """status.clientCACertificate comes from the ConfigMap trust-manager + syncs, as plain text rather than base64. A cluster only trusts this + gateway once it has it, so nothing reaches an engine before it appears. + """ + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), + resources={ + "gateway": _observed_gateway("gw.example.org", ready=True), + "client-ca-configmap": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "status": { + "atProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "data": {"ca.crt": "-----BEGIN CERTIFICATE-----\nclient\n"}, } - }, - } - ), + } + }, + } ), - }, - ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = await self.runner.RunFunction(req, None) + ), + }, + ), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - self.assertEqual( - resource.struct_to_dict(got.desired.composite.resource)["status"]["clientCACertificate"], - "-----BEGIN CERTIFICATE-----\nclient\n", - ) + assert ( + resource.struct_to_dict(got.desired.composite.resource)["status"]["clientCACertificate"] + == "-----BEGIN CERTIFICATE-----\nclient\n" + ) - async def test_no_client_ca_before_the_bundle_syncs(self) -> None: - """With no observed ConfigMap the gateway publishes no CA, so no cluster - trusts it yet and no cluster publishes a hostname on its account.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={"gateway": _observed_gateway("gw.example.org", ready=True)}, - ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = await self.runner.RunFunction(req, None) - self.assertNotIn( - "clientCACertificate", - resource.struct_to_dict(got.desired.composite.resource)["status"], - ) +def test_no_client_ca_before_the_bundle_syncs() -> None: + """With no observed ConfigMap the gateway publishes no CA, so no cluster + trusts it yet and no cluster publishes a hostname on its account.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), + resources={"gateway": _observed_gateway("gw.example.org", ready=True)}, + ), + required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + assert "clientCACertificate" not in resource.struct_to_dict(got.desired.composite.resource)["status"] diff --git a/functions/compose-model-cache/tests/test_fn.py b/functions/compose-model-cache/tests/test_fn.py index 33e460630..b1f607ff5 100644 --- a/functions/compose-model-cache/tests/test_fn.py +++ b/functions/compose-model-cache/tests/test_fn.py @@ -14,16 +14,18 @@ """Tests for the compose-model-cache function.""" +import asyncio import dataclasses import datetime -import unittest +import json from typing import Any -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.modelcache import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -41,10 +43,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - # The XR used across cases: a HuggingFace ModelCache in the ml-team namespace. # Both the PVC and Job derive their names from # resource.child_name("modelcache", "ml-team", "qwen", ...). @@ -292,607 +290,597 @@ def _auth_object(pc: str) -> dict: } -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: # noqa: PLR0915 - """The function composes a ModelCache.""" - # --- Case 1: GKE cluster, first pass. Composes the RWX PVC + hydration - # Job per matched cluster; nothing observed yet so phase is Pending and - # ArtifactReady is Hydrating. Emits the one-time "Staging" event. --- - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Pending"}], - }, +def _compose_cases() -> list[Case]: + """The cases for test_compose, built by code that derives them from shared parts.""" + # --- Case 1: GKE cluster, first pass. Composes the RWX PVC + hydration + # Job per matched cluster; nothing observed yet so phase is Pending and + # ArtifactReady is Hydrating. Emits the one-time "Staging" event. --- + want1 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "0/1"}, + "clusters": [{"name": "cluster-a", "phase": "Pending"}], }, - ), + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), - "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), - }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - want1.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 2: GKE cluster with a pinned revision + auth secret. The Job - # command gains --revision, and the function propagates the token to a - # workload-cluster Secret (auth-cluster-a) whose name the Job's HF_TOKEN - # env references - not the user's control-plane Secret name. --- - xr2 = _cache_xr(revision="main", authSecret=v1alpha1.AuthSecret(name="hf-token")) - env2 = [ - { - "name": "HF_TOKEN", - "valueFrom": {"secretKeyRef": {"name": _AUTH_NAME, "key": "HF_TOKEN"}}, + resources={ + "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), + "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), }, - ] - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Pending"}], - }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + ) + want1.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + + # --- Case 2: GKE cluster with a pinned revision + auth secret. The Job + # command gains --revision, and the function propagates the token to a + # workload-cluster Secret (auth-cluster-a) whose name the Job's HF_TOKEN + # env references - not the user's control-plane Secret name. --- + xr2 = _cache_xr(revision="main", authSecret=v1alpha1.AuthSecret(name="hf-token")) + env2 = [ + { + "name": "HF_TOKEN", + "valueFrom": {"secretKeyRef": {"name": _AUTH_NAME, "key": "HF_TOKEN"}}, + }, + ] + want2 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "0/1"}, + "clusters": [{"name": "cluster-a", "phase": "Pending"}], }, - ), + }, ), - resources={ - "auth-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_auth_object("cluster-a-pc"))), - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), - "hydrate-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct( - _job_object("cluster-a-pc", command=_HYDRATE_CMD_REVISION, env=env2), - ), - ), - }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - want2.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want2.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 3: EKS cluster reporting an EFS RWX class on status.cache. The - # PVC sources its storageClassName from status.cache (modelplane-rwx-efs), - # not the GKE/Filestore one. --- - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + resources={ + "auth-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_auth_object("cluster-a-pc"))), + "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), + "hydrate-cluster-a": fnv1.Resource( resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "eks-a", "phase": "Pending"}], - }, - }, + _job_object("cluster-a-pc", command=_HYDRATE_CMD_REVISION, env=env2), ), ), - resources={ - "pvc-eks-a": fnv1.Resource( - resource=resource.dict_to_struct( - _pvc_object("eks-a-pc", storage_class="modelplane-rwx-efs"), - ), - ), - "hydrate-eks-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("eks-a-pc"))), - }, + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: eks-a", - ), - ], - context=structpb.Struct(), - ) - want3.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 4: ready. The observed PVC + Job Objects each carry their own - # Ready condition (from DeriveFromCelQuery), and the wrapped manifest - # status shows PVC Bound + Job succeeded. Phase Ready, both Objects - # marked ready, summary 1/1, XR ready, ArtifactReady Staged. The - # already-composed PVC suppresses the "Staging" event; the - # not-previously-ready -> ready transition emits the "staged" event. --- - observed4 = { - "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), - } - pvc_ready = _pvc_object("cluster-a-pc") - want4 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, + ], + context=structpb.Struct(), + ) + want2.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + want2.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) + + # --- Case 3: EKS cluster reporting an EFS RWX class on status.cache. The + # PVC sources its storageClassName from status.cache (modelplane-rwx-efs), + # not the GKE/Filestore one. --- + want3 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "0/1"}, + "clusters": [{"name": "eks-a", "phase": "Pending"}], }, - ), - ready=fnv1.READY_TRUE, + }, ), - resources={ - # Job dropped once Ready; only the PVC remains composed. - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(pvc_ready), ready=fnv1.READY_TRUE), - }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - results=[ - fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), - ], - context=structpb.Struct(), - ) - want4.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 5: hydrating. PVC Bound (Object Ready) but the Job hasn't - # completed, so phase is Hydrating, only the PVC is marked ready, summary - # 0/1, and the XR is not ready. No transition event fires. --- - observed5 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} - want5 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + resources={ + "pvc-eks-a": fnv1.Resource( resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Hydrating"}], - }, - }, + _pvc_object("eks-a-pc", storage_class="modelplane-rwx-efs"), ), ), - resources={ - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE - ), - "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), - }, + "hydrate-eks-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("eks-a-pc"))), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: eks-a", ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - context=structpb.Struct(), - ) - want5.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 6: failed. The Job reports a Failed condition (and is NOT - # Ready). A Failed Job takes precedence over PVC binding, so phase is - # Failed, only the PVC is marked ready, summary 0/1, XR not ready, and - # ArtifactReady is False with reason Failed. --- - observed6 = { - "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Failed", "status": "True"}]}), - } - want6 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Failed"}], - }, + ], + context=structpb.Struct(), + ) + want3.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + + # --- Case 4: ready. The observed PVC + Job Objects each carry their own + # Ready condition (from DeriveFromCelQuery), and the wrapped manifest + # status shows PVC Bound + Job succeeded. Phase Ready, both Objects + # marked ready, summary 1/1, XR ready, ArtifactReady Staged. The + # already-composed PVC suppresses the "Staging" event; the + # not-previously-ready -> ready transition emits the "staged" event. --- + observed4 = { + "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), + "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), + } + pvc_ready = _pvc_object("cluster-a-pc") + want4 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "1/1"}, + "clusters": [{"name": "cluster-a", "phase": "Ready"}], }, - ), + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE - ), - "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), - }, + ready=fnv1.READY_TRUE, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Failed"), - ], - context=structpb.Struct(), - ) - want6.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 7: partial (1/2). Cluster a is Ready (PVC Bound + Job - # succeeded), cluster b is still Hydrating (PVC Bound only). Summary - # 1/2, ArtifactReady False with reason Partial, XR not ready. --- - observed7 = { - "pvc-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), - "pvc-b": _observed_object({"phase": "Bound"}, ready=True), - } - want7 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/2"}, - "clusters": [ - {"name": "a", "phase": "Ready"}, - {"name": "b", "phase": "Hydrating"}, - ], - }, + resources={ + # Job dropped once Ready; only the PVC remains composed. + "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(pvc_ready), ready=fnv1.READY_TRUE), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), + ], + context=structpb.Struct(), + ) + want4.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + + # --- Case 5: hydrating. PVC Bound (Object Ready) but the Job hasn't + # completed, so phase is Hydrating, only the PVC is marked ready, summary + # 0/1, and the XR is not ready. No transition event fires. --- + observed5 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} + want5 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "0/1"}, + "clusters": [{"name": "cluster-a", "phase": "Hydrating"}], }, - ), + }, ), - resources={ - # Cluster a is Ready, so its Job is dropped; b is still Hydrating. - "pvc-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("a-pc")), ready=fnv1.READY_TRUE - ), - "pvc-b": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("b-pc")), ready=fnv1.READY_TRUE - ), - "hydrate-b": fnv1.Resource(resource=resource.dict_to_struct(_job_object("b-pc"))), - }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Partial"), - ], - context=structpb.Struct(), - ) - want7.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 8: latch. A previously-Ready cluster whose hydration Job was - # dropped (and TTL-cleaned), so only the PVC is observed now. The status - # latch keeps phase Ready and the Job is not re-composed, PVC marked - # ready, summary 1/1, XR ready. Already-ready, so no transition event. --- - xr_ready = _cache_xr() - xr_ready.status = v1alpha1.Status( - summary=v1alpha1.Summary(ready="1/1"), - clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], - conditions=[ - v1alpha1.Condition( - type="Ready", - status="True", - reason="Available", - lastTransitionTime=_TRANSITION_TIME, + resources={ + "pvc-cluster-a": fnv1.Resource( + resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE ), - ], - ) - observed8 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} - want8 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, + "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + context=structpb.Struct(), + ) + want5.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + + # --- Case 6: failed. The Job reports a Failed condition (and is NOT + # Ready). A Failed Job takes precedence over PVC binding, so phase is + # Failed, only the PVC is marked ready, summary 0/1, XR not ready, and + # ArtifactReady is False with reason Failed. --- + observed6 = { + "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), + "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Failed", "status": "True"}]}), + } + want6 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "0/1"}, + "clusters": [{"name": "cluster-a", "phase": "Failed"}], }, - ), - ready=fnv1.READY_TRUE, + }, ), - resources={ - # Latched Ready with the Job already dropped: only the PVC. - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE - ), - }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - context=structpb.Struct(), - ) - want8.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 9: authSecret referenced but not yet resolved. The function - # requires both the clusters and the auth Secret, then returns early - # (no resources, status, or conditions) until Crossplane resolves the - # Secret and re-calls it. --- - xr9 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - want9 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(), - context=structpb.Struct(), - ) - want9.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want9.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 10: authSecret resolved but the Secret lacks the referenced - # key (here it carries OTHER, not HF_TOKEN). The PVC still composes - it - # doesn't depend on the token, so a cache isn't pruned for a missing one - # - but the hydration Job and token Secret are held back. ArtifactReady - # is False with reason AuthSecretMissing, and a warning names the Secret - # and key so the user can fix it instead of seeing the XR stall. The XR - # is marked not ready, since the PVC alone would make it ready once it - # binds. --- - want10 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Pending"}], - }, + resources={ + "pvc-cluster-a": fnv1.Resource( + resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE + ), + "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Failed"), + ], + context=structpb.Struct(), + ) + want6.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + + # --- Case 7: partial (1/2). Cluster a is Ready (PVC Bound + Job + # succeeded), cluster b is still Hydrating (PVC Bound only). Summary + # 1/2, ArtifactReady False with reason Partial, XR not ready. --- + observed7 = { + "pvc-a": _observed_object({"phase": "Bound"}, ready=True), + "hydrate-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), + "pvc-b": _observed_object({"phase": "Bound"}, ready=True), + } + want7 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "1/2"}, + "clusters": [ + {"name": "a", "phase": "Ready"}, + {"name": "b", "phase": "Hydrating"}, + ], }, - ), - ready=fnv1.READY_FALSE, + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), - }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="AuthSecretMissing"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_WARNING, - message="authSecret ml-team/hf-token is missing or has no key 'HF_TOKEN'", - ), - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - want10.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want10.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 11: Ready cluster with an authSecret. The token is only needed - # while hydrating, so once the cluster is Ready the auth Secret is dropped - # alongside the Job (only the PVC remains composed), even though the - # control-plane Secret still resolves. Keeps the token from lingering on - # the inference cluster after hydration. --- - xr11 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - observed11 = { - "auth-cluster-a": _observed_object({}, ready=True), - "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), - } - want11 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, + resources={ + # Cluster a is Ready, so its Job is dropped; b is still Hydrating. + "pvc-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("a-pc")), ready=fnv1.READY_TRUE), + "pvc-b": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("b-pc")), ready=fnv1.READY_TRUE), + "hydrate-b": fnv1.Resource(resource=resource.dict_to_struct(_job_object("b-pc"))), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Partial"), + ], + context=structpb.Struct(), + ) + want7.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + + # --- Case 8: latch. A previously-Ready cluster whose hydration Job was + # dropped (and TTL-cleaned), so only the PVC is observed now. The status + # latch keeps phase Ready and the Job is not re-composed, PVC marked + # ready, summary 1/1, XR ready. Already-ready, so no transition event. --- + xr_ready = _cache_xr() + xr_ready.status = v1alpha1.Status( + summary=v1alpha1.Summary(ready="1/1"), + clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], + conditions=[ + v1alpha1.Condition( + type="Ready", + status="True", + reason="Available", + lastTransitionTime=_TRANSITION_TIME, + ), + ], + ) + observed8 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} + want8 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "1/1"}, + "clusters": [{"name": "cluster-a", "phase": "Ready"}], }, - ), - ready=fnv1.READY_TRUE, + }, ), - resources={ - # Ready: the auth Secret and Job are both dropped, only the PVC remains. - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE - ), - }, + ready=fnv1.READY_TRUE, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - results=[ - fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), - ], - context=structpb.Struct(), - ) - want11.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want11.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 12: token rotated away after a cache is Ready. A latched-Ready - # cluster whose authSecret now resolves without the key. The PVC keeps - # composing (and stays Ready via the status latch) rather than being - # pruned, and because hydration is already done the missing token is - # neither reported (ArtifactReady stays Staged) nor warned. --- - xr12 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - xr12.status = v1alpha1.Status( - summary=v1alpha1.Summary(ready="1/1"), - clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], - conditions=[ - v1alpha1.Condition( - type="Ready", - status="True", - reason="Available", - lastTransitionTime=_TRANSITION_TIME, + resources={ + # Latched Ready with the Job already dropped: only the PVC. + "pvc-cluster-a": fnv1.Resource( + resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE ), - ], - ) - observed12 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} - want12 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + context=structpb.Struct(), + ) + want8.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + + # --- Case 9: authSecret referenced but not yet resolved. The function + # requires both the clusters and the auth Secret, then returns early + # (no resources, status, or conditions) until Crossplane resolves the + # Secret and re-calls it. --- + xr9 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) + want9 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(), + context=structpb.Struct(), + ) + want9.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + want9.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) + + # --- Case 10: authSecret resolved but the Secret lacks the referenced + # key (here it carries OTHER, not HF_TOKEN). The PVC still composes - it + # doesn't depend on the token, so a cache isn't pruned for a missing one + # - but the hydration Job and token Secret are held back. ArtifactReady + # is False with reason AuthSecretMissing, and a warning names the Secret + # and key so the user can fix it instead of seeing the XR stall. The XR + # is marked not ready, since the PVC alone would make it ready once it + # binds. --- + want10 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "0/1"}, + "clusters": [{"name": "cluster-a", "phase": "Pending"}], }, - ), - ready=fnv1.READY_TRUE, - ), - resources={ - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE - ), - }, - ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - context=structpb.Struct(), - ) - want12.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want12.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 13: authSecret missing AND no clusters matched. NoClusters is - # the dominant signal - the cache can't progress regardless of the token - # - so both conditions report NoClusters and the missing token is neither - # reported nor warned. --- - xr13 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - want13 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"summary": {"ready": "0/0"}, "clusters": []}}), - ready=fnv1.READY_FALSE, + }, ), + ready=fnv1.READY_FALSE, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), - ], - context=structpb.Struct(), - ) - want13.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want13.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - cases = [ - Case( - name="GKE cluster first pass composes RWX PVC and hydration Job", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")]), - want=want1, - ), - Case( - name="HuggingFace revision and auth secret wire --revision and HF_TOKEN", - req=_req(xr2, [_cluster_dict("cluster-a", "cluster-a-pc")], auth=_auth_secret()), - want=want2, - ), - Case( - name="EKS cluster PVC sources the EFS class from status.cache", - req=_req(_cache_xr(), [_cluster_dict("eks-a", "eks-a-pc", source="EKS")]), - want=want3, - ), - Case( - name="PVC bound and Job complete reports Ready", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed4), - want=want4, - ), - Case( - name="PVC bound but Job running reports Hydrating", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed5), - want=want5, - ), - Case( - name="failed Job reports Failed and takes precedence over PVC binding", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed6), - want=want6, - ), - Case( - name="one of two clusters ready reports partial", - req=_req(_cache_xr(), [_cluster_dict("a", "a-pc"), _cluster_dict("b", "b-pc")], observed7), - want=want7, - ), - Case( - name="hydrated cluster stays Ready after its Job is TTL-cleaned", - req=_req(xr_ready, [_cluster_dict("cluster-a", "cluster-a-pc")], observed8), - want=want8, + resources={ + "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="AuthSecretMissing"), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="authSecret ml-team/hf-token is missing or has no key 'HF_TOKEN'", ), - Case( - name="authSecret unresolved requires it and returns early", - req=_req(xr9, [_cluster_dict("cluster-a", "cluster-a-pc")]), - want=want9, + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", ), - Case( - name="authSecret resolved without the referenced key composes PVC and warns", - req=_req( - xr9, - [_cluster_dict("cluster-a", "cluster-a-pc")], - auth=_auth_secret(data={"OTHER": _TOKEN_B64}), + ], + context=structpb.Struct(), + ) + want10.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + want10.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) + + # --- Case 11: Ready cluster with an authSecret. The token is only needed + # while hydrating, so once the cluster is Ready the auth Secret is dropped + # alongside the Job (only the PVC remains composed), even though the + # control-plane Secret still resolves. Keeps the token from lingering on + # the inference cluster after hydration. --- + xr11 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) + observed11 = { + "auth-cluster-a": _observed_object({}, ready=True), + "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), + "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), + } + want11 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "1/1"}, + "clusters": [{"name": "cluster-a", "phase": "Ready"}], + }, + }, ), - want=want10, + ready=fnv1.READY_TRUE, ), - Case( - name="authSecret resolved with an empty token value composes PVC and warns", - req=_req( - xr9, - [_cluster_dict("cluster-a", "cluster-a-pc")], - auth=_auth_secret(data={"HF_TOKEN": ""}), + resources={ + # Ready: the auth Secret and Job are both dropped, only the PVC remains. + "pvc-cluster-a": fnv1.Resource( + resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE ), - want=want10, + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), + ], + context=structpb.Struct(), + ) + want11.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + want11.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) + + # --- Case 12: token rotated away after a cache is Ready. A latched-Ready + # cluster whose authSecret now resolves without the key. The PVC keeps + # composing (and stays Ready via the status latch) rather than being + # pruned, and because hydration is already done the missing token is + # neither reported (ArtifactReady stays Staged) nor warned. --- + xr12 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) + xr12.status = v1alpha1.Status( + summary=v1alpha1.Summary(ready="1/1"), + clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], + conditions=[ + v1alpha1.Condition( + type="Ready", + status="True", + reason="Available", + lastTransitionTime=_TRANSITION_TIME, ), - Case( - name="Ready cluster drops the auth Secret with the Job", - req=_req( - xr11, - [_cluster_dict("cluster-a", "cluster-a-pc")], - observed11, - auth=_auth_secret(), + ], + ) + observed12 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} + want12 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "summary": {"ready": "1/1"}, + "clusters": [{"name": "cluster-a", "phase": "Ready"}], + }, + }, ), - want=want11, + ready=fnv1.READY_TRUE, ), - Case( - name="token rotated away after Ready keeps the PVC and stays Ready", - req=_req( - xr12, - [_cluster_dict("cluster-a", "cluster-a-pc")], - observed12, - auth=_auth_secret(data={"OTHER": _TOKEN_B64}), + resources={ + "pvc-cluster-a": fnv1.Resource( + resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE ), - want=want12, + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + context=structpb.Struct(), + ) + want12.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + want12.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) + + # --- Case 13: authSecret missing AND no clusters matched. NoClusters is + # the dominant signal - the cache can't progress regardless of the token + # - so both conditions report NoClusters and the missing token is neither + # reported nor warned. --- + xr13 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) + want13 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"summary": {"ready": "0/0"}, "clusters": []}}), + ready=fnv1.READY_FALSE, + ), + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), + ], + context=structpb.Struct(), + ) + want13.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) + want13.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) + + return [ + Case( + name="GKE cluster first pass composes RWX PVC and hydration Job", + req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")]), + want=want1, + ), + Case( + name="HuggingFace revision and auth secret wire --revision and HF_TOKEN", + req=_req(xr2, [_cluster_dict("cluster-a", "cluster-a-pc")], auth=_auth_secret()), + want=want2, + ), + Case( + name="EKS cluster PVC sources the EFS class from status.cache", + req=_req(_cache_xr(), [_cluster_dict("eks-a", "eks-a-pc", source="EKS")]), + want=want3, + ), + Case( + name="PVC bound and Job complete reports Ready", + req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed4), + want=want4, + ), + Case( + name="PVC bound but Job running reports Hydrating", + req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed5), + want=want5, + ), + Case( + name="failed Job reports Failed and takes precedence over PVC binding", + req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed6), + want=want6, + ), + Case( + name="one of two clusters ready reports partial", + req=_req(_cache_xr(), [_cluster_dict("a", "a-pc"), _cluster_dict("b", "b-pc")], observed7), + want=want7, + ), + Case( + name="hydrated cluster stays Ready after its Job is TTL-cleaned", + req=_req(xr_ready, [_cluster_dict("cluster-a", "cluster-a-pc")], observed8), + want=want8, + ), + Case( + name="authSecret unresolved requires it and returns early", + req=_req(xr9, [_cluster_dict("cluster-a", "cluster-a-pc")]), + want=want9, + ), + Case( + name="authSecret resolved without the referenced key composes PVC and warns", + req=_req( + xr9, + [_cluster_dict("cluster-a", "cluster-a-pc")], + auth=_auth_secret(data={"OTHER": _TOKEN_B64}), ), - Case( - name="authSecret missing with no clusters reports NoClusters not AuthSecretMissing", - req=_req(xr13, [], auth=_auth_secret(data={"OTHER": _TOKEN_B64})), - want=want13, + want=want10, + ), + Case( + name="authSecret resolved with an empty token value composes PVC and warns", + req=_req( + xr9, + [_cluster_dict("cluster-a", "cluster-a-pc")], + auth=_auth_secret(data={"HF_TOKEN": ""}), ), - ] + want=want10, + ), + Case( + name="Ready cluster drops the auth Secret with the Job", + req=_req( + xr11, + [_cluster_dict("cluster-a", "cluster-a-pc")], + observed11, + auth=_auth_secret(), + ), + want=want11, + ), + Case( + name="token rotated away after Ready keeps the PVC and stays Ready", + req=_req( + xr12, + [_cluster_dict("cluster-a", "cluster-a-pc")], + observed12, + auth=_auth_secret(data={"OTHER": _TOKEN_B64}), + ), + want=want12, + ), + Case( + name="authSecret missing with no clusters reports NoClusters not AuthSecretMissing", + req=_req(xr13, [], auth=_auth_secret(data={"OTHER": _TOKEN_B64})), + want=want13, + ), + ] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes a ModelCache's PVC and hydration Job per cluster and reports their progress.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-model-deployment/tests/__init__.py b/functions/compose-model-deployment/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-model-deployment/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-model-deployment/tests/test_cel.py b/functions/compose-model-deployment/tests/test_cel.py index 9b4979755..7a2e3fd9b 100644 --- a/functions/compose-model-deployment/tests/test_cel.py +++ b/functions/compose-model-deployment/tests/test_cel.py @@ -21,8 +21,8 @@ """ import dataclasses -import unittest +import pytest from function import cel @@ -86,237 +86,232 @@ class Case: ) -class TestMatches(unittest.TestCase): - def test_matches(self) -> None: - cases = [ - # driver. - Case(name="driver equals", expr='device.driver == "gpu.nvidia.com"', device=_GPU, want=True), - Case(name="driver not equals", expr='device.driver == "nic.nvidia.com"', device=_GPU, want=False), - # Quantity comparison + methods. - Case( - name="quantity compareTo ge", - expr=f'{_CAP}.memory.compareTo(quantity("141Gi")) >= 0', - device=_GPU, - want=True, - ), - Case( - name="quantity compareTo too big", - expr=f'{_CAP}.memory.compareTo(quantity("200Gi")) >= 0', - device=_GPU, - want=False, - ), - Case( - name="quantity isGreaterThan", - expr=f'{_CAP}.memory.isGreaterThan(quantity("80Gi"))', - device=_GPU, - want=True, - ), - Case( - name="quantity isLessThan", expr=f'{_CAP}.memory.isLessThan(quantity("200Gi"))', device=_GPU, want=True - ), - Case(name="quantity sign", expr=f"{_CAP}.memory.sign() == 1", device=_GPU, want=True), - Case(name="quantity asInteger", expr=f"{_CAP}.memory.asInteger() == {141 * 2**30}", device=_GPU, want=True), - Case(name="quantity isInteger", expr=f"{_CAP}.memory.isInteger()", device=_GPU, want=True), - Case( - name="quantity add", - expr=f'{_CAP}.memory.add(quantity("1Gi")).compareTo(quantity("142Gi")) == 0', - device=_GPU, - want=True, - ), - Case(name="isQuantity true", expr='isQuantity("1.3Gi")', device=_GPU, want=True), - Case(name="isQuantity false", expr='isQuantity("200K")', device=_GPU, want=False), - # Semver comparison + methods. - Case( - name="semver isGreaterThan", - expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.0.0"))', - device=_GPU, - want=True, - ), - Case( - name="semver not greater", - expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.9.0"))', - device=_GPU, - want=False, - ), - Case(name="semver major", expr=f"{_ATTR}.cudaComputeCapability.major() == 9", device=_GPU, want=True), - Case(name="semver minor", expr=f"{_ATTR}.cudaComputeCapability.minor() == 5", device=_GPU, want=True), - Case(name="semver patch", expr=f"{_ATTR}.cudaComputeCapability.patch() == 3", device=_GPU, want=True), - Case( - name="semver equality", expr=f'{_ATTR}.cudaComputeCapability == semver("9.5.3")', device=_GPU, want=True - ), - Case(name="isSemver strict true", expr='isSemver("1.0.0")', device=_GPU, want=True), - Case(name="isSemver strict rejects short", expr='isSemver("1.0")', device=_GPU, want=False), - Case(name="isSemver normalize accepts short", expr='isSemver("1.0", true)', device=_GPU, want=True), - Case(name="semver normalize overload", expr='semver("v1.0", true).major() == 1', device=_GPU, want=True), - # Typed scalar attributes (resolve straight to the value, no .string). - Case(name="string attribute", expr=f'{_ATTR}.architecture == "Hopper"', device=_GPU, want=True), - Case( - name="string attribute mismatch", - expr='device.attributes["nic.nvidia.com"].linkType == "infiniband"', - device=_NIC, - want=True, - ), - Case( - name="bool attribute true", - expr=f"{_ATTR}.x", - device=_device(attributes={"x": {"bool": True}}), - want=True, - ), - Case( - name="bool attribute false", - expr=f"{_ATTR}.x", - device=_device(attributes={"x": {"bool": False}}), - want=False, - ), - Case( - name="int attribute", - expr=f"{_ATTR}.x >= 8", - device=_device(attributes={"x": {"int": 8}}), - want=True, - ), - Case( - name="int attribute below", - expr=f"{_ATTR}.x >= 8", - device=_device(attributes={"x": {"int": 4}}), - want=False, - ), - # Qualified names split into their own domain. - Case( - name="qualified name under its domain", - expr='device.attributes["resource.kubernetes.io"].pcieRoot == "pci0"', - device=_GPU, - want=True, - ), - Case( - name="bare name under driver domain", expr=f'{_ATTR}.architecture == "Hopper"', device=_GPU, want=True - ), - # Non-matches that must not raise. - Case( - name="two-component version is non-match", - expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("8.0.0"))', - device=_device(attributes={"cudaComputeCapability": {"version": "9.0"}}), - want=False, - ), - Case( - name="malformed quantity is non-match", - expr=f'{_CAP}.memory.compareTo(quantity("1Gi")) >= 0', - device=_device(capacity={"memory": {"value": "10Mo"}}), - want=False, - ), - Case(name="unknown id is non-match", expr=f'{_ATTR}.nope == "x"', device=_GPU, want=False), - # A non-bool selector must not spuriously match. Upstream rejects it - # at compile time; we treat a non-bool result as a non-match. - Case(name="non-bool string selector is non-match", expr='"5"', device=_GPU, want=False), - Case( - name="non-bool int selector is non-match", - expr=f"{_ATTR}.x", - device=_device(attributes={"x": {"int": 5}}), - want=False, - ), - # Domain presence. Upstream's domain-presence idiom is "" in - # device.attributes, not has(device.attributes[""]): cel-go's - # has() macro rejects an index argument, so the has() form is a - # compile error on a real cluster (celpy accepts it - see cel.py's - # documented divergences). An unknown domain is simply absent (False), - # not present-but-empty. - Case(name="unknown domain absent", expr='"other.com" in device.attributes', device=_GPU, want=False), - Case(name="known domain present", expr='"gpu.nvidia.com" in device.attributes', device=_GPU, want=True), - # Reading an unknown domain resolves to an empty map (not an error), - # so an id lookup under it is a non-match rather than a failure. - Case( - name="unknown domain id is non-match", - expr='device.attributes["other.com"].x == "y"', - device=_GPU, - want=False, - ), - # Guard a domain read with the in idiom before indexing it. - Case( - name="guarded known domain", - expr=f'"gpu.nvidia.com" in device.attributes && {_ATTR}.architecture == "Hopper"', - device=_GPU, - want=True, - ), - # The full design selector. - Case( - name="full design expression", - expr=( - f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.0.0")) && ' - f'{_CAP}.memory.compareTo(quantity("141Gi")) >= 0' - ), - device=_GPU, - want=True, - ), - # Verbatim selector examples from the DRA docs, each against a device - # that should and should not match. - Case( - name="docs: large-black subrequest matches", - expr=( - 'device.attributes["resource-driver.example.com"].color == "black" && ' - 'device.attributes["resource-driver.example.com"].size == "large"' - ), - device=_LARGE_BLACK, - want=True, - ), - Case( - name="docs: large-black subrequest rejects small-white", - expr=( - 'device.attributes["resource-driver.example.com"].color == "black" && ' - 'device.attributes["resource-driver.example.com"].size == "large"' - ), - device=_SMALL_WHITE, - want=False, - ), - Case( - name="docs: small-white subrequest matches", - expr=( - 'device.attributes["resource-driver.example.com"].color == "white" && ' - 'device.attributes["resource-driver.example.com"].size == "small"' - ), - device=_SMALL_WHITE, - want=True, - ), - Case( - name="docs: extended-resource DeviceClass selector matches", - expr="device.driver == 'gpu.example.com' && device.attributes['gpu.example.com'].type == 'gpu'", - device=_EXAMPLE_GPU, - want=True, - ), - Case( - name="docs: extended-resource DeviceClass selector rejects other driver", - expr="device.driver == 'gpu.example.com' && device.attributes['gpu.example.com'].type == 'gpu'", - device=_NIC, - want=False, - ), - Case( - name="docs: ResourceClaim type+memory selector matches", - expr=( - 'device.attributes["driver.example.com"].type == "gpu" && ' - 'device.capacity["driver.example.com"].memory == quantity("64Gi")' - ), - device=_GPU_64GI, - want=True, - ), - Case( - name="docs: ResourceClaim type+memory selector rejects wrong memory", - expr=( - 'device.attributes["driver.example.com"].type == "gpu" && ' - 'device.capacity["driver.example.com"].memory == quantity("64Gi")' - ), - device=_device( - driver="driver.example.com", - attributes={"type": {"string": "gpu"}}, - capacity={"memory": {"value": "32Gi"}}, - ), - want=False, - ), - ] - for case in cases: - with self.subTest(case.name): - got = cel.Program(case.expr).matches(case.device) - self.assertEqual(case.want, got, f"{case.name}: -want, +got") +MATCHES_CASES = [ + # driver. + Case(name="driver equals", expr='device.driver == "gpu.nvidia.com"', device=_GPU, want=True), + Case(name="driver not equals", expr='device.driver == "nic.nvidia.com"', device=_GPU, want=False), + # Quantity comparison + methods. + Case( + name="quantity compareTo ge", + expr=f'{_CAP}.memory.compareTo(quantity("141Gi")) >= 0', + device=_GPU, + want=True, + ), + Case( + name="quantity compareTo too big", + expr=f'{_CAP}.memory.compareTo(quantity("200Gi")) >= 0', + device=_GPU, + want=False, + ), + Case( + name="quantity isGreaterThan", + expr=f'{_CAP}.memory.isGreaterThan(quantity("80Gi"))', + device=_GPU, + want=True, + ), + Case(name="quantity isLessThan", expr=f'{_CAP}.memory.isLessThan(quantity("200Gi"))', device=_GPU, want=True), + Case(name="quantity sign", expr=f"{_CAP}.memory.sign() == 1", device=_GPU, want=True), + Case(name="quantity asInteger", expr=f"{_CAP}.memory.asInteger() == {141 * 2**30}", device=_GPU, want=True), + Case(name="quantity isInteger", expr=f"{_CAP}.memory.isInteger()", device=_GPU, want=True), + Case( + name="quantity add", + expr=f'{_CAP}.memory.add(quantity("1Gi")).compareTo(quantity("142Gi")) == 0', + device=_GPU, + want=True, + ), + Case(name="isQuantity true", expr='isQuantity("1.3Gi")', device=_GPU, want=True), + Case(name="isQuantity false", expr='isQuantity("200K")', device=_GPU, want=False), + # Semver comparison + methods. + Case( + name="semver isGreaterThan", + expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.0.0"))', + device=_GPU, + want=True, + ), + Case( + name="semver not greater", + expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.9.0"))', + device=_GPU, + want=False, + ), + Case(name="semver major", expr=f"{_ATTR}.cudaComputeCapability.major() == 9", device=_GPU, want=True), + Case(name="semver minor", expr=f"{_ATTR}.cudaComputeCapability.minor() == 5", device=_GPU, want=True), + Case(name="semver patch", expr=f"{_ATTR}.cudaComputeCapability.patch() == 3", device=_GPU, want=True), + Case(name="semver equality", expr=f'{_ATTR}.cudaComputeCapability == semver("9.5.3")', device=_GPU, want=True), + Case(name="isSemver strict true", expr='isSemver("1.0.0")', device=_GPU, want=True), + Case(name="isSemver strict rejects short", expr='isSemver("1.0")', device=_GPU, want=False), + Case(name="isSemver normalize accepts short", expr='isSemver("1.0", true)', device=_GPU, want=True), + Case(name="semver normalize overload", expr='semver("v1.0", true).major() == 1', device=_GPU, want=True), + # Typed scalar attributes (resolve straight to the value, no .string). + Case(name="string attribute", expr=f'{_ATTR}.architecture == "Hopper"', device=_GPU, want=True), + Case( + name="string attribute mismatch", + expr='device.attributes["nic.nvidia.com"].linkType == "infiniband"', + device=_NIC, + want=True, + ), + Case( + name="bool attribute true", + expr=f"{_ATTR}.x", + device=_device(attributes={"x": {"bool": True}}), + want=True, + ), + Case( + name="bool attribute false", + expr=f"{_ATTR}.x", + device=_device(attributes={"x": {"bool": False}}), + want=False, + ), + Case( + name="int attribute", + expr=f"{_ATTR}.x >= 8", + device=_device(attributes={"x": {"int": 8}}), + want=True, + ), + Case( + name="int attribute below", + expr=f"{_ATTR}.x >= 8", + device=_device(attributes={"x": {"int": 4}}), + want=False, + ), + # Qualified names split into their own domain. + Case( + name="qualified name under its domain", + expr='device.attributes["resource.kubernetes.io"].pcieRoot == "pci0"', + device=_GPU, + want=True, + ), + Case(name="bare name under driver domain", expr=f'{_ATTR}.architecture == "Hopper"', device=_GPU, want=True), + # Non-matches that must not raise. + Case( + name="two-component version is non-match", + expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("8.0.0"))', + device=_device(attributes={"cudaComputeCapability": {"version": "9.0"}}), + want=False, + ), + Case( + name="malformed quantity is non-match", + expr=f'{_CAP}.memory.compareTo(quantity("1Gi")) >= 0', + device=_device(capacity={"memory": {"value": "10Mo"}}), + want=False, + ), + Case(name="unknown id is non-match", expr=f'{_ATTR}.nope == "x"', device=_GPU, want=False), + # A non-bool selector must not spuriously match. Upstream rejects it + # at compile time; we treat a non-bool result as a non-match. + Case(name="non-bool string selector is non-match", expr='"5"', device=_GPU, want=False), + Case( + name="non-bool int selector is non-match", + expr=f"{_ATTR}.x", + device=_device(attributes={"x": {"int": 5}}), + want=False, + ), + # Domain presence. Upstream's domain-presence idiom is "" in + # device.attributes, not has(device.attributes[""]): cel-go's + # has() macro rejects an index argument, so the has() form is a + # compile error on a real cluster (celpy accepts it - see cel.py's + # documented divergences). An unknown domain is simply absent (False), + # not present-but-empty. + Case(name="unknown domain absent", expr='"other.com" in device.attributes', device=_GPU, want=False), + Case(name="known domain present", expr='"gpu.nvidia.com" in device.attributes', device=_GPU, want=True), + # Reading an unknown domain resolves to an empty map (not an error), + # so an id lookup under it is a non-match rather than a failure. + Case( + name="unknown domain id is non-match", + expr='device.attributes["other.com"].x == "y"', + device=_GPU, + want=False, + ), + # Guard a domain read with the in idiom before indexing it. + Case( + name="guarded known domain", + expr=f'"gpu.nvidia.com" in device.attributes && {_ATTR}.architecture == "Hopper"', + device=_GPU, + want=True, + ), + # The full design selector. + Case( + name="full design expression", + expr=( + f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.0.0")) && ' + f'{_CAP}.memory.compareTo(quantity("141Gi")) >= 0' + ), + device=_GPU, + want=True, + ), + # Verbatim selector examples from the DRA docs, each against a device + # that should and should not match. + Case( + name="docs: large-black subrequest matches", + expr=( + 'device.attributes["resource-driver.example.com"].color == "black" && ' + 'device.attributes["resource-driver.example.com"].size == "large"' + ), + device=_LARGE_BLACK, + want=True, + ), + Case( + name="docs: large-black subrequest rejects small-white", + expr=( + 'device.attributes["resource-driver.example.com"].color == "black" && ' + 'device.attributes["resource-driver.example.com"].size == "large"' + ), + device=_SMALL_WHITE, + want=False, + ), + Case( + name="docs: small-white subrequest matches", + expr=( + 'device.attributes["resource-driver.example.com"].color == "white" && ' + 'device.attributes["resource-driver.example.com"].size == "small"' + ), + device=_SMALL_WHITE, + want=True, + ), + Case( + name="docs: extended-resource DeviceClass selector matches", + expr="device.driver == 'gpu.example.com' && device.attributes['gpu.example.com'].type == 'gpu'", + device=_EXAMPLE_GPU, + want=True, + ), + Case( + name="docs: extended-resource DeviceClass selector rejects other driver", + expr="device.driver == 'gpu.example.com' && device.attributes['gpu.example.com'].type == 'gpu'", + device=_NIC, + want=False, + ), + Case( + name="docs: ResourceClaim type+memory selector matches", + expr=( + 'device.attributes["driver.example.com"].type == "gpu" && ' + 'device.capacity["driver.example.com"].memory == quantity("64Gi")' + ), + device=_GPU_64GI, + want=True, + ), + Case( + name="docs: ResourceClaim type+memory selector rejects wrong memory", + expr=( + 'device.attributes["driver.example.com"].type == "gpu" && ' + 'device.capacity["driver.example.com"].memory == quantity("64Gi")' + ), + device=_device( + driver="driver.example.com", + attributes={"type": {"string": "gpu"}}, + capacity={"memory": {"value": "32Gi"}}, + ), + want=False, + ), +] -class TestCompile(unittest.TestCase): - def test_invalid_expression_raises(self) -> None: - with self.assertRaises(cel.CELCompileError): - cel.Program("not ) valid (") +@pytest.mark.parametrize("case", MATCHES_CASES, ids=lambda case: case.name) +def test_matches(case: Case) -> None: + """A DRA CEL selector matches a device as it does upstream.""" + got = cel.Program(case.expr).matches(case.device) + assert got == case.want + + +def test_compile_invalid_expression_raises() -> None: + """A malformed expression fails to compile.""" + with pytest.raises(cel.CELCompileError, match=r"not \) valid \("): + cel.Program("not ) valid (") diff --git a/functions/compose-model-deployment/tests/test_fn.py b/functions/compose-model-deployment/tests/test_fn.py index 5e1e4dfc9..f937959ab 100644 --- a/functions/compose-model-deployment/tests/test_fn.py +++ b/functions/compose-model-deployment/tests/test_fn.py @@ -14,16 +14,18 @@ """Tests for the compose-model-deployment function.""" +import asyncio import dataclasses import datetime -import unittest +import json from typing import Any -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 from models.ai.modelplane.modelcache import v1alpha1 as mcv1alpha1 @@ -405,1134 +407,1121 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) +def _compose_cases() -> list[Case]: + """The cases for test_compose, and the deployments they share.""" + # A deployment that sets spec.modelCacheRef. + xr_cached = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=1, + template=v1alpha1.TemplateModel( + spec=v1alpha1.SpecModel( + modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), + engines=[_ENGINE], + ) + ), + ), + ).model_dump(exclude_none=True, mode="json") + # A cached deployment that also sets its own clusterSelector, so the + # scheduler intersects it with the cache's footprint. + xr_cached_selector = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=1, + template=v1alpha1.TemplateModel( + spec=v1alpha1.SpecModel( + clusterSelector=v1alpha1.ClusterSelector(matchLabels={"region": "us-east"}), + modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), + engines=[_ENGINE], + ) + ), + ), + ).model_dump(exclude_none=True, mode="json") -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" + # A two-replica deployment (no container args) for the co-location case. + xr_two = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=2, + template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE_NO_ARGS])), + ), + ).model_dump(exclude_none=True, mode="json") - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() + # A disaggregated (PrefillDecode) deployment: a Prefill and a Decode engine. + xr_pd = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=1, + template=v1alpha1.TemplateModel( + spec=v1alpha1.SpecModel( + serving=v1alpha1.Serving(mode="PrefillDecode"), + engines=[ + _ENGINE.model_copy(update={"name": "prefill", "phase": "Prefill"}), + _ENGINE.model_copy(update={"name": "decode", "phase": "Decode"}), + ], + ) + ), + ), + ).model_dump(exclude_none=True, mode="json") - async def test_compose(self) -> None: - """The function fans out ModelReplicas and, once they're Ready, ModelEndpoints.""" + # A deployment parked at zero replicas. + xr_zero = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=0, + template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE])), + ), + ).model_dump(exclude_none=True, mode="json") - # A deployment that sets spec.modelCacheRef. - xr_cached = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - spec=v1alpha1.SpecModel( - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), - engines=[_ENGINE], - ) - ), + return [ + Case( + # First reconcile: the replica is composed but not yet observed + # Ready, so its endpoint is withheld - routing must not advertise + # a backend whose pods are still warming up (#102). + name="freshly scheduled replica composes no endpoint until ready", + req=_req(_XR, clusters=[_CLUSTER_A]), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + }, + "spec": { + "clusterName": "cluster-a", + "engines": _REPLICA_ENGINES, + }, + } + ), + ), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + ) ), - ).model_dump(exclude_none=True, mode="json") - - # A cached deployment that also sets its own clusterSelector, so the - # scheduler intersects it with the cache's footprint. - xr_cached_selector = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - spec=v1alpha1.SpecModel( - clusterSelector=v1alpha1.ClusterSelector(matchLabels={"region": "us-east"}), - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), - engines=[_ENGINE], - ) + ), + Case( + # A replica that has gone not-Ready (e.g. a crash-loop after + # once serving) has its endpoint withdrawn: the previously + # observed endpoint is absent from desired, so Crossplane + # deletes it and traffic stops routing to the dead backend + # (#102). Omitting it from desired - not composing it - is what + # drives the deletion. + name="not-ready replica withdraws its endpoint", + req=_req( + _XR, + clusters=[_CLUSTER_A], + replicas=[_EXISTING_REPLICA], + observed={ + "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=False), + "endpoint-cluster-a-0": { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, + }, + }, + ), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + }, + "spec": { + "clusterName": "cluster-a", + "engines": _REPLICA_ENGINES, + }, + } + ), + ), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + context=structpb.Struct(), + ) + ), + ), + Case( + name="no clusters produces warning", + req=_req(_XR, clusters=[]), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoClusters", + ), + ], + results=[ + fnv1.Result(severity=fnv1.SEVERITY_WARNING, message="No InferenceClusters found"), + ], + context=structpb.Struct(), + ) + ), + ), + Case( + name="insufficient capacity produces no replicas", + req=_req(_XR, clusters=[_cluster("cluster-a", nodes=0)]), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), + ready=fnv1.READY_FALSE, + ), + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="InsufficientCapacity", + message="0 of 1 replicas scheduled (checked 1 clusters)", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReplicasScheduled", + ), + ], + context=structpb.Struct(), + ) + ), + ), + Case( + # Zero desired parks the deployment before resolve_inputs runs: + # no requirements are declared (the want carries none), nothing + # is composed, and both conditions read True with the + # NoReplicasDesired reason rather than a capacity failure. + name="scaled to zero composes nothing and reports NoReplicasDesired", + req=_req(xr_zero), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), + ready=fnv1.READY_TRUE, + ), ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + ], + context=structpb.Struct(), ), - ).model_dump(exclude_none=True, mode="json") - - # A two-replica deployment (no container args) for the co-location case. - xr_two = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=2, - template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE_NO_ARGS])), + ), + Case( + # Scaling an existing deployment to zero: the observed replica + # and endpoint are absent from desired (pruned), and the + # transition is announced while they still exist. + name="scale to zero prunes observed replicas and emits an event", + req=_req( + xr_zero, + observed={ + "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True), + "endpoint-cluster-a-0": { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, + }, + }, ), - ).model_dump(exclude_none=True, mode="json") - - # A disaggregated (PrefillDecode) deployment: a Prefill and a Decode engine. - xr_pd = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - spec=v1alpha1.SpecModel( - serving=v1alpha1.Serving(mode="PrefillDecode"), - engines=[ - _ENGINE.model_copy(update={"name": "prefill", "phase": "Prefill"}), - _ENGINE.model_copy(update={"name": "decode", "phase": "Decode"}), - ], - ) + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), + ready=fnv1.READY_TRUE, + ), ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scaled to zero: removing all replicas", + ), + ], + context=structpb.Struct(), ), - ).model_dump(exclude_none=True, mode="json") - - # A deployment parked at zero replicas. - xr_zero = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=0, - template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE])), + ), + Case( + name="ready replica is preserved and keeps its endpoint", + req=_req( + _XR, + clusters=[_CLUSTER_A], + replicas=[_EXISTING_REPLICA], + observed={ + "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True), + "endpoint-cluster-a-0": { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, + }, + }, ), - ).model_dump(exclude_none=True, mode="json") - - cases = [ - Case( - # First reconcile: the replica is composed but not yet observed - # Ready, so its endpoint is withheld - routing must not advertise - # a backend whose pods are still warming up (#102). - name="freshly scheduled replica composes no endpoint until ready", - req=_req(_XR, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 1}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + }, + "spec": { + "clusterName": "cluster-a", + "engines": _REPLICA_ENGINES, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "endpoint-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, + }, + "spec": { + "origin": "https://cluster.clusters.example.com", + "api": { + "schema": "OpenAI", + "prefix": "/ml-team/my-model-5ab63/v1", }, - } - ), + "model": "ml-team/my-model", + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="AllReplicasReady", + message="1 of 1 ready", + ), + ], + context=structpb.Struct(), + ) + ), + ), + Case( + name="offline pinned cluster keeps replica but drops endpoint", + req=_req( + _XR, + clusters=[_cluster("cluster-a", ready=False, hostname=None)], + replicas=[_EXISTING_REPLICA], + observed={"replica-cluster-a-0": _EXISTING_REPLICA}, + ), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + }, + "spec": { + "clusterName": "cluster-a", + "engines": _REPLICA_ENGINES, + }, + } + ), ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + context=structpb.Struct(), + ) + ), + ), + Case( + name="deleted pinned cluster triggers replica re-placement", + req=_req( + _XR, + clusters=[_cluster("cluster-b", hostname="cluster-b.clusters.example.com")], + replicas=[_EXISTING_REPLICA], + observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, + ), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-b-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-f0b76", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-b", + "modelplane.ai/replica-index": "0", + }, + }, + "spec": { + "clusterName": "cluster-b", + "engines": _REPLICA_ENGINES, + }, + } + ), ), - ], - context=structpb.Struct(), - ) - ), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-b", + ), + ], + context=structpb.Struct(), + ) ), - Case( - # A replica that has gone not-Ready (e.g. a crash-loop after - # once serving) has its endpoint withdrawn: the previously - # observed endpoint is absent from desired, so Crossplane - # deletes it and traffic stops routing to the dead backend - # (#102). Omitting it from desired - not composing it - is what - # drives the deletion. - name="not-ready replica withdraws its endpoint", - req=_req( - _XR, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={ - "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=False), - "endpoint-cluster-a-0": { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, + ), + Case( + name="modelCacheRef is propagated onto the composed replica", + req=_req(xr_cached, clusters=[_CLUSTER_A_CACHE], cache=_cache("qwen")), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + }, + "spec": { + "clusterName": "cluster-a", + "modelCacheRef": {"name": "qwen"}, + "engines": _REPLICA_ENGINES, + }, + } + ), + ), }, - }, + ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, + cache_name="qwen", + ), + ), + Case( + # The cache stages only to a subset of clusters; the scheduler + # intersects the cache's footprint with the deployment's own + # clusterSelector so replicas never land where the cache isn't. + name="cache clusterSelector is intersected with the deployment's", + req=_req( + xr_cached_selector, + clusters=[_CLUSTER_A_CACHE], + cache=_cache("qwen", match_labels={"tier": "gpu"}), + ), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, - } - ), + }, + "spec": { + "clusterName": "cluster-a", + "modelCacheRef": {"name": "qwen"}, + "engines": _REPLICA_ENGINES, + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", ), - ], - context=structpb.Struct(), - ) + }, + ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), ), + cluster_labels={"region": "us-east", "tier": "gpu"}, + cache_name="qwen", ), - Case( - name="no clusters produces warning", - req=_req(_XR, clusters=[]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoClusters", - ), - ], - results=[ - fnv1.Result(severity=fnv1.SEVERITY_WARNING, message="No InferenceClusters found"), - ], - context=structpb.Struct(), - ) + ), + Case( + # compose-model-cache stages only onto clusters that report cache + # storage, so cluster-a, which reports none, can't host the + # replica's PVC. The replica lands on cluster-b, though cluster-a + # would win the tiebreak by name. + name="a cached replica lands only on a cluster with cache storage", + req=_req( + xr_cached, + clusters=[_CLUSTER_A, _cluster("cluster-b", cache_storage=True)], + cache=_cache("qwen"), + ), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-b-0": fnv1.Resource(resource=resource.dict_to_struct(_CACHED_REPLICA_B)), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-b", + ), + ], + context=structpb.Struct(), ), + cache_name="qwen", ), - Case( - name="insufficient capacity produces no replicas", - req=_req(_XR, clusters=[_cluster("cluster-a", nodes=0)]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_FALSE, - ), + ), + Case( + # A replica is running on cluster-a, which has no cache storage, + # say because the bug this guards against put it there. Its PVC + # never appears, so it's dropped and re-placed on cluster-b, the + # way a replica is when the cache's selector stops matching. + name="a running cached replica on a cluster without cache storage is re-placed", + req=_req( + xr_cached, + clusters=[_CLUSTER_A, _cluster("cluster-b", cache_storage=True)], + replicas=[_EXISTING_REPLICA], + observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, + cache=_cache("qwen"), + ), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), + ), + resources={ + "replica-cluster-b-0": fnv1.Resource(resource=resource.dict_to_struct(_CACHED_REPLICA_B)), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-b", ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="InsufficientCapacity", - message="0 of 1 replicas scheduled (checked 1 clusters)", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoReplicasScheduled", - ), - ], - context=structpb.Struct(), - ) + ], + context=structpb.Struct(), ), + cache_name="qwen", ), - Case( - # Zero desired parks the deployment before resolve_inputs runs: - # no requirements are declared (the want carries none), nothing - # is composed, and both conditions read True with the - # NoReplicasDesired reason rather than a capacity failure. - name="scaled to zero composes nothing and reports NoReplicasDesired", - req=_req(xr_zero), - want=fnv1.RunFunctionResponse( + ), + Case( + # The only candidate has no cache storage, so the cache can't + # stage there and nothing is placed. ReplicasScheduled says why + # rather than blaming capacity. + name="no candidate with cache storage places nothing", + req=_req(xr_cached, clusters=[_CLUSTER_A], cache=_cache("qwen")), + want=_want( + fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( composite=fnv1.Resource( resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_FALSE, ), ), conditions=[ fnv1.Condition( - type="ReplicasScheduled", + type="ModelCacheResolved", status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoCacheStorage", + message="0 of 1 replicas scheduled: no candidate cluster has storage for ModelCache qwen", ), fnv1.Condition( type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReplicasScheduled", ), ], context=structpb.Struct(), ), + cache_name="qwen", ), - Case( - # Scaling an existing deployment to zero: the observed replica - # and endpoint are absent from desired (pruned), and the - # transition is announced while they still exist. - name="scale to zero prunes observed replicas and emits an event", - req=_req( - xr_zero, - observed={ - "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True), - "endpoint-cluster-a-0": { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, - }, - }, - ), - want=fnv1.RunFunctionResponse( + ), + Case( + # A referenced cache Crossplane hasn't fetched yet leaves the + # footprint unknown. With no replicas to retain, the function + # holds off placing any rather than risk landing them outside the + # footprint: fill is suppressed, so nothing is composed, and + # ModelCacheResolved=False (Unresolved) says why. The wait is + # transient and self-clearing, so it's a condition, not an event. + # The cluster and replica requirements are still declared so the + # cache can resolve alongside them. + name="unresolved cache suppresses new placement", + req=_req(xr_cached, clusters=[_CLUSTER_A]), + want=_want( + fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( composite=fnv1.Resource( resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_FALSE, ), ), conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelCacheUnresolved", + message="Waiting for ModelCache qwen", + ), fnv1.Condition( type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", + status=fnv1.STATUS_CONDITION_FALSE, + reason="InsufficientCapacity", + message="0 of 1 replicas scheduled (checked 1 clusters)", ), fnv1.Condition( type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scaled to zero: removing all replicas", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReplicasScheduled", ), ], context=structpb.Struct(), ), + cache_name="qwen", ), - Case( - name="ready replica is preserved and keeps its endpoint", - req=_req( - _XR, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={ - "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True), - "endpoint-cluster-a-0": { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, - }, - }, - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 1}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "endpoint-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "origin": "https://cluster.clusters.example.com", - "api": { - "schema": "OpenAI", - "prefix": "/ml-team/my-model-5ab63/v1", - }, - "model": "ml-team/my-model", - }, - } - ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="AllReplicasReady", - message="1 of 1 ready", - ), - ], - context=structpb.Struct(), - ) - ), + ), + Case( + # The cache a live deployment depends on is deleted (the cache + # requirement resolves but matches nothing - ABSENT). The cache + # only matters when loading weights, which already happened, so + # its disappearance must not tear the deployment down: the + # existing replica is retained (retain ignores fill) even as + # ModelCacheResolved goes False (NotFound) and new placement is + # suppressed. + name="deleted cache retains existing replicas", + req=_req( + xr_cached, + clusters=[_CLUSTER_A], + replicas=[_EXISTING_REPLICA], + observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, + cache_resolved_empty=True, ), - Case( - name="offline pinned cluster keeps replica but drops endpoint", - req=_req( - _XR, - clusters=[_cluster("cluster-a", ready=False, hostname=None)], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _EXISTING_REPLICA}, - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 1}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, - } - ), + }, + "spec": { + "clusterName": "cluster-a", + "modelCacheRef": {"name": "qwen"}, + "engines": _REPLICA_ENGINES, + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - context=structpb.Struct(), - ) - ), - ), - Case( - name="deleted pinned cluster triggers replica re-placement", - req=_req( - _XR, - clusters=[_cluster("cluster-b", hostname="cluster-b.clusters.example.com")], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-b-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-f0b76", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-b", - "modelplane.ai/replica-index": "0", - }, + ready=fnv1.READY_TRUE, + ), + "endpoint-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, - "spec": { - "clusterName": "cluster-b", - "engines": _REPLICA_ENGINES, + }, + "spec": { + "origin": "https://cluster.clusters.example.com", + "api": { + "schema": "OpenAI", + "prefix": "/ml-team/my-model-5ab63/v1", }, - } - ), + "model": "ml-team/my-model", + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-b", ), - ], - context=structpb.Struct(), - ) + }, + ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelCacheNotFound", + message="ModelCache qwen not found; holding replica placement", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="AllReplicasReady", + message="1 of 1 ready", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="ModelCache qwen not found; holding replica placement", + ), + ], + context=structpb.Struct(), ), + cache_name="qwen", ), - Case( - name="modelCacheRef is propagated onto the composed replica", - req=_req(xr_cached, clusters=[_CLUSTER_A_CACHE], cache=_cache("qwen")), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "modelCacheRef": {"name": "qwen"}, - "engines": _REPLICA_ENGINES, + ), + Case( + name="two replicas co-locate on one cluster as distinct resources", + req=_req(xr_two, clusters=[_CLUSTER_A]), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 2, "ready": 0}}}), + ), + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, - } - ), + }, + "spec": { + "clusterName": "cluster-a", + "engines": _REPLICA_ENGINES_NO_ARGS, + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ), - cache_name="qwen", - ), - ), - Case( - # The cache stages only to a subset of clusters; the scheduler - # intersects the cache's footprint with the deployment's own - # clusterSelector so replicas never land where the cache isn't. - name="cache clusterSelector is intersected with the deployment's", - req=_req( - xr_cached_selector, - clusters=[_CLUSTER_A_CACHE], - cache=_cache("qwen", match_labels={"tier": "gpu"}), - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "modelCacheRef": {"name": "qwen"}, - "engines": _REPLICA_ENGINES, + "replica-cluster-a-1": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-609c5", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "1", }, - } - ), + }, + "spec": { + "clusterName": "cluster-a", + "engines": _REPLICA_ENGINES_NO_ARGS, + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", ), - ], - context=structpb.Struct(), + }, ), - cluster_labels={"region": "us-east", "tier": "gpu"}, - cache_name="qwen", - ), - ), - Case( - # compose-model-cache stages only onto clusters that report cache - # storage, so cluster-a, which reports none, can't host the - # replica's PVC. The replica lands on cluster-b, though cluster-a - # would win the tiebreak by name. - name="a cached replica lands only on a cluster with cache storage", - req=_req( - xr_cached, - clusters=[_CLUSTER_A, _cluster("cluster-b", cache_storage=True)], - cache=_cache("qwen"), - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-b-0": fnv1.Resource( - resource=resource.dict_to_struct(_CACHED_REPLICA_B) - ), - }, + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-b", - ), - ], - context=structpb.Struct(), - ), - cache_name="qwen", - ), - ), - Case( - # A replica is running on cluster-a, which has no cache storage, - # say because the bug this guards against put it there. Its PVC - # never appears, so it's dropped and re-placed on cluster-b, the - # way a replica is when the cache's selector stops matching. - name="a running cached replica on a cluster without cache storage is re-placed", - req=_req( - xr_cached, - clusters=[_CLUSTER_A, _cluster("cluster-b", cache_storage=True)], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - cache=_cache("qwen"), - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-b-0": fnv1.Resource( - resource=resource.dict_to_struct(_CACHED_REPLICA_B) - ), - }, + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 2 ready", ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-b", - ), - ], - context=structpb.Struct(), - ), - cache_name="qwen", - ), - ), - Case( - # The only candidate has no cache storage, so the cache can't - # stage there and nothing is placed. ReplicasScheduled says why - # rather than blaming capacity. - name="no candidate with cache storage places nothing", - req=_req(xr_cached, clusters=[_CLUSTER_A], cache=_cache("qwen")), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_FALSE, - ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 2 replicas across 1 clusters: cluster-a", ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoCacheStorage", - message="0 of 1 replicas scheduled: no candidate cluster has storage for ModelCache qwen", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoReplicasScheduled", - ), - ], - context=structpb.Struct(), - ), - cache_name="qwen", - ), + ], + context=structpb.Struct(), + ) ), - Case( - # A referenced cache Crossplane hasn't fetched yet leaves the - # footprint unknown. With no replicas to retain, the function - # holds off placing any rather than risk landing them outside the - # footprint: fill is suppressed, so nothing is composed, and - # ModelCacheResolved=False (Unresolved) says why. The wait is - # transient and self-clearing, so it's a condition, not an event. - # The cluster and replica requirements are still declared so the - # cache can resolve alongside them. - name="unresolved cache suppresses new placement", - req=_req(xr_cached, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_FALSE, - ), + ), + Case( + # PrefillDecode copies serving and each engine's phase onto the + # replica; the replica backend reads them to front the engines + # with an InferencePool + endpoint picker rather than a Service. + name="PrefillDecode copies serving and engine phases onto the replica", + req=_req(xr_pd, clusters=[_CLUSTER_A]), + want=_want( + fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelCacheUnresolved", - message="Waiting for ModelCache qwen", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="InsufficientCapacity", - message="0 of 1 replicas scheduled (checked 1 clusters)", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoReplicasScheduled", - ), - ], - context=structpb.Struct(), - ), - cache_name="qwen", - ), - ), - Case( - # The cache a live deployment depends on is deleted (the cache - # requirement resolves but matches nothing - ABSENT). The cache - # only matters when loading weights, which already happened, so - # its disappearance must not tear the deployment down: the - # existing replica is retained (retain ignores fill) even as - # ModelCacheResolved goes False (NotFound) and new placement is - # suppressed. - name="deleted cache retains existing replicas", - req=_req( - xr_cached, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - cache_resolved_empty=True, - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 1}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "modelCacheRef": {"name": "qwen"}, - "engines": _REPLICA_ENGINES, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "endpoint-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "origin": "https://cluster.clusters.example.com", - "api": { - "schema": "OpenAI", - "prefix": "/ml-team/my-model-5ab63/v1", - }, - "model": "ml-team/my-model", + resources={ + "replica-cluster-a-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, - } - ), + }, + "spec": { + "clusterName": "cluster-a", + "serving": {"mode": "PrefillDecode"}, + "engines": _PD_REPLICA_ENGINES, + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelCacheNotFound", - message="ModelCache qwen not found; holding replica placement", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="AllReplicasReady", - message="1 of 1 ready", ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_WARNING, - message="ModelCache qwen not found; holding replica placement", - ), - ], - context=structpb.Struct(), + }, ), - cache_name="qwen", - ), - ), - Case( - name="two replicas co-locate on one cluster as distinct resources", - req=_req(xr_two, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 2, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES_NO_ARGS, - }, - } - ), - ), - "replica-cluster-a-1": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-609c5", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "1", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES_NO_ARGS, - }, - } - ), - ), - }, + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 2 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 2 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - ), - ), - Case( - # PrefillDecode copies serving and each engine's phase onto the - # replica; the replica backend reads them to front the engines - # with an InferencePool + endpoint picker rather than a Service. - name="PrefillDecode copies serving and engine phases onto the replica", - req=_req(xr_pd, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "serving": {"mode": "PrefillDecode"}, - "engines": _PD_REPLICA_ENGINES, - }, - } - ), - ), - }, + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + ) ), - ] + ), + ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """The function fans out ModelReplicas and, once they're Ready, ModelEndpoints.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) def _composed(resp: fnv1.RunFunctionResponse, kind: str) -> list[dict]: @@ -1545,183 +1534,191 @@ def _composed(resp: fnv1.RunFunctionResponse, kind: str) -> list[dict]: return result -class TestTemplateLabels(unittest.IsolatedAsyncioTestCase): - """spec.template.metadata.labels land on the composed ModelReplicas and - ModelEndpoints, alongside the labels Modelplane manages.""" +# spec.template.metadata.labels land on the composed ModelReplicas and +# ModelEndpoints, alongside the labels Modelplane manages. - async def test_stamped_on_replica_and_endpoint(self) -> None: - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - metadata=v1alpha1.Metadata(labels={"tier": "prod", "team": "search"}), - spec=v1alpha1.SpecModel(engines=[_ENGINE]), - ), + +def test_template_labels_stamped_on_replica_and_endpoint() -> None: + """Template labels land on the replica and endpoint beside the managed labels.""" + xr = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=1, + template=v1alpha1.TemplateModel( + metadata=v1alpha1.Metadata(labels={"tier": "prod", "team": "search"}), + spec=v1alpha1.SpecModel(engines=[_ENGINE]), ), - ).model_dump(exclude_none=True, mode="json") - # An observed, Ready replica lets the endpoint compose this reconcile. - req = _req( - xr, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - ) - got = await fn.FunctionRunner().RunFunction(req, None) - - composed = _composed(got, "ModelReplica") + _composed(got, "ModelEndpoint") - self.assertEqual(len(composed), 2, "expected one ModelReplica and one ModelEndpoint") - for obj in composed: - labels = obj["metadata"]["labels"] - self.assertEqual(labels.get("tier"), "prod") - self.assertEqual(labels.get("team"), "search") - self.assertEqual(labels.get("modelplane.ai/deployment"), "my-model") - self.assertEqual(labels.get("modelplane.ai/cluster"), "cluster-a") - self.assertEqual(labels.get("modelplane.ai/replica-index"), "0") - - async def test_managed_labels_win_a_collision(self) -> None: - """The XRD's CEL rejects a template label under the modelplane.ai/ prefix, - but the invariant lives in the function too: managed labels are stamped - last, so a colliding label can't override them even if that CEL rule is - relaxed or the function is reused elsewhere.""" - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - metadata=v1alpha1.Metadata(labels={"modelplane.ai/cluster": "wrong", "tier": "prod"}), - spec=v1alpha1.SpecModel(engines=[_ENGINE]), - ), + ), + ).model_dump(exclude_none=True, mode="json") + # An observed, Ready replica lets the endpoint compose this reconcile. + req = _req( + xr, + clusters=[_CLUSTER_A], + replicas=[_EXISTING_REPLICA], + observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + composed = _composed(got, "ModelReplica") + _composed(got, "ModelEndpoint") + assert len(composed) == 2, "expected one ModelReplica and one ModelEndpoint" + for obj in composed: + labels = obj["metadata"]["labels"] + assert labels.get("tier") == "prod" + assert labels.get("team") == "search" + assert labels.get("modelplane.ai/deployment") == "my-model" + assert labels.get("modelplane.ai/cluster") == "cluster-a" + assert labels.get("modelplane.ai/replica-index") == "0" + + +def test_template_labels_managed_labels_win_a_collision() -> None: + """A managed label beats a template label of the same key.""" + # The XRD's CEL rejects a template label under the modelplane.ai/ prefix, + # but the invariant lives in the function too: managed labels are stamped + # last, so a colliding label can't override them even if that CEL rule is + # relaxed or the function is reused elsewhere. + xr = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=1, + template=v1alpha1.TemplateModel( + metadata=v1alpha1.Metadata(labels={"modelplane.ai/cluster": "wrong", "tier": "prod"}), + spec=v1alpha1.SpecModel(engines=[_ENGINE]), ), - ).model_dump(exclude_none=True, mode="json") - got = await fn.FunctionRunner().RunFunction(_req(xr, clusters=[_CLUSTER_A]), None) + ), + ).model_dump(exclude_none=True, mode="json") + got = asyncio.run(fn.FunctionRunner().RunFunction(_req(xr, clusters=[_CLUSTER_A]), None)) - replica = _composed(got, "ModelReplica")[0] - self.assertEqual(replica["metadata"]["labels"]["modelplane.ai/cluster"], "cluster-a") - self.assertEqual(replica["metadata"]["labels"]["tier"], "prod") + replica = _composed(got, "ModelReplica")[0] + assert replica["metadata"]["labels"]["modelplane.ai/cluster"] == "cluster-a" + assert replica["metadata"]["labels"]["tier"] == "prod" -class TestResolveRequired(unittest.TestCase): - """Tests for fn.resolve_required - the three-state required-resource read.""" +def test_resolve_required() -> None: + """resolve_required tells a found, a missing, and an unfetched requirement apart.""" + cache = {"apiVersion": "modelplane.ai/v1alpha1", "kind": "ModelCache", "metadata": {"name": "qwen"}} - def test_resolve_required(self) -> None: - cache = {"apiVersion": "modelplane.ai/v1alpha1", "kind": "ModelCache", "metadata": {"name": "qwen"}} + # PRESENT: the requirement resolved and matched a resource. + req = fnv1.RunFunctionRequest() + req.required_resources["cache"].items.append(fnv1.Resource(resource=resource.dict_to_struct(cache))) + assert fn.resolve_required(req, "cache") == (fn.Resolution.PRESENT, cache) - # PRESENT: the requirement resolved and matched a resource. - req = fnv1.RunFunctionRequest() - req.required_resources["cache"].items.append(fnv1.Resource(resource=resource.dict_to_struct(cache))) - self.assertEqual((fn.Resolution.PRESENT, cache), fn.resolve_required(req, "cache")) + # ABSENT: the requirement resolved but matched nothing (key present, no items). + req = fnv1.RunFunctionRequest() + req.required_resources["cache"].SetInParent() + assert fn.resolve_required(req, "cache") == (fn.Resolution.ABSENT, None) - # ABSENT: the requirement resolved but matched nothing (key present, no items). - req = fnv1.RunFunctionRequest() - req.required_resources["cache"].SetInParent() - self.assertEqual((fn.Resolution.ABSENT, None), fn.resolve_required(req, "cache")) - - # UNRESOLVED: Crossplane has not fetched the requirement (key absent). - req = fnv1.RunFunctionRequest() - self.assertEqual((fn.Resolution.UNRESOLVED, None), fn.resolve_required(req, "cache")) - - -class TestServedModelName(unittest.TestCase): - """The name an engine is started under, and how it gets there.""" - - def test_it_goes_ahead_of_the_users_env(self) -> None: - """Env expansion is left to right, so an arg or a later entry - referencing $(MODELPLANE_SERVED_MODEL_NAME) only resolves if it's - first.""" - template = mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container( - name="engine", - image="vllm/vllm-openai:latest", - env=[mrv1alpha1.EnvItem(name="HF_TOKEN", value="x")], - ) - ] - ) - ) - fn._inject_served_model_name(template, "ml-team/kimi-k2") - assert template.spec is not None - self.assertEqual( - [(e.name, e.value) for e in template.spec.containers[0].env or []], - [("MODELPLANE_SERVED_MODEL_NAME", "ml-team/kimi-k2"), ("HF_TOKEN", "x")], - ) + # UNRESOLVED: Crossplane has not fetched the requirement (key absent). + req = fnv1.RunFunctionRequest() + assert fn.resolve_required(req, "cache") == (fn.Resolution.UNRESOLVED, None) - def test_a_user_override_is_dropped(self) -> None: - """Modelplane decides this value. Honouring an override would let the - engine answer to a name nothing routes to, which surfaces as a 404 from - the engine rather than anything visible in status.""" - template = mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container( - name="engine", - image="vllm/vllm-openai:latest", - env=[mrv1alpha1.EnvItem(name="MODELPLANE_SERVED_MODEL_NAME", value="mine")], - ) - ] - ) - ) - fn._inject_served_model_name(template, "ml-team/kimi-k2") - assert template.spec is not None - self.assertEqual( - [(e.name, e.value) for e in template.spec.containers[0].env or []], - [("MODELPLANE_SERVED_MODEL_NAME", "ml-team/kimi-k2")], - ) - def test_it_is_namespaced(self) -> None: - """So two deployments in different namespaces can't collide, and a - ModelService can rewrite one name for a whole deployment.""" - self.assertEqual(fn.served_model_name("ml-team", "kimi-k2"), "ml-team/kimi-k2") +# The name an engine is started under, and how it gets there. -class TestPlacementLabels(unittest.IsolatedAsyncioTestCase): - """A cluster's spec.placement.metadata.labels land on the ModelReplicas and - ModelEndpoints composed there. +def test_served_model_name_goes_ahead_of_the_users_env() -> None: + """The served model name env var comes before the container's own env.""" + # Env expansion is left to right, so an arg or a later entry referencing + # $(MODELPLANE_SERVED_MODEL_NAME) only resolves if it's first. + template = mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + env=[mrv1alpha1.EnvItem(name="HF_TOKEN", value="x")], + ) + ] + ) + ) + fn._inject_served_model_name(template, "ml-team/kimi-k2") + assert template.spec is not None + assert [(e.name, e.value) for e in template.spec.containers[0].env or []] == [ + ("MODELPLANE_SERVED_MODEL_NAME", "ml-team/kimi-k2"), + ("HF_TOKEN", "x"), + ] - This is the endpoint half of residency: a ModelService selects endpoints by - label, so without it a region-scoped service can't select its own replicas, - and nobody can label them by hand because Modelplane owns them. The gateway - half is an InferenceGateway's serviceSelector. - """ - async def test_stamped_on_replica_and_endpoint(self) -> None: - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE])), - ), - ).model_dump(exclude_none=True, mode="json") - req = _req( - xr, - clusters=[_cluster("cluster-a", placement_labels={"example.org/region": "eu"})], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, +def test_served_model_name_user_override_is_dropped() -> None: + """A user's own MODELPLANE_SERVED_MODEL_NAME is replaced, not kept.""" + # Modelplane decides this value. Honouring an override would let the engine + # answer to a name nothing routes to, which surfaces as a 404 from the + # engine rather than anything visible in status. + template = mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + env=[mrv1alpha1.EnvItem(name="MODELPLANE_SERVED_MODEL_NAME", value="mine")], + ) + ] ) - got = await fn.FunctionRunner().RunFunction(req, None) - - composed = _composed(got, "ModelReplica") + _composed(got, "ModelEndpoint") - self.assertEqual(len(composed), 2, "expected one ModelReplica and one ModelEndpoint") - for obj in composed: - self.assertEqual(obj["metadata"]["labels"].get("example.org/region"), "eu") - - async def test_a_cluster_label_beats_a_template_label(self) -> None: - """The cluster is the authority on where it is, so its placement labels - are stamped after the deployment's own template labels.""" - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - metadata=v1alpha1.Metadata(labels={"example.org/region": "wrong"}), - spec=v1alpha1.SpecModel(engines=[_ENGINE]), - ), + ) + fn._inject_served_model_name(template, "ml-team/kimi-k2") + assert template.spec is not None + assert [(e.name, e.value) for e in template.spec.containers[0].env or []] == [ + ("MODELPLANE_SERVED_MODEL_NAME", "ml-team/kimi-k2"), + ] + + +def test_served_model_name_is_namespaced() -> None: + """served_model_name prefixes the deployment's name with its namespace.""" + # So two deployments in different namespaces can't collide, and a + # ModelService can rewrite one name for a whole deployment. + assert fn.served_model_name("ml-team", "kimi-k2") == "ml-team/kimi-k2" + + +# A cluster's spec.placement.metadata.labels land on the ModelReplicas and +# ModelEndpoints composed there. +# +# This is the endpoint half of residency: a ModelService selects endpoints by +# label, so without it a region-scoped service can't select its own replicas, +# and nobody can label them by hand because Modelplane owns them. The gateway +# half is an InferenceGateway's serviceSelector. + + +def test_placement_labels_stamped_on_replica_and_endpoint() -> None: + """A cluster's placement labels land on the replica and endpoint composed there.""" + xr = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=1, + template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE])), + ), + ).model_dump(exclude_none=True, mode="json") + req = _req( + xr, + clusters=[_cluster("cluster-a", placement_labels={"example.org/region": "eu"})], + replicas=[_EXISTING_REPLICA], + observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + composed = _composed(got, "ModelReplica") + _composed(got, "ModelEndpoint") + assert len(composed) == 2, "expected one ModelReplica and one ModelEndpoint" + for obj in composed: + assert obj["metadata"]["labels"].get("example.org/region") == "eu" + + +def test_placement_labels_cluster_label_beats_a_template_label() -> None: + """A cluster's placement label beats a template label of the same key.""" + # The cluster is the authority on where it is, so its placement labels are + # stamped after the deployment's own template labels. + xr = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=1, + template=v1alpha1.TemplateModel( + metadata=v1alpha1.Metadata(labels={"example.org/region": "wrong"}), + spec=v1alpha1.SpecModel(engines=[_ENGINE]), ), - ).model_dump(exclude_none=True, mode="json") - got = await fn.FunctionRunner().RunFunction( + ), + ).model_dump(exclude_none=True, mode="json") + got = asyncio.run( + fn.FunctionRunner().RunFunction( _req(xr, clusters=[_cluster("cluster-a", placement_labels={"example.org/region": "eu"})]), None, ) - replica = _composed(got, "ModelReplica")[0] - self.assertEqual(replica["metadata"]["labels"]["example.org/region"], "eu") + ) + replica = _composed(got, "ModelReplica")[0] + assert replica["metadata"]["labels"]["example.org/region"] == "eu" diff --git a/functions/compose-model-deployment/tests/test_quantity.py b/functions/compose-model-deployment/tests/test_quantity.py index d196962ff..5c0429415 100644 --- a/functions/compose-model-deployment/tests/test_quantity.py +++ b/functions/compose-model-deployment/tests/test_quantity.py @@ -37,8 +37,8 @@ """ import dataclasses -import unittest +import pytest from function import cel, quantity @@ -60,169 +60,164 @@ class ParseErrCase: input: str -class TestQuantityCEL(unittest.TestCase): - """Mirrors quantity_test.go TestQuantity (and the doc-comment examples).""" - - def test_quantity(self) -> None: - cases = [ - # parse + isQuantity. - Case(name="parse", expr='quantity("12Mi").compareTo(quantity("12Mi")) == 0', want=True), - Case(name="isQuantity int string", expr='isQuantity("20")', want=True), - Case(name="isQuantity megabytes", expr='isQuantity("20M")', want=True), - Case(name="isQuantity mebibytes", expr='isQuantity("20Mi")', want=True), - Case(name="isQuantity invalid suffix", expr='isQuantity("20Mo")', want=False), - Case(name="isQuantity passing regex bad suffix", expr='isQuantity("10Mm")', want=False), - # resource.Quantity accepts decimal exponents and nano/micro suffixes. - Case(name="isQuantity exponent lowercase", expr='isQuantity("256e3")', want=True), - Case(name="isQuantity exponent uppercase", expr='isQuantity("1E3")', want=True), - Case(name="exponent value", expr='quantity("256e3").compareTo(quantity("256000")) == 0', want=True), - Case(name="isQuantity nano", expr='isQuantity("100n")', want=True), - Case(name="isQuantity micro", expr='isQuantity("100u")', want=True), - Case(name="isQuantity trailing dot", expr='isQuantity("5.")', want=True), - # The quantity() constructor does NOT trim whitespace. - Case(name="isQuantity leading whitespace false", expr='isQuantity(" 5Gi")', want=False), - Case(name="isQuantity trailing whitespace false", expr='isQuantity("5Gi ")', want=False), - # Values equal at nano resolution compare equal (resource.Quantity.Cmp - # rounds to nano). - Case( - name="nano rounding equality", - expr='quantity("0.0000000004").compareTo(quantity("0.000000001")) == 0', - want=True, - ), - # doc-comment isQuantity examples. - Case(name="isQuantity 1.3G", expr='isQuantity("1.3G")', want=True), - Case(name="isQuantity 1.3Gi", expr='isQuantity("1.3Gi")', want=True), - Case(name="isQuantity comma", expr='isQuantity("1,3G")', want=False), - Case(name="isQuantity 10000k", expr='isQuantity("10000k")', want=True), - Case(name="isQuantity capital K", expr='isQuantity("200K")', want=False), - Case(name="isQuantity Three", expr='isQuantity("Three")', want=False), - Case(name="isQuantity bare suffix", expr='isQuantity("Mi")', want=False), - # equality. - Case(name="equality reflexivity", expr='quantity("200M") == quantity("200M")', want=True), - Case( - name="equality symmetry", - expr='quantity("200M") == quantity("0.2G") && quantity("0.2G") == quantity("200M")', - want=True, - ), - Case( - name="equality transitivity", - expr=( - 'quantity("2M") == quantity("0.002G") && quantity("2000k") == quantity("2M") && ' - 'quantity("0.002G") == quantity("2000k")' - ), - want=True, - ), - Case(name="inequality", expr='quantity("200M") == quantity("0.3G")', want=False), - # isLessThan / isGreaterThan. - Case(name="less", expr='quantity("50M").isLessThan(quantity("50Mi"))', want=True), - Case(name="less obvious", expr='quantity("50M").isLessThan(quantity("100M"))', want=True), - Case(name="less false", expr='quantity("100M").isLessThan(quantity("50M"))', want=False), - Case(name="greater", expr='quantity("50Mi").isGreaterThan(quantity("50M"))', want=True), - Case(name="greater obvious", expr='quantity("150Mi").isGreaterThan(quantity("100Mi"))', want=True), - Case(name="greater false", expr='quantity("50M").isGreaterThan(quantity("100M"))', want=False), - # compareTo. - Case(name="compare equal", expr='quantity("200M").compareTo(quantity("0.2G")) == 0', want=True), - Case(name="compare less", expr='quantity("50M").compareTo(quantity("50Mi")) == -1', want=True), - Case(name="compare greater", expr='quantity("50Mi").compareTo(quantity("50M")) == 1', want=True), - # add / sub (quantity and int overloads). - Case(name="add quantity", expr='quantity("50k").add(quantity("20")) == quantity("50.02k")', want=True), - Case(name="add int not less", expr='quantity("50k").add(20).isLessThan(quantity("50020"))', want=False), - Case(name="sub quantity", expr='quantity("50k").sub(quantity("20")) == quantity("49.98k")', want=True), - Case(name="sub int", expr='quantity("50k").sub(20) == quantity("49980")', want=True), - Case( - name="arith chain 1", - expr='quantity("50k").add(20).sub(quantity("100k")).asInteger() == -49980', - want=True, - ), - Case( - name="arith chain 2", - expr='quantity("50k").add(20).sub(quantity("100k")).sub(-50000).asInteger() == 20', - want=True, - ), - # sign (doc comment). Upstream declares sign as a GLOBAL function - # (sign(q)), not a member (q.sign()); the global form is the parity - # surface. celpy can't tell the two call styles apart, so we accept - # both, but the test asserts the upstream-correct global form (see - # cel.py's documented divergences). - Case(name="sign positive", expr='sign(quantity("50k")) == 1', want=True), - Case(name="sign negative", expr='sign(quantity("-50k")) == -1', want=True), - Case(name="sign zero", expr='sign(quantity("0")) == 0', want=True), - # Binary-suffix overflow saturates to int64-max, keeping sign, so - # 8Ei/10Ei/100Ei all compare equal to int64-max (resource.Quantity - # stores BinarySI in an int64). Confirmed against resource.Quantity. - Case( - name="Ei saturates to int64 max", - expr='quantity("8Ei").compareTo(quantity("9223372036854775807")) == 0', - want=True, - ), - Case(name="8Ei equals 10Ei", expr='quantity("8Ei").compareTo(quantity("10Ei")) == 0', want=True), - Case(name="10Ei equals 100Ei", expr='quantity("10Ei").compareTo(quantity("100Ei")) == 0', want=True), - Case( - name="negative Ei saturates", - expr='quantity("-10Ei").compareTo(quantity("-9223372036854775807")) == 0', - want=True, - ), - # 7Ei is below int64-max, so it does NOT saturate and stays less. - Case(name="7Ei below saturation", expr='quantity("7Ei").isLessThan(quantity("8Ei"))', want=True), - # Large DECIMAL-path values do not saturate (only the binary path - # does) and must not raise on nano-rounding. isQuantity must be true - # and the value must round-trip. - Case(name="isQuantity 256E", expr='isQuantity("256E")', want=True), - Case(name="isQuantity 10E", expr='isQuantity("10E")', want=True), - Case( - name="256E value", expr='quantity("256E").compareTo(quantity("256000000000000000000")) == 0', want=True - ), - Case(name="256E greater than 1Ei", expr='quantity("256E").isGreaterThan(quantity("1Ei"))', want=True), - # asInteger / isInteger. - Case(name="as integer", expr='quantity("50k").asInteger() == 50000', want=True), - Case(name="is integer true small", expr='quantity("50").isInteger()', want=True), - Case(name="is integer true big magnitude", expr='quantity("50000000G").isInteger()', want=True), - Case( - name="is integer false overflow", - expr='quantity("9999999999999999999999999999999999999G").isInteger()', - want=False, - ), - # asInteger overflow is a runtime error upstream -> non-match here. - Case( - name="as integer overflow is non-match", - expr='quantity("9999999999999999999999999999999999999G").asInteger() > 0', - want=False, - ), - # asApproximateFloat. - Case(name="as approximate float", expr='quantity("50.703k").asApproximateFloat() == 50703.0', want=True), - # An invalid suffix is a runtime error upstream -> non-match here. - # (Uses a member method upstream accepts, isGreaterThan, so the - # non-match is the parse failure, not a rejected call form.) - Case( - name="invalid suffix is non-match", - expr='quantity("10Mo").isGreaterThan(quantity("1"))', - want=False, - ), - ] - for case in cases: - with self.subTest(case.name): - self.assertEqual(case.want, _eval(case.expr), f"{case.name}: -want, +got") - - -class TestParseRejects(unittest.TestCase): - """parse() rejects what resource.Quantity rejects (drives the non-matches above). - - The bare-suffix row ("Mi") is a DELIBERATE divergence, not parity: upstream - parses most bare suffixes as 0 but inconsistently errors on a few (see - parse()'s docstring). We reject every bare suffix; no device capacity is - ever a bare suffix. - """ - - def test_parse_rejects(self) -> None: - cases = [ - ParseErrCase(name="invalid suffix Mo", input="10Mo"), - ParseErrCase(name="passing regex bad suffix Mm", input="10Mm"), - ParseErrCase(name="capital K", input="200K"), - ParseErrCase(name="comma", input="1,3G"), - ParseErrCase(name="word", input="Three"), - ParseErrCase(name="bare suffix (deliberate divergence)", input="Mi"), - ParseErrCase(name="empty", input=""), - ] - for case in cases: - with self.subTest(case.name), self.assertRaises(ValueError): - quantity.parse(case.input) +QUANTITY_CASES = [ + # parse + isQuantity. + Case(name="parse", expr='quantity("12Mi").compareTo(quantity("12Mi")) == 0', want=True), + Case(name="isQuantity int string", expr='isQuantity("20")', want=True), + Case(name="isQuantity megabytes", expr='isQuantity("20M")', want=True), + Case(name="isQuantity mebibytes", expr='isQuantity("20Mi")', want=True), + Case(name="isQuantity invalid suffix", expr='isQuantity("20Mo")', want=False), + Case(name="isQuantity passing regex bad suffix", expr='isQuantity("10Mm")', want=False), + # resource.Quantity accepts decimal exponents and nano/micro suffixes. + Case(name="isQuantity exponent lowercase", expr='isQuantity("256e3")', want=True), + Case(name="isQuantity exponent uppercase", expr='isQuantity("1E3")', want=True), + Case(name="exponent value", expr='quantity("256e3").compareTo(quantity("256000")) == 0', want=True), + Case(name="isQuantity nano", expr='isQuantity("100n")', want=True), + Case(name="isQuantity micro", expr='isQuantity("100u")', want=True), + Case(name="isQuantity trailing dot", expr='isQuantity("5.")', want=True), + # The quantity() constructor does NOT trim whitespace. + Case(name="isQuantity leading whitespace false", expr='isQuantity(" 5Gi")', want=False), + Case(name="isQuantity trailing whitespace false", expr='isQuantity("5Gi ")', want=False), + # Values equal at nano resolution compare equal (resource.Quantity.Cmp + # rounds to nano). + Case( + name="nano rounding equality", + expr='quantity("0.0000000004").compareTo(quantity("0.000000001")) == 0', + want=True, + ), + # doc-comment isQuantity examples. + Case(name="isQuantity 1.3G", expr='isQuantity("1.3G")', want=True), + Case(name="isQuantity 1.3Gi", expr='isQuantity("1.3Gi")', want=True), + Case(name="isQuantity comma", expr='isQuantity("1,3G")', want=False), + Case(name="isQuantity 10000k", expr='isQuantity("10000k")', want=True), + Case(name="isQuantity capital K", expr='isQuantity("200K")', want=False), + Case(name="isQuantity Three", expr='isQuantity("Three")', want=False), + Case(name="isQuantity bare suffix", expr='isQuantity("Mi")', want=False), + # equality. + Case(name="equality reflexivity", expr='quantity("200M") == quantity("200M")', want=True), + Case( + name="equality symmetry", + expr='quantity("200M") == quantity("0.2G") && quantity("0.2G") == quantity("200M")', + want=True, + ), + Case( + name="equality transitivity", + expr=( + 'quantity("2M") == quantity("0.002G") && quantity("2000k") == quantity("2M") && ' + 'quantity("0.002G") == quantity("2000k")' + ), + want=True, + ), + Case(name="inequality", expr='quantity("200M") == quantity("0.3G")', want=False), + # isLessThan / isGreaterThan. + Case(name="less", expr='quantity("50M").isLessThan(quantity("50Mi"))', want=True), + Case(name="less obvious", expr='quantity("50M").isLessThan(quantity("100M"))', want=True), + Case(name="less false", expr='quantity("100M").isLessThan(quantity("50M"))', want=False), + Case(name="greater", expr='quantity("50Mi").isGreaterThan(quantity("50M"))', want=True), + Case(name="greater obvious", expr='quantity("150Mi").isGreaterThan(quantity("100Mi"))', want=True), + Case(name="greater false", expr='quantity("50M").isGreaterThan(quantity("100M"))', want=False), + # compareTo. + Case(name="compare equal", expr='quantity("200M").compareTo(quantity("0.2G")) == 0', want=True), + Case(name="compare less", expr='quantity("50M").compareTo(quantity("50Mi")) == -1', want=True), + Case(name="compare greater", expr='quantity("50Mi").compareTo(quantity("50M")) == 1', want=True), + # add / sub (quantity and int overloads). + Case(name="add quantity", expr='quantity("50k").add(quantity("20")) == quantity("50.02k")', want=True), + Case(name="add int not less", expr='quantity("50k").add(20).isLessThan(quantity("50020"))', want=False), + Case(name="sub quantity", expr='quantity("50k").sub(quantity("20")) == quantity("49.98k")', want=True), + Case(name="sub int", expr='quantity("50k").sub(20) == quantity("49980")', want=True), + Case( + name="arith chain 1", + expr='quantity("50k").add(20).sub(quantity("100k")).asInteger() == -49980', + want=True, + ), + Case( + name="arith chain 2", + expr='quantity("50k").add(20).sub(quantity("100k")).sub(-50000).asInteger() == 20', + want=True, + ), + # sign (doc comment). Upstream declares sign as a GLOBAL function + # (sign(q)), not a member (q.sign()); the global form is the parity + # surface. celpy can't tell the two call styles apart, so we accept + # both, but the test asserts the upstream-correct global form (see + # cel.py's documented divergences). + Case(name="sign positive", expr='sign(quantity("50k")) == 1', want=True), + Case(name="sign negative", expr='sign(quantity("-50k")) == -1', want=True), + Case(name="sign zero", expr='sign(quantity("0")) == 0', want=True), + # Binary-suffix overflow saturates to int64-max, keeping sign, so + # 8Ei/10Ei/100Ei all compare equal to int64-max (resource.Quantity + # stores BinarySI in an int64). Confirmed against resource.Quantity. + Case( + name="Ei saturates to int64 max", + expr='quantity("8Ei").compareTo(quantity("9223372036854775807")) == 0', + want=True, + ), + Case(name="8Ei equals 10Ei", expr='quantity("8Ei").compareTo(quantity("10Ei")) == 0', want=True), + Case(name="10Ei equals 100Ei", expr='quantity("10Ei").compareTo(quantity("100Ei")) == 0', want=True), + Case( + name="negative Ei saturates", + expr='quantity("-10Ei").compareTo(quantity("-9223372036854775807")) == 0', + want=True, + ), + # 7Ei is below int64-max, so it does NOT saturate and stays less. + Case(name="7Ei below saturation", expr='quantity("7Ei").isLessThan(quantity("8Ei"))', want=True), + # Large DECIMAL-path values do not saturate (only the binary path + # does) and must not raise on nano-rounding. isQuantity must be true + # and the value must round-trip. + Case(name="isQuantity 256E", expr='isQuantity("256E")', want=True), + Case(name="isQuantity 10E", expr='isQuantity("10E")', want=True), + Case(name="256E value", expr='quantity("256E").compareTo(quantity("256000000000000000000")) == 0', want=True), + Case(name="256E greater than 1Ei", expr='quantity("256E").isGreaterThan(quantity("1Ei"))', want=True), + # asInteger / isInteger. + Case(name="as integer", expr='quantity("50k").asInteger() == 50000', want=True), + Case(name="is integer true small", expr='quantity("50").isInteger()', want=True), + Case(name="is integer true big magnitude", expr='quantity("50000000G").isInteger()', want=True), + Case( + name="is integer false overflow", + expr='quantity("9999999999999999999999999999999999999G").isInteger()', + want=False, + ), + # asInteger overflow is a runtime error upstream -> non-match here. + Case( + name="as integer overflow is non-match", + expr='quantity("9999999999999999999999999999999999999G").asInteger() > 0', + want=False, + ), + # asApproximateFloat. + Case(name="as approximate float", expr='quantity("50.703k").asApproximateFloat() == 50703.0', want=True), + # An invalid suffix is a runtime error upstream -> non-match here. + # (Uses a member method upstream accepts, isGreaterThan, so the + # non-match is the parse failure, not a rejected call form.) + Case( + name="invalid suffix is non-match", + expr='quantity("10Mo").isGreaterThan(quantity("1"))', + want=False, + ), +] + + +@pytest.mark.parametrize("case", QUANTITY_CASES, ids=lambda case: case.name) +def test_quantity(case: Case) -> None: + """A quantity CEL expression evaluates as it does upstream.""" + assert _eval(case.expr) == case.want + + +# The bare-suffix row ("Mi") is a DELIBERATE divergence, not parity: upstream +# parses most bare suffixes as 0 but inconsistently errors on a few (see +# parse()'s docstring). We reject every bare suffix; no device capacity is +# ever a bare suffix. +PARSE_REJECTS_CASES = [ + ParseErrCase(name="invalid suffix Mo", input="10Mo"), + ParseErrCase(name="passing regex bad suffix Mm", input="10Mm"), + ParseErrCase(name="capital K", input="200K"), + ParseErrCase(name="comma", input="1,3G"), + ParseErrCase(name="word", input="Three"), + ParseErrCase(name="bare suffix (deliberate divergence)", input="Mi"), + ParseErrCase(name="empty", input=""), +] + + +@pytest.mark.parametrize("case", PARSE_REJECTS_CASES, ids=lambda case: case.name) +def test_parse_rejects(case: ParseErrCase) -> None: + """parse() rejects what resource.Quantity rejects, which drives the non-matches above.""" + with pytest.raises(ValueError, match="invalid quantity"): + quantity.parse(case.input) diff --git a/functions/compose-model-deployment/tests/test_scheduling.py b/functions/compose-model-deployment/tests/test_scheduling.py index 31449aa39..a584c9c9f 100644 --- a/functions/compose-model-deployment/tests/test_scheduling.py +++ b/functions/compose-model-deployment/tests/test_scheduling.py @@ -25,8 +25,8 @@ import dataclasses import datetime -import unittest +import pytest from function import cel, scheduling from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 from models.ai.modelplane.modeldeployment import v1alpha1 as mdv1alpha1 @@ -405,810 +405,803 @@ def _cand( ) -class TestSchedule(unittest.TestCase): - """Tests for scheduling.schedule placement: retain, spread, scale, capacity. - - Deployments use the default single-GPU nodeSelector request (any pool's GPU - device satisfies it), so these focus on placement rather than pool matching; - TestScheduleNodeSelector covers request-to-device matching. - """ - - def test_schedule(self) -> None: - """The scheduler retains existing pins and places new replicas.""" - - cases = [ - Case( - name="no clusters returns no candidates", - deployment=_deployment(), - clusters=[], - all_replicas=[], - want=[], - ), - Case( - name="single ready cluster is picked", - deployment=_deployment(), - clusters=[_cluster("cluster-a")], - all_replicas=[], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default")], - ), - Case( - name="not-ready cluster is not picked for a new replica", - deployment=_deployment(), - clusters=[_cluster("cluster-a", ready=False)], - all_replicas=[], - want=[], - ), - Case( - name="cluster without gateway address is not picked", - deployment=_deployment(), - clusters=[_cluster("cluster-a", gateway_hostname="")], - all_replicas=[], - want=[], - ), - Case( - name="multi-node deployment needs enough nodes", - deployment=_deployment(pipeline=4), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], - all_replicas=[], - want=[], - ), - Case( - name="existing replica is retained on its pinned cluster", - deployment=_deployment(), - clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - # cluster-a wins even though cluster-b is also viable. The pin - # still matches, so it's retained with its resolved pool/requests. - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="degraded pinned cluster is retained with empty gateway", - deployment=_deployment(), - clusters=[_cluster("cluster-a", ready=False, gateway_hostname="")], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="")], - ), - Case( - name="deleted pinned cluster triggers re-placement", - deployment=_deployment(), - clusters=[_cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com")], - all_replicas=[_replica("my-model", "cluster-a")], - want=[_cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default")], - ), - Case( - name="scale up places new replicas on additional clusters", - deployment=_deployment(replicas=2), - clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ], - all_replicas=[_replica("my-model", "cluster-a")], - want=[ - _cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com"), - _cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="scale up with no extra capacity returns only retained", - deployment=_deployment(replicas=2), - # Single-node pool, already filled by the retained replica, so no - # second replica can be placed - not even on the same cluster. - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default")], - ), - Case( - name="two replicas pack onto one cluster when it is the only option", - deployment=_deployment(replicas=2), - # One cluster, a 2-node pool, two 1-node replicas. With nowhere - # to spread, both pack onto cluster-a at indices 0 and 1. - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], - all_replicas=[], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - ], - ), - Case( - name="two replicas spread across two clusters before packing", - deployment=_deployment(replicas=2), - # Both clusters can hold two replicas, but we prefer one each. - clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=2)]), - _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=2)], - ), - ], - all_replicas=[], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="three replicas spread first then pack the remainder", - deployment=_deployment(replicas=3), - # Two clusters, plenty of room. Spread gives a, b one each, then - # the third lands back on cluster-a (lowest load, name tiebreak). - clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), - _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], - ), - ], - all_replicas=[], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="capacity forces packing past the spread preference", - deployment=_deployment(replicas=3), - # cluster-b holds one replica; cluster-a has room for the rest. - # Spread puts one on each, then the third can't fit on b (full), - # so it packs onto a. - clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), - _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=1)], - ), - ], - all_replicas=[], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="new replica spreads onto an empty cluster before doubling up", - deployment=_deployment(replicas=2), - # cluster-a already hosts a replica; cluster-b is empty. The new - # replica prefers empty cluster-b over packing onto a. - clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), - _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], - ), - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="new replica takes the lowest free index on a packed cluster", - deployment=_deployment(replicas=3), - # Only cluster-a exists, already hosting indices 0 and 2 (1 was - # deleted). The new replica fills the gap at index 1. - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], - all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=2), - ], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=2, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - ], - ), - Case( - name="scale down packs off by dropping the highest index first", - deployment=_deployment(replicas=2), - # cluster-a hosts indices 0 and 1; cluster-b hosts index 0. Three - # replicas, want two. Highest index (a/1) is dropped, keeping the - # spread across a/0 and b/0. - clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), - _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], - ), - ], - all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=1), - _replica_with_pool("my-model", "cluster-b", pool="default", index=0), - ], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="retained replica is charged at its own node cost, not the new shape", - # The deployment's workers grew to pipeline=4 (4 nodes/replica), - # but the existing replica was created at pipeline=2 and is - # retained (no nodeSelector change rolls it). It still consumes - # only its original 2 nodes. The pool has 6, so a second replica - # at the new 4-node cost must still fit (6 - 2 = 4). Regression: - # charging the retained replica at the new shape (4) would leave - # 2 free and wrongly refuse the placement. - deployment=_deployment(replicas=2, pipeline=4), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=6)])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default", pipeline=2)], - # The retained replica is re-stamped to the deployment's current - # pipeline=4 shape but still charged its observed 2 nodes in the - # ledger. - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pipeline=4), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pipeline=4), - ], - ), - Case( - name="scale down drops from the most-loaded cluster to preserve spread", - deployment=_deployment(replicas=2), - # cluster-a hosts two replicas, cluster-b one. Scaling 3->2 must - # drop a's extra (a/1), NOT b's sole replica - otherwise we'd - # leave a packed and b empty, the opposite of spread. b's index - # is 3 (higher than a/1) to prove we drop by cluster load, not by - # a global index comparison. - clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), - _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], - ), - ], - all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=1), - _replica_with_pool("my-model", "cluster-b", pool="default", index=3), - ], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=3, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="co-located replicas are both retained across a reconcile", - deployment=_deployment(replicas=2), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], - all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=1), - ], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - ], - ), - Case( - name="scale down across clusters drops higher cluster name at equal index", - deployment=_deployment(replicas=1), - clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ], - all_replicas=[ - _replica("my-model", "cluster-b"), - _replica("my-model", "cluster-a"), - ], - # Both at index 0, so the (index, name) tiebreak keeps cluster-a. - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="new placement is alphabetical for determinism", - deployment=_deployment(replicas=2), - clusters=[ - _cluster("cluster-c", gateway_hostname="cluster-c.clusters.example.com"), - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), +# Deployments use the default single-GPU nodeSelector request (any pool's GPU +# device satisfies it), so these focus on placement rather than pool matching; +# NODE_SELECTOR_CASES covers request-to-device matching. +SCHEDULE_CASES = [ + Case( + name="no clusters returns no candidates", + deployment=_deployment(), + clusters=[], + all_replicas=[], + want=[], + ), + Case( + name="single ready cluster is picked", + deployment=_deployment(), + clusters=[_cluster("cluster-a")], + all_replicas=[], + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default")], + ), + Case( + name="not-ready cluster is not picked for a new replica", + deployment=_deployment(), + clusters=[_cluster("cluster-a", ready=False)], + all_replicas=[], + want=[], + ), + Case( + name="cluster without gateway address is not picked", + deployment=_deployment(), + clusters=[_cluster("cluster-a", gateway_hostname="")], + all_replicas=[], + want=[], + ), + Case( + name="multi-node deployment needs enough nodes", + deployment=_deployment(pipeline=4), + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], + all_replicas=[], + want=[], + ), + Case( + name="existing replica is retained on its pinned cluster", + deployment=_deployment(), + clusters=[ + _cluster("cluster-a"), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + # cluster-a wins even though cluster-b is also viable. The pin + # still matches, so it's retained with its resolved pool/requests. + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], + ), + Case( + name="degraded pinned cluster is retained with empty gateway", + deployment=_deployment(), + clusters=[_cluster("cluster-a", ready=False, gateway_hostname="")], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + want=[_cand(name="cluster-a", gateway_hostname="")], + ), + Case( + name="deleted pinned cluster triggers re-placement", + deployment=_deployment(), + clusters=[_cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com")], + all_replicas=[_replica("my-model", "cluster-a")], + want=[_cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default")], + ), + Case( + name="scale up places new replicas on additional clusters", + deployment=_deployment(replicas=2), + clusters=[ + _cluster("cluster-a"), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ], + all_replicas=[_replica("my-model", "cluster-a")], + want=[ + _cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com"), + _cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="scale up with no extra capacity returns only retained", + deployment=_deployment(replicas=2), + # Single-node pool, already filled by the retained replica, so no + # second replica can be placed - not even on the same cluster. + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default")], + ), + Case( + name="two replicas pack onto one cluster when it is the only option", + deployment=_deployment(replicas=2), + # One cluster, a 2-node pool, two 1-node replicas. With nowhere + # to spread, both pack onto cluster-a at indices 0 and 1. + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], + all_replicas=[], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + ], + ), + Case( + name="two replicas spread across two clusters before packing", + deployment=_deployment(replicas=2), + # Both clusters can hold two replicas, but we prefer one each. + clusters=[ + _cluster("cluster-a", pools=[_pool("default", nodes=2)]), + _cluster( + "cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + pools=[_pool("default", nodes=2)], + ), + ], + all_replicas=[], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="three replicas spread first then pack the remainder", + deployment=_deployment(replicas=3), + # Two clusters, plenty of room. Spread gives a, b one each, then + # the third lands back on cluster-a (lowest load, name tiebreak). + clusters=[ + _cluster("cluster-a", pools=[_pool("default", nodes=4)]), + _cluster( + "cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + pools=[_pool("default", nodes=4)], + ), + ], + all_replicas=[], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="capacity forces packing past the spread preference", + deployment=_deployment(replicas=3), + # cluster-b holds one replica; cluster-a has room for the rest. + # Spread puts one on each, then the third can't fit on b (full), + # so it packs onto a. + clusters=[ + _cluster("cluster-a", pools=[_pool("default", nodes=4)]), + _cluster( + "cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + pools=[_pool("default", nodes=1)], + ), + ], + all_replicas=[], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="new replica spreads onto an empty cluster before doubling up", + deployment=_deployment(replicas=2), + # cluster-a already hosts a replica; cluster-b is empty. The new + # replica prefers empty cluster-b over packing onto a. + clusters=[ + _cluster("cluster-a", pools=[_pool("default", nodes=4)]), + _cluster( + "cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + pools=[_pool("default", nodes=4)], + ), + ], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="new replica takes the lowest free index on a packed cluster", + deployment=_deployment(replicas=3), + # Only cluster-a exists, already hosting indices 0 and 2 (1 was + # deleted). The new replica fills the gap at index 1. + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], + all_replicas=[ + _replica_with_pool("my-model", "cluster-a", pool="default", index=0), + _replica_with_pool("my-model", "cluster-a", pool="default", index=2), + ], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-a", index=2, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + ], + ), + Case( + name="scale down packs off by dropping the highest index first", + deployment=_deployment(replicas=2), + # cluster-a hosts indices 0 and 1; cluster-b hosts index 0. Three + # replicas, want two. Highest index (a/1) is dropped, keeping the + # spread across a/0 and b/0. + clusters=[ + _cluster("cluster-a", pools=[_pool("default", nodes=4)]), + _cluster( + "cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + pools=[_pool("default", nodes=4)], + ), + ], + all_replicas=[ + _replica_with_pool("my-model", "cluster-a", pool="default", index=0), + _replica_with_pool("my-model", "cluster-a", pool="default", index=1), + _replica_with_pool("my-model", "cluster-b", pool="default", index=0), + ], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="retained replica is charged at its own node cost, not the new shape", + # The deployment's workers grew to pipeline=4 (4 nodes/replica), + # but the existing replica was created at pipeline=2 and is + # retained (no nodeSelector change rolls it). It still consumes + # only its original 2 nodes. The pool has 6, so a second replica + # at the new 4-node cost must still fit (6 - 2 = 4). Regression: + # charging the retained replica at the new shape (4) would leave + # 2 free and wrongly refuse the placement. + deployment=_deployment(replicas=2, pipeline=4), + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=6)])], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default", pipeline=2)], + # The retained replica is re-stamped to the deployment's current + # pipeline=4 shape but still charged its observed 2 nodes in the + # ledger. + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pipeline=4), + _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pipeline=4), + ], + ), + Case( + name="scale down drops from the most-loaded cluster to preserve spread", + deployment=_deployment(replicas=2), + # cluster-a hosts two replicas, cluster-b one. Scaling 3->2 must + # drop a's extra (a/1), NOT b's sole replica - otherwise we'd + # leave a packed and b empty, the opposite of spread. b's index + # is 3 (higher than a/1) to prove we drop by cluster load, not by + # a global index comparison. + clusters=[ + _cluster("cluster-a", pools=[_pool("default", nodes=4)]), + _cluster( + "cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + pools=[_pool("default", nodes=4)], + ), + ], + all_replicas=[ + _replica_with_pool("my-model", "cluster-a", pool="default", index=0), + _replica_with_pool("my-model", "cluster-a", pool="default", index=1), + _replica_with_pool("my-model", "cluster-b", pool="default", index=3), + ], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-b", index=3, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="co-located replicas are both retained across a reconcile", + deployment=_deployment(replicas=2), + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], + all_replicas=[ + _replica_with_pool("my-model", "cluster-a", pool="default", index=0), + _replica_with_pool("my-model", "cluster-a", pool="default", index=1), + ], + want=[ + _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + ], + ), + Case( + name="scale down across clusters drops higher cluster name at equal index", + deployment=_deployment(replicas=1), + clusters=[ + _cluster("cluster-a"), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ], + all_replicas=[ + _replica("my-model", "cluster-b"), + _replica("my-model", "cluster-a"), + ], + # Both at index 0, so the (index, name) tiebreak keeps cluster-a. + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], + ), + Case( + name="new placement is alphabetical for determinism", + deployment=_deployment(replicas=2), + clusters=[ + _cluster("cluster-c", gateway_hostname="cluster-c.clusters.example.com"), + _cluster("cluster-a"), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ], + all_replicas=[], + want=[ + _cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ], + ), + Case( + name="other deployment's replicas consume node capacity", + deployment=_deployment(pipeline=1), + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], + # other-model occupies the single node on cluster-a. + all_replicas=[_replica("other-model", "cluster-a")], + want=[], + ), + Case( + name="our own observed replicas don't double-count against us", + deployment=_deployment(pipeline=1), + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + # Retained on its pin: the single node it already occupies isn't + # charged against itself, so it stays rather than being evicted. + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], + ), + Case( + name="replica labeled for our deployment but pinned to unknown cluster is ignored", + deployment=_deployment(), + clusters=[_cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com")], + all_replicas=[_replica("my-model", "cluster-a")], + want=[_cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default")], + ), + Case( + name="another deployment pinned to a deleted pool consumes no capacity", + # other-model is pinned to pool "gone", which the cluster no + # longer publishes. Its pods are pinned to a node label no node + # carries, so they're unschedulable and occupy nothing. The one + # published node on "frontier" is therefore free for our replica. + # Charging the unattributable replica would wrongly report the + # cluster full. + deployment=_deployment(requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], + all_replicas=[_replica_with_pool("other-model", "cluster-a", pool="gone")], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved()], + ) + ], + ), + Case( + name="colliding (cluster, index) retains deterministically by replica name", + # Two of our replicas collide on (cluster-a, index 0) with + # different pinned pools. Retain keeps the first by replica name + # (my-model-cluster-a-0 on "a" sorts before the "-dup" replica on + # "b"), independent of input order, so the schedule is a function + # of state not of delivery order. Both pools match, so either + # would be a valid placement - only determinism is under test. + deployment=_deployment(requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("a", devices=[_gpu_device()]), + _pool("b", devices=[_gpu_device()]), ], - all_replicas=[], - want=[ - _cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ) + ], + all_replicas=[ + _collision_replica("my-model-cluster-a-0-dup", "cluster-a", pool="b", index=0), + _replica_with_pool("my-model", "cluster-a", pool="a", index=0), + ], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="a", + device_requests=[_resolved()], + ) + ], + ), +] + + +@pytest.mark.parametrize("case", SCHEDULE_CASES, ids=lambda case: case.name) +def test_schedule(case: Case) -> None: + """The scheduler retains existing pins and places new replicas.""" + got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) + assert got == case.want + + +# A caller passes fill=False when it can't yet trust the candidate set (for the +# ModelDeployment, when a referenced ModelCache is unresolved). Retain runs +# unconditionally; only the placement of new replicas is held. +FILL_FALSE_CASES = [ + Case( + name="no replicas yet: nothing is placed", + deployment=_deployment(), + clusters=[_cluster("cluster-a")], + all_replicas=[], + want=[], + ), + Case( + name="existing replica is retained despite fill=False", + deployment=_deployment(), + clusters=[_cluster("cluster-a")], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], + ), + Case( + name="scale-up shortfall is not filled, only the retained replica remains", + deployment=_deployment(replicas=3), + clusters=[ + _cluster("cluster-a"), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], + ), +] + + +@pytest.mark.parametrize("case", FILL_FALSE_CASES, ids=lambda case: case.name) +def test_fill_false_is_retain_only(case: Case) -> None: + """fill=False retains existing replicas but places no new ones.""" + got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas, fill=False) + assert got == case.want + + +# nodeSelector device-request matching and pool pinning. +NODE_SELECTOR_CASES = [ + Case( + name="matching request picks the cluster and records the pool", + deployment=_deployment(requests=[_request(cel_exprs=[_MEM_141])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], + all_replicas=[], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved()], + ) + ], + ), + Case( + name="non-matching request filters the cluster out", + deployment=_deployment(requests=[_request(cel_exprs=[_MEM_200])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], + all_replicas=[], + want=[], + ), + Case( + name="device count not covered filters out", + # Request 8 GPUs, pool device has only 4. + deployment=_deployment(requests=[_request(count=8, cel_exprs=[_MEM_141])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=4)])])], + all_replicas=[], + want=[], + ), + Case( + name="published device count of zero satisfies no request", + # A pool device published with count 0 must read as "none + # available", not default to 1. Regression: `d.count or 1` + # treated 0 as 1 and placed a replica whose ResourceClaim no + # device could satisfy. The status schema permits 0 even though + # an InferenceClass device count is floored at 1. + deployment=_deployment(requests=[_request(count=1, cel_exprs=[_MEM_141])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=0)])])], + all_replicas=[], + want=[], + ), + Case( + name="published pool node count of zero hosts nothing", + # An autoscaled-to-zero pool has a matching GPU device but no + # nodes, so it can host no replica. + deployment=_deployment(requests=[_request(count=1, cel_exprs=[_MEM_141])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=0)])], + all_replicas=[], + want=[], + ), + Case( + name="synthetic NIC device matches but is not in resolved requests", + deployment=_deployment( + requests=[ + _request(name="gpu", cel_exprs=[_MEM_141]), + _request(name="nic", cel_exprs=[_IB]), + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[_pool("frontier", devices=[_gpu_device(), _nic_device()])], + ) + ], + all_replicas=[], + # Only the claim: DRA gpu request is resolved; the synthetic nic + # matched for scheduling but isn't claimed. + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved(name="gpu")], + ) + ], + ), + Case( + name="multi-device: missing NIC filters the pool out", + deployment=_deployment( + requests=[ + _request(name="gpu", cel_exprs=[_MEM_141]), + _request(name="nic", cel_exprs=[_IB]), + ] + ), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device()])])], + all_replicas=[], + want=[], + ), + Case( + name="two requests cannot both claim one single-count device", + # Two distinct requests, each matching the same single GPU + # device. DRA allocates distinct devices per request, so a + # count:1 device can satisfy only one. The pool must not match. + deployment=_deployment( + requests=[ + _request(name="gpu-a", cel_exprs=[_MEM_141]), + _request(name="gpu-b", cel_exprs=[_MEM_141]), + ] + ), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=1)])])], + all_replicas=[], + want=[], + ), + Case( + name="two requests against one device must fit within its count", + # Two count:5 requests need 10 GPUs total; the device has 8. + # Capacity is consumed across requests, so the pool must not + # match (regression: an earlier version checked each request + # against the full device count independently). + deployment=_deployment( + requests=[ + _request(name="gpu-a", count=5, cel_exprs=[_MEM_141]), + _request(name="gpu-b", count=5, cel_exprs=[_MEM_141]), + ] + ), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=8)])])], + all_replicas=[], + want=[], + ), + Case( + name="two requests sharing a device fit when count covers both", + # 8-GPU device, two count:4 requests = 8 total. Both resolve. + deployment=_deployment( + requests=[ + _request(name="gpu-a", count=4, cel_exprs=[_MEM_141]), + _request(name="gpu-b", count=4, cel_exprs=[_MEM_141]), + ] + ), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=8)])])], + all_replicas=[], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[ + _resolved(name="gpu-a", count=4), + _resolved(name="gpu-b", count=4), ], - ), - Case( - name="other deployment's replicas consume node capacity", - deployment=_deployment(pipeline=1), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], - # other-model occupies the single node on cluster-a. - all_replicas=[_replica("other-model", "cluster-a")], - want=[], - ), - Case( - name="our own observed replicas don't double-count against us", - deployment=_deployment(pipeline=1), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - # Retained on its pin: the single node it already occupies isn't - # charged against itself, so it stays rather than being evicted. - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="replica labeled for our deployment but pinned to unknown cluster is ignored", - deployment=_deployment(), - clusters=[_cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com")], - all_replicas=[_replica("my-model", "cluster-a")], - want=[_cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default")], - ), - Case( - name="another deployment pinned to a deleted pool consumes no capacity", - # other-model is pinned to pool "gone", which the cluster no - # longer publishes. Its pods are pinned to a node label no node - # carries, so they're unschedulable and occupy nothing. The one - # published node on "frontier" is therefore free for our replica. - # Charging the unattributable replica would wrongly report the - # cluster full. - deployment=_deployment(requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], - all_replicas=[_replica_with_pool("other-model", "cluster-a", pool="gone")], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], - ) + ) + ], + ), + Case( + name="first matching pool wins (deterministic)", + # Both pools carry a claimable GPU; the synthetic NIC's link type + # is the discriminator. Only the infiniband pool satisfies the + # nic selector, so it wins regardless of pool order. + deployment=_deployment( + requests=[ + _request(name="gpu", cel_exprs=[_MEM_141]), + _request(name="nic", cel_exprs=[_IB]), + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("dev", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")]), + _pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), ], - ), - Case( - name="colliding (cluster, index) retains deterministically by replica name", - # Two of our replicas collide on (cluster-a, index 0) with - # different pinned pools. Retain keeps the first by replica name - # (my-model-cluster-a-0 on "a" sorts before the "-dup" replica on - # "b"), independent of input order, so the schedule is a function - # of state not of delivery order. Both pools match, so either - # would be a valid placement - only determinism is under test. - deployment=_deployment(requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("a", devices=[_gpu_device()]), - _pool("b", devices=[_gpu_device()]), - ], - ) + ) + ], + all_replicas=[], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved(name="gpu")], + ) + ], + ), + Case( + name="synthetic-only selector leaves nothing to claim, pool ineligible", + # The sole request matches a synthetic NIC. The replica's serving + # workload would have no ResourceClaim to bind GPUs through, so + # the pool is not a viable host and nothing is scheduled. + deployment=_deployment(requests=[_request(name="nic", cel_exprs=[_IB])]), + clusters=[ + _cluster( + "cluster-a", + pools=[_pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")])], + ) + ], + all_replicas=[], + want=[], + ), + Case( + name="retained replica keeps its pinned pool", + deployment=_deployment(requests=[_request(cel_exprs=[_MEM_141])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="frontier")], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved()], + ) + ], + ), + Case( + name="selector drift re-places replica onto a now-matching pool", + # A claimable GPU keeps both pools viable hosts; the synthetic + # NIC's link type is the drifting discriminator. + deployment=_deployment( + requests=[ + _request(name="gpu", cel_exprs=[_MEM_141]), + _request(name="nic", cel_exprs=[_IB]), + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("a", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")]), + _pool("b", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), ], - all_replicas=[ - _collision_replica("my-model-cluster-a-0-dup", "cluster-a", pool="b", index=0), - _replica_with_pool("my-model", "cluster-a", pool="a", index=0), + ) + ], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="b", + device_requests=[_resolved(name="gpu")], + ) + ], + ), + Case( + name="pinned pool that still matches stays pinned (attribute drift is sticky)", + deployment=_deployment( + requests=[ + _request(name="gpu", cel_exprs=[_MEM_141]), + _request(name="nic", cel_exprs=[_IB]), + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("a", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), + _pool("b", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), ], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="a", - device_requests=[_resolved()], - ) + ) + ], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="a", + device_requests=[_resolved(name="gpu")], + ) + ], + ), + Case( + name="no matching pool anywhere drops the replica entirely", + deployment=_deployment( + requests=[ + _request(name="gpu", cel_exprs=[_MEM_141]), + _request(name="nic", cel_exprs=[_IB]), + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[_pool("a", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")])], + ) + ], + all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], + want=[], + ), + Case( + name="replica with no pool pin is re-placed when a selector now applies", + deployment=_deployment( + requests=[ + _request(name="gpu", cel_exprs=[_MEM_141]), + _request(name="nic", cel_exprs=[_IB]), + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[_pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")])], + ) + ], + all_replicas=[_replica("my-model", "cluster-a")], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved(name="gpu")], + ) + ], + ), + Case( + name="dropping a non-matching replica frees its node for the refill", + # a/0 is pinned to a pool that still matches (retained). a/1 is + # pinned to a pool no longer published, so it's dropped and will + # be re-placed. The pool has just 2 nodes; both are notionally in + # use by a/0 and a/1. The refill must see a/1's node freeing up + # (it's being deleted) and re-place onto frontier at index 1. + # Regression: the ledger must not charge dropped replicas. + deployment=_deployment(replicas=2, requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=2)])], + all_replicas=[ + _replica_with_pool("my-model", "cluster-a", pool="frontier", index=0), + _replica_with_pool("my-model", "cluster-a", pool="gone", index=1), + ], + want=[ + _cand( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved()], + ), + _cand( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + pool="frontier", + device_requests=[_resolved()], + ), + ], + ), + Case( + name="device count is checked against the pinned pool, not a cluster-wide sum", + # Request 8 GPUs. Pool 'a' has 4/node (doesn't fit); pool 'b' + # has 8 and does. The replica must pin to 'b'. + deployment=_deployment(requests=[_request(count=8, cel_exprs=[_MEM_141])]), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("a", devices=[_gpu_device(count=4)]), + _pool("b", devices=[_gpu_device(count=8)]), ], - ), - ] + ) + ], + all_replicas=[], + want=[ + _cand( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pool="b", + device_requests=[_resolved(count=8)], + ) + ], + ), +] - for case in cases: - with self.subTest(case.name): - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) - self.assertEqual(case.want, got, f"{case.name}: -want, +got") - - def test_fill_false_is_retain_only(self) -> None: - """fill=False retains existing replicas but places no new ones. - - A caller passes fill=False when it can't yet trust the candidate set - (for the ModelDeployment, when a referenced ModelCache is unresolved). - Retain runs unconditionally; only the placement of new replicas is held. - """ - cases = [ - Case( - name="no replicas yet: nothing is placed", - deployment=_deployment(), - clusters=[_cluster("cluster-a")], - all_replicas=[], - want=[], - ), - Case( - name="existing replica is retained despite fill=False", - deployment=_deployment(), - clusters=[_cluster("cluster-a")], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="scale-up shortfall is not filled, only the retained replica remains", - deployment=_deployment(replicas=3), - clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - ] - for case in cases: - with self.subTest(case.name): - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas, fill=False) - self.assertEqual(case.want, got, f"{case.name}: -want, +got") - - -class TestScheduleNodeSelector(unittest.TestCase): - """Tests for nodeSelector device-request matching and pool pinning.""" - - def test_node_selector(self) -> None: - cases = [ - Case( - name="matching request picks the cluster and records the pool", - deployment=_deployment(requests=[_request(cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], - all_replicas=[], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], - ) - ], - ), - Case( - name="non-matching request filters the cluster out", - deployment=_deployment(requests=[_request(cel_exprs=[_MEM_200])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], - all_replicas=[], - want=[], - ), - Case( - name="device count not covered filters out", - # Request 8 GPUs, pool device has only 4. - deployment=_deployment(requests=[_request(count=8, cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=4)])])], - all_replicas=[], - want=[], - ), - Case( - name="published device count of zero satisfies no request", - # A pool device published with count 0 must read as "none - # available", not default to 1. Regression: `d.count or 1` - # treated 0 as 1 and placed a replica whose ResourceClaim no - # device could satisfy. The status schema permits 0 even though - # an InferenceClass device count is floored at 1. - deployment=_deployment(requests=[_request(count=1, cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=0)])])], - all_replicas=[], - want=[], - ), - Case( - name="published pool node count of zero hosts nothing", - # An autoscaled-to-zero pool has a matching GPU device but no - # nodes, so it can host no replica. - deployment=_deployment(requests=[_request(count=1, cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=0)])], - all_replicas=[], - want=[], - ), - Case( - name="synthetic NIC device matches but is not in resolved requests", - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device()])], - ) - ], - all_replicas=[], - # Only the claim: DRA gpu request is resolved; the synthetic nic - # matched for scheduling but isn't claimed. - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved(name="gpu")], - ) - ], - ), - Case( - name="multi-device: missing NIC filters the pool out", - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device()])])], - all_replicas=[], - want=[], - ), - Case( - name="two requests cannot both claim one single-count device", - # Two distinct requests, each matching the same single GPU - # device. DRA allocates distinct devices per request, so a - # count:1 device can satisfy only one. The pool must not match. - deployment=_deployment( - requests=[ - _request(name="gpu-a", cel_exprs=[_MEM_141]), - _request(name="gpu-b", cel_exprs=[_MEM_141]), - ] - ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=1)])])], - all_replicas=[], - want=[], - ), - Case( - name="two requests against one device must fit within its count", - # Two count:5 requests need 10 GPUs total; the device has 8. - # Capacity is consumed across requests, so the pool must not - # match (regression: an earlier version checked each request - # against the full device count independently). - deployment=_deployment( - requests=[ - _request(name="gpu-a", count=5, cel_exprs=[_MEM_141]), - _request(name="gpu-b", count=5, cel_exprs=[_MEM_141]), - ] - ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=8)])])], - all_replicas=[], - want=[], - ), - Case( - name="two requests sharing a device fit when count covers both", - # 8-GPU device, two count:4 requests = 8 total. Both resolve. - deployment=_deployment( - requests=[ - _request(name="gpu-a", count=4, cel_exprs=[_MEM_141]), - _request(name="gpu-b", count=4, cel_exprs=[_MEM_141]), - ] - ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=8)])])], - all_replicas=[], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[ - _resolved(name="gpu-a", count=4), - _resolved(name="gpu-b", count=4), - ], - ) - ], - ), - Case( - name="first matching pool wins (deterministic)", - # Both pools carry a claimable GPU; the synthetic NIC's link type - # is the discriminator. Only the infiniband pool satisfies the - # nic selector, so it wins regardless of pool order. - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("dev", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")]), - _pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), - ], - ) - ], - all_replicas=[], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved(name="gpu")], - ) - ], - ), - Case( - name="synthetic-only selector leaves nothing to claim, pool ineligible", - # The sole request matches a synthetic NIC. The replica's serving - # workload would have no ResourceClaim to bind GPUs through, so - # the pool is not a viable host and nothing is scheduled. - deployment=_deployment(requests=[_request(name="nic", cel_exprs=[_IB])]), - clusters=[ - _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")])], - ) - ], - all_replicas=[], - want=[], - ), - Case( - name="retained replica keeps its pinned pool", - deployment=_deployment(requests=[_request(cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="frontier")], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], - ) - ], - ), - Case( - name="selector drift re-places replica onto a now-matching pool", - # A claimable GPU keeps both pools viable hosts; the synthetic - # NIC's link type is the drifting discriminator. - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("a", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")]), - _pool("b", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), - ], - ) - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="b", - device_requests=[_resolved(name="gpu")], - ) - ], - ), - Case( - name="pinned pool that still matches stays pinned (attribute drift is sticky)", - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("a", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), - _pool("b", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), - ], - ) - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="a", - device_requests=[_resolved(name="gpu")], - ) - ], - ), - Case( - name="no matching pool anywhere drops the replica entirely", - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[_pool("a", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")])], - ) - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], - want=[], - ), - Case( - name="replica with no pool pin is re-placed when a selector now applies", - deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")])], - ) - ], - all_replicas=[_replica("my-model", "cluster-a")], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved(name="gpu")], - ) - ], - ), - Case( - name="dropping a non-matching replica frees its node for the refill", - # a/0 is pinned to a pool that still matches (retained). a/1 is - # pinned to a pool no longer published, so it's dropped and will - # be re-placed. The pool has just 2 nodes; both are notionally in - # use by a/0 and a/1. The refill must see a/1's node freeing up - # (it's being deleted) and re-place onto frontier at index 1. - # Regression: the ledger must not charge dropped replicas. - deployment=_deployment(replicas=2, requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=2)])], - all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="frontier", index=0), - _replica_with_pool("my-model", "cluster-a", pool="gone", index=1), - ], - want=[ - _cand( - name="cluster-a", - index=0, - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], - ), - _cand( - name="cluster-a", - index=1, - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], - ), - ], - ), - Case( - name="device count is checked against the pinned pool, not a cluster-wide sum", - # Request 8 GPUs. Pool 'a' has 4/node (doesn't fit); pool 'b' - # has 8 and does. The replica must pin to 'b'. - deployment=_deployment(requests=[_request(count=8, cel_exprs=[_MEM_141])]), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("a", devices=[_gpu_device(count=4)]), - _pool("b", devices=[_gpu_device(count=8)]), - ], - ) - ], - all_replicas=[], - want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="b", - device_requests=[_resolved(count=8)], - ) - ], - ), - ] +@pytest.mark.parametrize("case", NODE_SELECTOR_CASES, ids=lambda case: case.name) +def test_node_selector(case: Case) -> None: + """The scheduler places replicas only on pools whose devices satisfy the nodeSelector.""" + got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) + assert got == case.want - for case in cases: - with self.subTest(case.name): - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) - self.assertEqual(case.want, got, f"{case.name}: -want, +got") - def test_invalid_cel_raises(self) -> None: - """A malformed expression raises CELCompileError (caller handles it).""" - deployment = _deployment(requests=[_request(cel_exprs=["this is ) not valid ("])]) - with self.assertRaises(cel.CELCompileError): - scheduling.schedule(deployment, [_cluster("cluster-a", pools=[_pool("frontier")])], []) +def test_node_selector_invalid_cel_raises() -> None: + """A malformed expression raises CELCompileError, which the caller handles.""" + deployment = _deployment(requests=[_request(cel_exprs=["this is ) not valid ("])]) + with pytest.raises(cel.CELCompileError, match=r"this is \) not valid \("): + scheduling.schedule(deployment, [_cluster("cluster-a", pools=[_pool("frontier")])], []) def _gang( @@ -1231,561 +1224,558 @@ def _gang( return mdv1alpha1.Engine(name=_ENGINE, members=[leader, worker]) -class TestScheduleMembers(unittest.TestCase): - """Tests for per-member placement: single-pool engines, rejection when no - pool fits, and claimless ride-along members.""" - - def test_members(self) -> None: - cases = [ - Case( - name="a single pool satisfying every member hosts the whole engine", - # The leader's request matches both pools; the worker's only - # matches big. The whole-engine pass must put both members on - # big - the one pool that satisfies them all. - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_200])], - ) - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("small", devices=[_gpu_device(memory="141Gi")]), - _pool("big", devices=[_gpu_device(memory="200Gi")]), - ], - ) +# Per-member placement: single-pool engines, rejection when no pool fits, and +# claimless ride-along members. +MEMBERS_CASES = [ + Case( + name="a single pool satisfying every member hosts the whole engine", + # The leader's request matches both pools; the worker's only + # matches big. The whole-engine pass must put both members on + # big - the one pool that satisfies them all. + deployment=_deployment( + engines=[ + _gang( + [_request(cel_exprs=[_MEM_141])], + [_request(cel_exprs=[_MEM_200])], + ) + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("small", devices=[_gpu_device(memory="141Gi")]), + _pool("big", devices=[_gpu_device(memory="200Gi")]), ], - all_replicas=[], - want=[ - scheduling.Candidate( - name="cluster-a", - index=0, - gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement( - role="Leader", pool="big", device_requests=[_resolved()] - ), - scheduling.MemberPlacement( - role="Worker", - pool="big", - device_requests=[_resolved(cel_exprs=[_MEM_200])], - ), - ], - ) + ) + ], + all_replicas=[], + want=[ + scheduling.Candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + engines=[ + scheduling.EnginePlacement( + name=_ENGINE, + members=[ + scheduling.MemberPlacement(role="Leader", pool="big", device_requests=[_resolved()]), + scheduling.MemberPlacement( + role="Worker", + pool="big", + device_requests=[_resolved(cel_exprs=[_MEM_200])], + ), ], ) ], - ), - Case( - name="members no single pool satisfies are not scheduled", - # The leader only fits big (>= 200Gi); the worker only fits - # small (< 200Gi). No single pool satisfies both. The scheduler - # never splits an engine across pools - it can't tell whether - # big and small share a fabric - so the engine is rejected and - # the replica goes unplaced (#149). - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_200])], - [_request(cel_exprs=[_MEM_LT_200])], - ) - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("small", devices=[_gpu_device(memory="141Gi")]), - _pool("big", devices=[_gpu_device(memory="200Gi")]), - ], - ) + ) + ], + ), + Case( + name="members no single pool satisfies are not scheduled", + # The leader only fits big (>= 200Gi); the worker only fits + # small (< 200Gi). No single pool satisfies both. The scheduler + # never splits an engine across pools - it can't tell whether + # big and small share a fabric - so the engine is rejected and + # the replica goes unplaced (#149). + deployment=_deployment( + engines=[ + _gang( + [_request(cel_exprs=[_MEM_200])], + [_request(cel_exprs=[_MEM_LT_200])], + ) + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("small", devices=[_gpu_device(memory="141Gi")]), + _pool("big", devices=[_gpu_device(memory="200Gi")]), ], - all_replicas=[], - want=[], - ), - Case( - name="a gang too big for its only matching pool is rejected", - # Both members match only big (>= 141Gi); small (40Gi) matches - # neither. big has one free node but the gang needs two. The - # engine doesn't fit any single pool, so it's rejected; with big - # the only cluster the replica goes unplaced. - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_141])], - ) - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool("small", nodes=8, devices=[_gpu_device(memory="40Gi")]), - _pool("big", nodes=1, devices=[_gpu_device(memory="141Gi")]), + ) + ], + all_replicas=[], + want=[], + ), + Case( + name="a gang too big for its only matching pool is rejected", + # Both members match only big (>= 141Gi); small (40Gi) matches + # neither. big has one free node but the gang needs two. The + # engine doesn't fit any single pool, so it's rejected; with big + # the only cluster the replica goes unplaced. + deployment=_deployment( + engines=[ + _gang( + [_request(cel_exprs=[_MEM_141])], + [_request(cel_exprs=[_MEM_141])], + ) + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool("small", nodes=8, devices=[_gpu_device(memory="40Gi")]), + _pool("big", nodes=1, devices=[_gpu_device(memory="141Gi")]), + ], + ) + ], + all_replicas=[], + want=[], + ), + Case( + name="a gang too big for one cluster's pool lands whole on another", + # Same gang. cluster-a's matching pool has only one free node + # (too few for the two-member gang), so the scheduler rejects + # cluster-a and places the whole gang on cluster-b, whose pool + # has room for both members. + deployment=_deployment( + engines=[ + _gang( + [_request(cel_exprs=[_MEM_141])], + [_request(cel_exprs=[_MEM_141])], + ) + ] + ), + clusters=[ + _cluster( + "cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + pools=[ + _pool("small", nodes=8, devices=[_gpu_device(memory="40Gi")]), + _pool("big", nodes=1, devices=[_gpu_device(memory="141Gi")]), + ], + ), + _cluster( + "cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + pools=[_pool("big", nodes=2, devices=[_gpu_device(memory="141Gi")])], + ), + ], + all_replicas=[], + want=[ + scheduling.Candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + engines=[ + scheduling.EnginePlacement( + name=_ENGINE, + members=[ + scheduling.MemberPlacement(role="Leader", pool="big", device_requests=[_resolved()]), + scheduling.MemberPlacement(role="Worker", pool="big", device_requests=[_resolved()]), ], ) ], - all_replicas=[], - want=[], - ), - Case( - name="a gang too big for one cluster's pool lands whole on another", - # Same gang. cluster-a's matching pool has only one free node - # (too few for the two-member gang), so the scheduler rejects - # cluster-a and places the whole gang on cluster-b, whose pool - # has room for both members. - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_141])], - ) - ] - ), - clusters=[ - _cluster( - "cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pools=[ - _pool("small", nodes=8, devices=[_gpu_device(memory="40Gi")]), - _pool("big", nodes=1, devices=[_gpu_device(memory="141Gi")]), + ) + ], + ), + Case( + name="a member claimable elsewhere is not stranded on a synthetic match", + # On pool-a the leader's request matches only a Synthetic + # device (nothing to claim) while the worker claims, so the + # whole engine *could* land there - but pool-b satisfies the + # leader claimably. The engine must go to pool-b; placing on + # pool-a would run the leader without the GPU it asked for. + deployment=_deployment( + engines=[ + _gang( + [_request(cel_exprs=[_MEM_200])], + [_request(cel_exprs=[_MEM_141])], + ) + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[ + _pool( + "a", + devices=[ + _gpu_device(memory="141Gi"), + _gpu_device(name="syn", claim="Synthetic", memory="200Gi"), ], ), - _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("big", nodes=2, devices=[_gpu_device(memory="141Gi")])], - ), - ], - all_replicas=[], - want=[ - scheduling.Candidate( - name="cluster-b", - index=0, - gateway_hostname="cluster-b.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement( - role="Leader", pool="big", device_requests=[_resolved()] - ), - scheduling.MemberPlacement( - role="Worker", pool="big", device_requests=[_resolved()] - ), - ], - ) - ], - ) + _pool("b", devices=[_gpu_device(memory="200Gi")]), ], - ), - Case( - name="a member claimable elsewhere is not stranded on a synthetic match", - # On pool-a the leader's request matches only a Synthetic - # device (nothing to claim) while the worker claims, so the - # whole engine *could* land there - but pool-b satisfies the - # leader claimably. The engine must go to pool-b; placing on - # pool-a would run the leader without the GPU it asked for. - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_200])], - [_request(cel_exprs=[_MEM_141])], - ) - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[ - _pool( - "a", - devices=[ - _gpu_device(memory="141Gi"), - _gpu_device(name="syn", claim="Synthetic", memory="200Gi"), - ], + ) + ], + all_replicas=[], + want=[ + scheduling.Candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + engines=[ + scheduling.EnginePlacement( + name=_ENGINE, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="b", + device_requests=[_resolved(cel_exprs=[_MEM_200])], + ), + scheduling.MemberPlacement( + role="Worker", + pool="b", + device_requests=[_resolved(cel_exprs=[_MEM_141])], ), - _pool("b", devices=[_gpu_device(memory="200Gi")]), - ], - ) - ], - all_replicas=[], - want=[ - scheduling.Candidate( - name="cluster-a", - index=0, - gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement( - role="Leader", - pool="b", - device_requests=[_resolved(cel_exprs=[_MEM_200])], - ), - scheduling.MemberPlacement( - role="Worker", - pool="b", - device_requests=[_resolved(cel_exprs=[_MEM_141])], - ), - ], - ) ], ) ], - ), - Case( - name="a member synthetic-only everywhere places claimless with its gang", - # The leader's request matches only the pool's synthetic NIC on - # every pool - deliberate (a selector that pins without - # claiming). It places claimless alongside the claiming worker. - deployment=_deployment( - engines=[ - _gang( - [_request(name="nic", cel_exprs=[_IB])], - [_request(cel_exprs=[_MEM_141])], - ) - ] - ), - clusters=[ - _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device()])], - ) - ], - all_replicas=[], - want=[ - scheduling.Candidate( - name="cluster-a", - index=0, - gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), - scheduling.MemberPlacement( - role="Worker", pool="frontier", device_requests=[_resolved()] - ), - ], - ) + ) + ], + ), + Case( + name="a member synthetic-only everywhere places claimless with its gang", + # The leader's request matches only the pool's synthetic NIC on + # every pool - deliberate (a selector that pins without + # claiming). It places claimless alongside the claiming worker. + deployment=_deployment( + engines=[ + _gang( + [_request(name="nic", cel_exprs=[_IB])], + [_request(cel_exprs=[_MEM_141])], + ) + ] + ), + clusters=[ + _cluster( + "cluster-a", + pools=[_pool("frontier", devices=[_gpu_device(), _nic_device()])], + ) + ], + all_replicas=[], + want=[ + scheduling.Candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + engines=[ + scheduling.EnginePlacement( + name=_ENGINE, + members=[ + scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), + scheduling.MemberPlacement(role="Worker", pool="frontier", device_requests=[_resolved()]), ], ) ], - ), - Case( - name="a member that matches nowhere fails the whole replica", - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_200])], - ) - ] - ), - clusters=[_cluster("cluster-a", pools=[_pool("default", devices=[_gpu_device(memory="141Gi")])])], - all_replicas=[], - want=[], - ), - Case( - name="a claimless leader rides along on its gang's pool at zero cost", - # The leader carries no nodeSelector: it claims nothing, follows - # the worker's pool, and costs no nodes - the 1-node pool fits - # the whole gang because only the worker occupies a node. - deployment=_deployment(engines=[_gang(None, [_request(cel_exprs=[_MEM_141])])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], - all_replicas=[], - want=[ - scheduling.Candidate( - name="cluster-a", - index=0, - gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), - scheduling.MemberPlacement( - role="Worker", pool="frontier", device_requests=[_resolved()] - ), - ], - ) + ) + ], + ), + Case( + name="a member that matches nowhere fails the whole replica", + deployment=_deployment( + engines=[ + _gang( + [_request(cel_exprs=[_MEM_141])], + [_request(cel_exprs=[_MEM_200])], + ) + ] + ), + clusters=[_cluster("cluster-a", pools=[_pool("default", devices=[_gpu_device(memory="141Gi")])])], + all_replicas=[], + want=[], + ), + Case( + name="a claimless leader rides along on its gang's pool at zero cost", + # The leader carries no nodeSelector: it claims nothing, follows + # the worker's pool, and costs no nodes - the 1-node pool fits + # the whole gang because only the worker occupies a node. + deployment=_deployment(engines=[_gang(None, [_request(cel_exprs=[_MEM_141])])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], + all_replicas=[], + want=[ + scheduling.Candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + engines=[ + scheduling.EnginePlacement( + name=_ENGINE, + members=[ + scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), + scheduling.MemberPlacement(role="Worker", pool="frontier", device_requests=[_resolved()]), ], ) ], - ), - Case( - name="a retained replica's claimless member keeps its pin", - deployment=_deployment(engines=[_gang(None, [_request(cel_exprs=[_MEM_141])])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], - all_replicas=[ - _replica( - "my-model", - "cluster-a", - engines=[ - mrv1alpha1.Engine( - name=_ENGINE, - members=[ - mrv1alpha1.Member( - role="Leader", - nodePoolName="frontier", - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - mrv1alpha1.Member( - role="Worker", - worker=mrv1alpha1.Worker(nodes=1), - nodePoolName="frontier", - deviceRequests=_replica_device_requests(), - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - ], - ) + ) + ], + ), + Case( + name="a retained replica's claimless member keeps its pin", + deployment=_deployment(engines=[_gang(None, [_request(cel_exprs=[_MEM_141])])]), + clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], + all_replicas=[ + _replica( + "my-model", + "cluster-a", + engines=[ + mrv1alpha1.Engine( + name=_ENGINE, + members=[ + mrv1alpha1.Member( + role="Leader", + nodePoolName="frontier", + template=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") + ] + ) + ), + ), + mrv1alpha1.Member( + role="Worker", + worker=mrv1alpha1.Worker(nodes=1), + nodePoolName="frontier", + deviceRequests=_replica_device_requests(), + template=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") + ] + ) + ), + ), ], ) ], - want=[ - scheduling.Candidate( - name="cluster-a", - index=0, - gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), - scheduling.MemberPlacement( - role="Worker", pool="frontier", device_requests=[_resolved()] - ), - ], - ) + ) + ], + want=[ + scheduling.Candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + engines=[ + scheduling.EnginePlacement( + name=_ENGINE, + members=[ + scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), + scheduling.MemberPlacement(role="Worker", pool="frontier", device_requests=[_resolved()]), ], ) ], - ), - Case( - name="another deployment's claimless member consumes no capacity", - # other-model's gang occupies only its worker's node: its - # claimless leader shares that node. The 2-node pool has 1 node - # free, so our 1-node deployment fits. Charging the claimless - # leader a node would wrongly report insufficient capacity. - deployment=_deployment(), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], - all_replicas=[ - _replica( - "other-model", - "cluster-a", - engines=[ - mrv1alpha1.Engine( - name="main", - members=[ - mrv1alpha1.Member( - role="Leader", - nodePoolName="default", - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - mrv1alpha1.Member( - role="Worker", - worker=mrv1alpha1.Worker(nodes=1), - nodePoolName="default", - deviceRequests=_replica_device_requests(), - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - ], - ) + ) + ], + ), + Case( + name="another deployment's claimless member consumes no capacity", + # other-model's gang occupies only its worker's node: its + # claimless leader shares that node. The 2-node pool has 1 node + # free, so our 1-node deployment fits. Charging the claimless + # leader a node would wrongly report insufficient capacity. + deployment=_deployment(), + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], + all_replicas=[ + _replica( + "other-model", + "cluster-a", + engines=[ + mrv1alpha1.Engine( + name="main", + members=[ + mrv1alpha1.Member( + role="Leader", + nodePoolName="default", + template=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") + ] + ) + ), + ), + mrv1alpha1.Member( + role="Worker", + worker=mrv1alpha1.Worker(nodes=1), + nodePoolName="default", + deviceRequests=_replica_device_requests(), + template=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") + ] + ) + ), + ), ], ) ], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="a member shape change re-places the replica", - # The deployment grew a Worker (Standalone -> Leader+Worker). - # The observed single-member replica no longer lines up, so it - # is re-placed with the new shape. - deployment=_deployment(engines=[_gang([_request()], [_request()])]), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], - all_replicas=[_replica("my-model", "cluster-a")], - want=[ - scheduling.Candidate( - name="cluster-a", - index=0, - gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement( - role="Leader", pool="default", device_requests=[_resolved()] - ), - scheduling.MemberPlacement( - role="Worker", pool="default", device_requests=[_resolved()] - ), - ], - ) + ) + ], + want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], + ), + Case( + name="a member shape change re-places the replica", + # The deployment grew a Worker (Standalone -> Leader+Worker). + # The observed single-member replica no longer lines up, so it + # is re-placed with the new shape. + deployment=_deployment(engines=[_gang([_request()], [_request()])]), + clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], + all_replicas=[_replica("my-model", "cluster-a")], + want=[ + scheduling.Candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + engines=[ + scheduling.EnginePlacement( + name=_ENGINE, + members=[ + scheduling.MemberPlacement(role="Leader", pool="default", device_requests=[_resolved()]), + scheduling.MemberPlacement(role="Worker", pool="default", device_requests=[_resolved()]), ], ) ], - ), - ] + ) + ], + ), +] - for case in cases: - with self.subTest(case.name): - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) - self.assertEqual(case.want, got, f"{case.name}: -want, +got") +@pytest.mark.parametrize("case", MEMBERS_CASES, ids=lambda case: case.name) +def test_members(case: Case) -> None: + """The scheduler places every member of an engine on one pool that fits them all.""" + got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) + assert got == case.want -class TestScheduleTaints(unittest.TestCase): - """Taints on InferenceClusters gate placement; a matching toleration on the - ModelDeployment overrides them. NoSchedule keeps new replicas off a cluster - but leaves existing ones; NoExecute additionally drains the existing ones, - which fill reschedules onto a tolerated cluster.""" - _MAINT = icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule") - _DECOMM = icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute") +# Taints on InferenceClusters gate placement; a matching toleration on the +# ModelDeployment overrides them. NoSchedule keeps new replicas off a cluster but +# leaves existing ones; NoExecute additionally drains the existing ones, which +# fill reschedules onto a tolerated cluster. +_MAINT = icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule") +_DECOMM = icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute") - def _names(self, got: list[scheduling.Candidate]) -> list[tuple[str, int]]: - return [(c.name, c.index) for c in got] - def test_noschedule_keeps_new_replicas_off(self) -> None: - clusters = [ - _cluster("cluster-a", taints=[self._MAINT]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1), clusters, []) - self.assertEqual(self._names(got), [("cluster-b", 0)]) - - def test_noschedule_leaves_existing_replica_in_place(self) -> None: - existing = _replica("my-model", "cluster-a") - got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a", taints=[self._MAINT])], [existing]) - self.assertEqual(self._names(got), [("cluster-a", 0)]) - - def test_toleration_allows_placement_on_tainted(self) -> None: - tol = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists") - got = scheduling.schedule( - _deployment(replicas=1, tolerations=[tol]), [_cluster("cluster-a", taints=[self._MAINT])], [] - ) - self.assertEqual(self._names(got), [("cluster-a", 0)]) +def _names(got: list[scheduling.Candidate]) -> list[tuple[str, int]]: + return [(c.name, c.index) for c in got] - def test_noexecute_drains_and_reschedules(self) -> None: - existing = _replica("my-model", "cluster-a") - clusters = [ - _cluster("cluster-a", taints=[self._DECOMM]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1), clusters, [existing]) - self.assertEqual(self._names(got), [("cluster-b", 0)]) - - def test_noexecute_toleration_retains_in_place(self) -> None: - tol = mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists") - existing = _replica("my-model", "cluster-a") - got = scheduling.schedule( - _deployment(replicas=1, tolerations=[tol]), [_cluster("cluster-a", taints=[self._DECOMM])], [existing] - ) - self.assertEqual(self._names(got), [("cluster-a", 0)]) - - def test_noexecute_drain_leaves_count_unmet_when_nowhere_to_go(self) -> None: - """Draining with no tolerated cluster to reschedule onto yields fewer - than spec.replicas; the deploy function surfaces the shortfall.""" - existing = _replica("my-model", "cluster-a") - got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a", taints=[self._DECOMM])], [existing]) - self.assertEqual(got, []) - - def test_noschedule_toleration_does_not_cover_a_noexecute_taint(self) -> None: - """Matching the key but not the effect doesn't tolerate: an operator who - tolerates only NoSchedule is still drained by a NoExecute taint.""" - tol = mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists", effect="NoSchedule") - existing = _replica("my-model", "cluster-a") - clusters = [ - _cluster("cluster-a", taints=[self._DECOMM]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, [existing]) - self.assertEqual(self._names(got), [("cluster-b", 0)]) - - def test_untolerated_second_taint_still_repels(self) -> None: - """Tolerating one of a cluster's taints isn't enough; any untolerated - taint keeps new replicas off.""" - other = icv1alpha1.Taint(key="modelplane.ai/reserved", effect="NoSchedule") - tol = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists") - clusters = [ - _cluster("cluster-a", taints=[self._MAINT, other]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, []) - self.assertEqual(self._names(got), [("cluster-b", 0)]) - - def test_equal_toleration_matches_on_value(self) -> None: - """Equal tolerates only when key and value both match.""" - match = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="on") - placed = scheduling.schedule( - _deployment(replicas=1, tolerations=[match]), [_cluster("cluster-a", taints=[self._MAINT])], [] - ) - self.assertEqual(self._names(placed), [("cluster-a", 0)]) - mismatch = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="off") - repelled = scheduling.schedule( - _deployment(replicas=1, tolerations=[mismatch]), [_cluster("cluster-a", taints=[self._MAINT])], [] - ) - self.assertEqual(repelled, []) +def test_taints_noschedule_keeps_new_replicas_off() -> None: + """A NoSchedule taint keeps a new replica off the cluster.""" + clusters = [ + _cluster("cluster-a", taints=[_MAINT]), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ] + got = scheduling.schedule(_deployment(replicas=1), clusters, []) + assert _names(got) == [("cluster-b", 0)] - def test_keyless_exists_tolerates_every_taint(self) -> None: - """An Exists toleration with no key tolerates any taint on the cluster.""" - tol = mdv1alpha1.Toleration(operator="Exists") - clusters = [_cluster("cluster-a", taints=[self._MAINT, self._DECOMM])] - got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, []) - self.assertEqual(self._names(got), [("cluster-a", 0)]) +def test_taints_noschedule_leaves_existing_replica_in_place() -> None: + """A NoSchedule taint leaves an existing replica where it is.""" + existing = _replica("my-model", "cluster-a") + got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a", taints=[_MAINT])], [existing]) + assert _names(got) == [("cluster-a", 0)] -class TestPlacementLabels(unittest.TestCase): - """A cluster's placement labels reach the Candidate, and so the ModelReplica - and ModelEndpoint composed from it. - This is how a self-hosted endpoint gets its region: a ModelService selects - endpoints by label, so without it a region-scoped service can't select its - own replicas, and it can't label them by hand because Modelplane owns them. - """ +def test_taints_toleration_allows_placement_on_tainted() -> None: + """A matching toleration lets a new replica onto a tainted cluster.""" + tol = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists") + got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), [_cluster("cluster-a", taints=[_MAINT])], []) + assert _names(got) == [("cluster-a", 0)] + + +def test_taints_noexecute_drains_and_reschedules() -> None: + """A NoExecute taint drains an existing replica onto another cluster.""" + existing = _replica("my-model", "cluster-a") + clusters = [ + _cluster("cluster-a", taints=[_DECOMM]), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ] + got = scheduling.schedule(_deployment(replicas=1), clusters, [existing]) + assert _names(got) == [("cluster-b", 0)] + + +def test_taints_noexecute_toleration_retains_in_place() -> None: + """A NoExecute toleration keeps an existing replica on a draining cluster.""" + tol = mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists") + existing = _replica("my-model", "cluster-a") + got = scheduling.schedule( + _deployment(replicas=1, tolerations=[tol]), [_cluster("cluster-a", taints=[_DECOMM])], [existing] + ) + assert _names(got) == [("cluster-a", 0)] + + +def test_taints_noexecute_drain_leaves_count_unmet_when_nowhere_to_go() -> None: + """Draining with no tolerated cluster to go to yields fewer than spec.replicas.""" + # The deploy function surfaces the shortfall. + existing = _replica("my-model", "cluster-a") + got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a", taints=[_DECOMM])], [existing]) + assert got == [] + + +def test_taints_noschedule_toleration_does_not_cover_a_noexecute_taint() -> None: + """A toleration that matches the key but not the effect doesn't tolerate.""" + # An operator who tolerates only NoSchedule is still drained by a NoExecute + # taint. + tol = mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists", effect="NoSchedule") + existing = _replica("my-model", "cluster-a") + clusters = [ + _cluster("cluster-a", taints=[_DECOMM]), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ] + got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, [existing]) + assert _names(got) == [("cluster-b", 0)] + + +def test_taints_untolerated_second_taint_still_repels() -> None: + """Any untolerated taint keeps new replicas off, even if another is tolerated.""" + other = icv1alpha1.Taint(key="modelplane.ai/reserved", effect="NoSchedule") + tol = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists") + clusters = [ + _cluster("cluster-a", taints=[_MAINT, other]), + _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + ] + got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, []) + assert _names(got) == [("cluster-b", 0)] + + +def test_taints_equal_toleration_matches_on_value() -> None: + """An Equal toleration tolerates only when key and value both match.""" + match = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="on") + placed = scheduling.schedule( + _deployment(replicas=1, tolerations=[match]), [_cluster("cluster-a", taints=[_MAINT])], [] + ) + assert _names(placed) == [("cluster-a", 0)] + + mismatch = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="off") + repelled = scheduling.schedule( + _deployment(replicas=1, tolerations=[mismatch]), [_cluster("cluster-a", taints=[_MAINT])], [] + ) + assert repelled == [] + + +def test_taints_keyless_exists_tolerates_every_taint() -> None: + """An Exists toleration with no key tolerates any taint on the cluster.""" + tol = mdv1alpha1.Toleration(operator="Exists") + clusters = [_cluster("cluster-a", taints=[_MAINT, _DECOMM])] + got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, []) + assert _names(got) == [("cluster-a", 0)] + + +# A cluster's placement labels reach the Candidate, and so the ModelReplica and +# ModelEndpoint composed from it. +# +# This is how a self-hosted endpoint gets its region: a ModelService selects +# endpoints by label, so without it a region-scoped service can't select its own +# replicas, and it can't label them by hand because Modelplane owns them. + + +def test_placement_labels_reach_the_candidate() -> None: + """A cluster's placement labels reach the Candidate.""" + got = scheduling.schedule( + _deployment(replicas=1), + [_cluster("cluster-a", placement_labels={"example.org/region": "eu"})], + [], + ) + assert [c.placement_labels for c in got] == [{"example.org/region": "eu"}] - def test_labels_reach_the_candidate(self) -> None: - got = scheduling.schedule( - _deployment(replicas=1), - [_cluster("cluster-a", placement_labels={"example.org/region": "eu"})], - [], - ) - self.assertEqual([c.placement_labels for c in got], [{"example.org/region": "eu"}]) - def test_a_cluster_declaring_none_yields_none(self) -> None: - got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a")], []) - self.assertEqual([c.placement_labels for c in got], [{}]) +def test_placement_labels_a_cluster_declaring_none_yields_none() -> None: + """A cluster that declares no placement labels yields a Candidate with none.""" + got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a")], []) + assert [c.placement_labels for c in got] == [{}] diff --git a/functions/compose-model-deployment/tests/test_semver.py b/functions/compose-model-deployment/tests/test_semver.py index e7b58ae85..433a3f058 100644 --- a/functions/compose-model-deployment/tests/test_semver.py +++ b/functions/compose-model-deployment/tests/test_semver.py @@ -25,12 +25,12 @@ Upstream cases that don't apply: the compile-time overload error (isSemver([1,2,3])) - celpy doesn't type-check overloads; and the runtime parse error for semver("v1.0") - upstream raises, we treat a bad version as a -non-match (driven through the parse layer in TestParseRejects). +non-match (driven through the parse layer in test_parse_rejects). """ import dataclasses -import unittest +import pytest from function import cel, semver @@ -52,79 +52,80 @@ class ParseErrCase: input: str -class TestSemverCEL(unittest.TestCase): - """Mirrors semver_test.go TestSemver (and the doc-comment examples).""" - - def test_semver(self) -> None: - cases = [ - # parse + doc-comment examples. - Case(name="parse", expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', want=True), - Case(name="parse with prerelease", expr='semver("0.1.0-alpha.1").major() == 0', want=True), - # isSemver strict. - Case(name="isSemver full", expr='isSemver("1.2.3-beta.1+build.1")', want=True), - Case(name="isSemver simple", expr='isSemver("1.0.0")', want=True), - Case(name="isSemver hello", expr='isSemver("hello")', want=False), - Case(name="isSemver empty false", expr='isSemver("")', want=False), - Case(name="isSemver v prefix false", expr='isSemver("v1.0.0")', want=False), - Case(name="isSemver v1.0 false", expr='isSemver("v1.0")', want=False), - Case(name="isSemver leading whitespace false", expr='isSemver(" 1.0.0")', want=False), - Case(name="isSemver inner whitespace false", expr='isSemver("1. 0.0")', want=False), - Case(name="isSemver trailing whitespace false", expr='isSemver("1.0.0 ")', want=False), - Case(name="isSemver leading zeros false", expr='isSemver("01.01.01")', want=False), - Case(name="isSemver major only false", expr='isSemver("1")', want=False), - Case(name="isSemver major minor only false", expr='isSemver("1.1")', want=False), - Case(name="isSemver 200K", expr='isSemver("200K")', want=False), - Case(name="isSemver Mi", expr='isSemver("Mi")', want=False), - # isSemver normalize overload. Normalization does NOT trim whitespace. - Case(name="isSemver empty normalize false", expr='isSemver("", true)', want=False), - Case(name="isSemver leading whitespace normalize false", expr='isSemver(" 1.0.0", true)', want=False), - Case(name="isSemver inner whitespace normalize false", expr='isSemver("1. 0.0", true)', want=False), - Case(name="isSemver trailing whitespace normalize false", expr='isSemver("1.0.0 ", true)', want=False), - Case(name="isSemver v prefix normalize true", expr='isSemver("v1.0.0", true)', want=True), - Case(name="isSemver leading zeros normalize true", expr='isSemver("01.01.01", true)', want=True), - Case(name="isSemver major only normalize true", expr='isSemver("1", true)', want=True), - Case(name="isSemver major minor only normalize true", expr='isSemver("1.1", true)', want=True), - # normalize equality and semver(...) examples. - Case(name="equality normalize", expr='semver("v01.01", true) == semver("1.1.0")', want=True), - Case(name="semver v prefix normalize major", expr='semver("v1.0.0", true).major() == 1', want=True), - Case(name="semver short normalize patch", expr='semver("1.0", true).patch() == 0', want=True), - Case(name="semver leading zeros normalize", expr='semver("01.01.01", true).minor() == 1', want=True), - # equality / comparison. - Case(name="equality reflexivity", expr='semver("1.2.3") == semver("1.2.3")', want=True), - Case(name="inequality", expr='semver("1.2.3") == semver("1.0.0")', want=False), - Case(name="less", expr='semver("1.0.0").isLessThan(semver("1.2.3"))', want=True), - Case(name="less false", expr='semver("1.0.0").isLessThan(semver("1.0.0"))', want=False), - Case(name="greater", expr='semver("1.2.3").isGreaterThan(semver("1.0.0"))', want=True), - Case(name="greater false", expr='semver("1.0.0").isGreaterThan(semver("1.0.0"))', want=False), - Case(name="compare equal", expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', want=True), - Case(name="compare less", expr='semver("1.2.3").compareTo(semver("2.0.0")) == -1', want=True), - Case(name="compare greater", expr='semver("1.2.3").compareTo(semver("0.1.2")) == 1', want=True), - # major / minor / patch. - Case(name="major", expr='semver("1.2.3").major() == 1', want=True), - Case(name="minor", expr='semver("1.2.3").minor() == 2', want=True), - Case(name="patch", expr='semver("1.2.3").patch() == 3', want=True), - # A bad version is a runtime error upstream -> non-match here. - Case(name="bad version is non-match", expr='semver("v1.0").major() == 1', want=False), - ] - for case in cases: - with self.subTest(case.name): - self.assertEqual(case.want, _eval(case.expr), f"{case.name}: -want, +got") - - -class TestParseRejects(unittest.TestCase): +SEMVER_CASES = [ + # parse + doc-comment examples. + Case(name="parse", expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', want=True), + Case(name="parse with prerelease", expr='semver("0.1.0-alpha.1").major() == 0', want=True), + # isSemver strict. + Case(name="isSemver full", expr='isSemver("1.2.3-beta.1+build.1")', want=True), + Case(name="isSemver simple", expr='isSemver("1.0.0")', want=True), + Case(name="isSemver hello", expr='isSemver("hello")', want=False), + Case(name="isSemver empty false", expr='isSemver("")', want=False), + Case(name="isSemver v prefix false", expr='isSemver("v1.0.0")', want=False), + Case(name="isSemver v1.0 false", expr='isSemver("v1.0")', want=False), + Case(name="isSemver leading whitespace false", expr='isSemver(" 1.0.0")', want=False), + Case(name="isSemver inner whitespace false", expr='isSemver("1. 0.0")', want=False), + Case(name="isSemver trailing whitespace false", expr='isSemver("1.0.0 ")', want=False), + Case(name="isSemver leading zeros false", expr='isSemver("01.01.01")', want=False), + Case(name="isSemver major only false", expr='isSemver("1")', want=False), + Case(name="isSemver major minor only false", expr='isSemver("1.1")', want=False), + Case(name="isSemver 200K", expr='isSemver("200K")', want=False), + Case(name="isSemver Mi", expr='isSemver("Mi")', want=False), + # isSemver normalize overload. Normalization does NOT trim whitespace. + Case(name="isSemver empty normalize false", expr='isSemver("", true)', want=False), + Case(name="isSemver leading whitespace normalize false", expr='isSemver(" 1.0.0", true)', want=False), + Case(name="isSemver inner whitespace normalize false", expr='isSemver("1. 0.0", true)', want=False), + Case(name="isSemver trailing whitespace normalize false", expr='isSemver("1.0.0 ", true)', want=False), + Case(name="isSemver v prefix normalize true", expr='isSemver("v1.0.0", true)', want=True), + Case(name="isSemver leading zeros normalize true", expr='isSemver("01.01.01", true)', want=True), + Case(name="isSemver major only normalize true", expr='isSemver("1", true)', want=True), + Case(name="isSemver major minor only normalize true", expr='isSemver("1.1", true)', want=True), + # normalize equality and semver(...) examples. + Case(name="equality normalize", expr='semver("v01.01", true) == semver("1.1.0")', want=True), + Case(name="semver v prefix normalize major", expr='semver("v1.0.0", true).major() == 1', want=True), + Case(name="semver short normalize patch", expr='semver("1.0", true).patch() == 0', want=True), + Case(name="semver leading zeros normalize", expr='semver("01.01.01", true).minor() == 1', want=True), + # equality / comparison. + Case(name="equality reflexivity", expr='semver("1.2.3") == semver("1.2.3")', want=True), + Case(name="inequality", expr='semver("1.2.3") == semver("1.0.0")', want=False), + Case(name="less", expr='semver("1.0.0").isLessThan(semver("1.2.3"))', want=True), + Case(name="less false", expr='semver("1.0.0").isLessThan(semver("1.0.0"))', want=False), + Case(name="greater", expr='semver("1.2.3").isGreaterThan(semver("1.0.0"))', want=True), + Case(name="greater false", expr='semver("1.0.0").isGreaterThan(semver("1.0.0"))', want=False), + Case(name="compare equal", expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', want=True), + Case(name="compare less", expr='semver("1.2.3").compareTo(semver("2.0.0")) == -1', want=True), + Case(name="compare greater", expr='semver("1.2.3").compareTo(semver("0.1.2")) == 1', want=True), + # major / minor / patch. + Case(name="major", expr='semver("1.2.3").major() == 1', want=True), + Case(name="minor", expr='semver("1.2.3").minor() == 2', want=True), + Case(name="patch", expr='semver("1.2.3").patch() == 3', want=True), + # A bad version is a runtime error upstream -> non-match here. + Case(name="bad version is non-match", expr='semver("v1.0").major() == 1', want=False), +] + + +@pytest.mark.parametrize("case", SEMVER_CASES, ids=lambda case: case.name) +def test_semver(case: Case) -> None: + """A semver CEL expression evaluates as it does upstream.""" + assert _eval(case.expr) == case.want + + +PARSE_REJECTS_CASES = [ + ParseErrCase(name="v prefix", input="v1.0"), + ParseErrCase(name="major only", input="1"), + ParseErrCase(name="major minor only", input="1.1"), + ParseErrCase(name="leading zeros", input="01.01.01"), + ParseErrCase(name="leading whitespace", input=" 1.0.0"), + ParseErrCase(name="trailing whitespace", input="1.0.0 "), + ParseErrCase(name="empty", input=""), + ParseErrCase(name="word", input="hello"), +] + + +@pytest.mark.parametrize("case", PARSE_REJECTS_CASES, ids=lambda case: case.name) +def test_parse_rejects(case: ParseErrCase) -> None: """parse() (strict) rejects what blang/semver Parse rejects.""" - - def test_parse_rejects(self) -> None: - cases = [ - ParseErrCase(name="v prefix", input="v1.0"), - ParseErrCase(name="major only", input="1"), - ParseErrCase(name="major minor only", input="1.1"), - ParseErrCase(name="leading zeros", input="01.01.01"), - ParseErrCase(name="leading whitespace", input=" 1.0.0"), - ParseErrCase(name="trailing whitespace", input="1.0.0 "), - ParseErrCase(name="empty", input=""), - ParseErrCase(name="word", input="hello"), - ] - for case in cases: - with self.subTest(case.name), self.assertRaises(ValueError): - semver.parse(case.input) + with pytest.raises( + ValueError, match=r"version string empty|no Major\.Minor\.Patch|invalid character|leading zeroes" + ): + semver.parse(case.input) diff --git a/functions/compose-model-endpoint/tests/__init__.py b/functions/compose-model-endpoint/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-model-endpoint/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-model-endpoint/tests/test_fn.py b/functions/compose-model-endpoint/tests/test_fn.py index a52a88996..a149608a9 100644 --- a/functions/compose-model-endpoint/tests/test_fn.py +++ b/functions/compose-model-endpoint/tests/test_fn.py @@ -14,15 +14,17 @@ """Tests for the compose-model-endpoint function.""" +import asyncio import base64 import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.modelendpoint import v1alpha1 @@ -100,147 +102,131 @@ def _response( return rsp -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - cases = [ - Case( - name="no credential: usable as soon as it exists", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - ), - want=_response( - reason=fn.CONDITION_REASON_ENDPOINT_USABLE, - status=fnv1.STATUS_CONDITION_TRUE, - ), +COMPOSE_CASES = [ + Case( + name="no credential: usable as soon as it exists", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), + ), + want=_response( + reason=fn.CONDITION_REASON_ENDPOINT_USABLE, + status=fnv1.STATUS_CONDITION_TRUE, + ), + ), + Case( + name="a credential that resolves", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) ), - Case( - name="a credential that resolves", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key"))) + required_resources={ + "credential": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct(_secret("together-api-key", {"apiKey": "sk-abc"})) ) - ), - required_resources={ - "credential": fnv1.Resources( - items=[ - fnv1.Resource( - resource=resource.dict_to_struct(_secret("together-api-key", {"apiKey": "sk-abc"})) - ) - ] - ) - }, - ), - want=_response( - reason=fn.CONDITION_REASON_ENDPOINT_USABLE, - status=fnv1.STATUS_CONDITION_TRUE, - requirements=_credential_requirement("together-api-key"), - ), + ] + ) + }, + ), + want=_response( + reason=fn.CONDITION_REASON_ENDPOINT_USABLE, + status=fnv1.STATUS_CONDITION_TRUE, + requirements=_credential_requirement("together-api-key"), + ), + ), + Case( + name="a credential Secret that does not exist", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) ), - Case( - name="a credential Secret that does not exist", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key"))) - ) - ), - required_resources={"credential": fnv1.Resources(items=[])}, - ), - want=_response( - reason=fn.CONDITION_REASON_CREDENTIAL_MISSING, - status=fnv1.STATUS_CONDITION_FALSE, - message="Secret together-api-key does not exist", - requirements=_credential_requirement("together-api-key"), - ), + required_resources={"credential": fnv1.Resources(items=[])}, + ), + want=_response( + reason=fn.CONDITION_REASON_CREDENTIAL_MISSING, + status=fnv1.STATUS_CONDITION_FALSE, + message="Secret together-api-key does not exist", + requirements=_credential_requirement("together-api-key"), + ), + ), + Case( + # A Secret that exists but lacks the key is the likelier mistake, + # and would otherwise surface as a 401 from the provider. + name="a credential Secret missing the key", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) ), - Case( - # A Secret that exists but lacks the key is the likelier mistake, - # and would otherwise surface as a 401 from the provider. - name="a credential Secret missing the key", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key"))) - ) - ), - required_resources={ - "credential": fnv1.Resources( - items=[ - fnv1.Resource( - resource=resource.dict_to_struct(_secret("together-api-key", {"token": "sk-abc"})) - ) - ] + required_resources={ + "credential": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct(_secret("together-api-key", {"token": "sk-abc"})) ) - }, - ), - want=_response( - reason=fn.CONDITION_REASON_CREDENTIAL_MISSING, - status=fnv1.STATUS_CONDITION_FALSE, - message="Secret together-api-key has no key apiKey", - requirements=_credential_requirement("together-api-key"), - ), + ] + ) + }, + ), + want=_response( + reason=fn.CONDITION_REASON_CREDENTIAL_MISSING, + status=fnv1.STATUS_CONDITION_FALSE, + message="Secret together-api-key has no key apiKey", + requirements=_credential_requirement("together-api-key"), + ), + ), + Case( + name="a credential under a non-default key", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + _xr(credential=_api_key("together-api-key", key="TOGETHER_API_KEY")) + ) + ) ), - Case( - name="a credential under a non-default key", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( + required_resources={ + "credential": fnv1.Resources( + items=[ + fnv1.Resource( resource=resource.dict_to_struct( - _xr(credential=_api_key("together-api-key", key="TOGETHER_API_KEY")) + _secret("together-api-key", {"TOGETHER_API_KEY": "sk-abc"}) ) ) - ), - required_resources={ - "credential": fnv1.Resources( - items=[ - fnv1.Resource( - resource=resource.dict_to_struct( - _secret("together-api-key", {"TOGETHER_API_KEY": "sk-abc"}) - ) - ) - ] - ) - }, - ), - want=_response( - reason=fn.CONDITION_REASON_ENDPOINT_USABLE, - status=fnv1.STATUS_CONDITION_TRUE, - requirements=_credential_requirement("together-api-key"), - ), - ), - Case( - name="an unresolved credential requirement", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key"))) - ) - ), - ), - want=_response( - reason=fn.CONDITION_REASON_WAITING_FOR_CREDENTIAL, - status=fnv1.STATUS_CONDITION_FALSE, - message="Waiting for Secret together-api-key to resolve", - requirements=_credential_requirement("together-api-key"), - ), - ), - ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", + ] ) + }, + ), + want=_response( + reason=fn.CONDITION_REASON_ENDPOINT_USABLE, + status=fnv1.STATUS_CONDITION_TRUE, + requirements=_credential_requirement("together-api-key"), + ), + ), + Case( + name="an unresolved credential requirement", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) + ), + ), + want=_response( + reason=fn.CONDITION_REASON_WAITING_FOR_CREDENTIAL, + status=fnv1.STATUS_CONDITION_FALSE, + message="Waiting for Secret together-api-key to resolve", + requirements=_credential_requirement("together-api-key"), + ), + ), +] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction reports whether the endpoint's credential makes it usable.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-model-replica/tests/__init__.py b/functions/compose-model-replica/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-model-replica/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-model-replica/tests/test_backends.py b/functions/compose-model-replica/tests/test_backends.py index e00f7cef6..da5194def 100644 --- a/functions/compose-model-replica/tests/test_backends.py +++ b/functions/compose-model-replica/tests/test_backends.py @@ -23,9 +23,9 @@ """ import dataclasses -import unittest -from typing import Any, ClassVar +from typing import Any +import pytest from crossplane.function import resource from function import routing from function.backends import base, grove, llmd, native @@ -375,7 +375,7 @@ class Case: stack: str = "Standard" -_CASES = [ +MANIFESTS_CASES = [ Case( name="native Standalone engine composes a Deployment", backend=native.NativeBackend(), @@ -392,962 +392,1021 @@ class Case: ] -class TestBackendManifests(unittest.TestCase): - def test_manifests(self) -> None: - for case in _CASES: - with self.subTest(case.name): - replica = _replica(engines=[case.engine]) - out = case.backend.build(replica, case.engine, _PC, base.serving_label(replica), case.stack) - got = {key: obj.spec.forProvider.manifest for key, obj in out.items()} - self.assertEqual(case.want, got, "-want, +got") - - def test_leader_address_env_injected_but_not_rank(self) -> None: - # The Grove backend injects MODELPLANE_LEADER_ADDRESS (aliasing Grove's - # own GROVE_PCSG_* vars) but not MODELPLANE_RANK: Grove exposes no - # group-wide pod index yet (grove#755, open), so a gang engine's - # command computes its own rank from GROVE_PCLQ_POD_INDEX directly - # (see grove.py and the multinode example). - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - # Spelled out rather than compared against grove_leader_address_env(), - # which would pass whatever that function returned. The PCSG vars are - # what make the address vary per gang; the PCS-scoped ones are - # identical across gangs and would silently point every copy at gang - # 0's leader. - want = { - "name": "MODELPLANE_LEADER_ADDRESS", - "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", - } - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - self.assertEqual(container["env"], [want]) - - def test_user_env_passed_through(self) -> None: - # A member's own env passes through verbatim, after the leader-address - # alias (see test_leader_address_env_injected_but_not_rank). - engine = _gang_engine( - leader_command=_LEADER_CMD, - worker_command=_WORKER_CMD, - ) - spec = engine.members[0].template.spec - assert spec is not None - spec.containers[0].env = [v1alpha1.EnvItem(name="HF_TOKEN", value="x")] - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader")["spec"]["podSpec"] - env = leader["containers"][0]["env"] - self.assertEqual(env, [base.grove_leader_address_env(), {"name": "HF_TOKEN", "value": "x"}]) - - def test_fieldref_env_passes_through(self) -> None: - # A pod-field env (e.g. VLLM_HOST_IP from status.podIP, which multi-NIC - # RDMA nodes need so the engine binds the right interface — #141) survives - # model_dump into the composed manifest. - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - spec = engine.members[0].template.spec - assert spec is not None - spec.containers[0].env = [ - v1alpha1.EnvItem( - name="VLLM_HOST_IP", - valueFrom=v1alpha1.ValueFrom(fieldRef=v1alpha1.FieldRef(fieldPath="status.podIP")), - ) - ] - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader")["spec"]["podSpec"] - env = leader["containers"][0]["env"] - self.assertEqual( - env, - [ - base.grove_leader_address_env(), - {"name": "VLLM_HOST_IP", "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}}, - ], +@pytest.mark.parametrize("case", MANIFESTS_CASES, ids=lambda case: case.name) +def test_manifests(case: Case) -> None: + """A backend composes an engine's manifests.""" + replica = _replica(engines=[case.engine]) + out = case.backend.build(replica, case.engine, _PC, base.serving_label(replica), case.stack) + got = {key: obj.spec.forProvider.manifest for key, obj in out.items()} + assert got == case.want + + +def test_leader_address_env_injected_but_not_rank() -> None: + """The Grove backend injects a leader address alias, but no rank.""" + # The Grove backend injects MODELPLANE_LEADER_ADDRESS (aliasing Grove's + # own GROVE_PCSG_* vars) but not MODELPLANE_RANK: Grove exposes no + # group-wide pod index yet (grove#755, open), so a gang engine's + # command computes its own rank from GROVE_PCLQ_POD_INDEX directly + # (see grove.py and the multinode example). + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + replica = _replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + # Spelled out rather than compared against grove_leader_address_env(), + # which would pass whatever that function returned. The PCSG vars are + # what make the address vary per gang; the PCS-scoped ones are + # identical across gangs and would silently point every copy at gang + # 0's leader. + want = { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + for clique_name in ("leader", "worker"): + container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] + assert container["env"] == [want] + + +def test_user_env_passed_through() -> None: + """A Grove member's own env follows the leader address alias.""" + # A member's own env passes through verbatim, after the leader-address + # alias (see test_leader_address_env_injected_but_not_rank). + engine = _gang_engine( + leader_command=_LEADER_CMD, + worker_command=_WORKER_CMD, + ) + spec = engine.members[0].template.spec + assert spec is not None + spec.containers[0].env = [v1alpha1.EnvItem(name="HF_TOKEN", value="x")] + replica = _replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + leader = _clique(manifest, "leader")["spec"]["podSpec"] + env = leader["containers"][0]["env"] + assert env == [base.grove_leader_address_env(), {"name": "HF_TOKEN", "value": "x"}] + + +def test_fieldref_env_passes_through() -> None: + """A Grove member's pod-field env survives into the composed manifest.""" + # A pod-field env (e.g. VLLM_HOST_IP from status.podIP, which multi-NIC + # RDMA nodes need so the engine binds the right interface — #141) survives + # model_dump into the composed manifest. + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + spec = engine.members[0].template.spec + assert spec is not None + spec.containers[0].env = [ + v1alpha1.EnvItem( + name="VLLM_HOST_IP", + valueFrom=v1alpha1.ValueFrom(fieldRef=v1alpha1.FieldRef(fieldPath="status.podIP")), ) + ] + replica = _replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + leader = _clique(manifest, "leader")["spec"]["podSpec"] + env = leader["containers"][0]["env"] + assert env == [ + base.grove_leader_address_env(), + {"name": "VLLM_HOST_IP", "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}}, + ] - def test_member_metadata_propagates_to_native_pod_template(self) -> None: - # A Standalone member's template.metadata labels and annotations land - # on the Deployment's pod template, merged with the managed labels - # (#378). - engine = _standalone_engine() - engine.members[0].template.metadata = v1alpha1.Metadata( - labels={"example.com/role": "standalone"}, - annotations={"example.com/config": "standalone"}, - ) - replica = _replica(engines=[engine]) - out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - meta = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["metadata"] - self.assertEqual( - meta["labels"], - {"example.com/role": "standalone", _SERVING: "r", _WORKLOAD: _WORKLOAD_NAME}, - ) - self.assertEqual(meta["annotations"], {"example.com/config": "standalone"}) - - def test_member_metadata_propagates_to_cliques_independently(self) -> None: - # Leader metadata lands on the leader clique and worker metadata on the - # worker clique; neither leaks into the other. Grove propagates a - # clique's labels and annotations to its pods (#378). - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - engine.members[0].template.metadata = v1alpha1.Metadata( - labels={"example.com/role": "leader"}, annotations={"example.com/config": "leader"} - ) - engine.members[1].template.metadata = v1alpha1.Metadata( - labels={"example.com/role": "worker"}, annotations={"example.com/config": "worker"} - ) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader") - self.assertEqual( - leader["labels"], - {"example.com/role": "leader", _SERVING: "r", _QUEUE_LABEL: _QUEUE, _CLIQUE_ROLE: "leader"}, - ) - self.assertEqual(leader["annotations"], {"example.com/config": "leader"}) - worker = _clique(manifest, "worker") - self.assertEqual(worker["labels"], {"example.com/role": "worker", _QUEUE_LABEL: _QUEUE}) - self.assertEqual(worker["annotations"], {"example.com/config": "worker"}) - - def test_worker_without_metadata_composes_only_managed_labels(self) -> None: - # A worker member with no template.metadata composes a worker clique - # carrying only the managed queue label and no annotations key. - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - worker = _clique(manifest, "worker") - self.assertEqual(worker["labels"], {_QUEUE_LABEL: _QUEUE}) - self.assertNotIn("annotations", worker) - - @staticmethod - def _names(out: dict[str, k8sobjv1alpha1.Object]) -> set[str]: - return {o.spec.forProvider.manifest["metadata"]["name"] for o in out.values()} - - def test_co_located_replicas_get_distinct_names(self) -> None: - # Two replicas of one deployment on the same cluster must produce - # distinct resource names on the remote cluster. - a = _replica("dep-clusterA") - b = _replica("dep-clusterB") - out_a = native.NativeBackend().build(a, a.spec.engines[0], _PC, base.serving_label(a), "Standard") - out_b = native.NativeBackend().build(b, b.spec.engines[0], _PC, base.serving_label(b), "Standard") - self.assertEqual(self._names(out_a) & self._names(out_b), set()) - - def test_multi_engine_qualifies_workload_names(self) -> None: - # A replica with two engines names each engine's workload distinctly so - # they don't collide on the remote cluster. - engines = [_standalone_engine("prefill"), _standalone_engine("decode")] - replica = _replica(engines=engines) - names = set() - for g in engines: - out = native.NativeBackend().build(replica, g, _PC, base.serving_label(replica), "Standard") - names |= self._names(out) - self.assertEqual(len(names), 4) # 2 deployments + 2 claim templates - - def test_workload_readiness_policies(self) -> None: - # A Deployment reports readiness from its Available condition; a - # PodCliqueSet publishes no such condition, so it's derived from its - # replica counters instead (base.GROVE_AVAILABLE_CEL). Either way the - # claim templates are ready on create. - for name, backend, engine, stack, want_cel in ( - ("native", native.NativeBackend(), _standalone_engine(), "Standard", base.AVAILABLE_CEL), - ( - "grove", - grove.GroveBackend(), - _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD), - "Dynamo", - base.GROVE_AVAILABLE_CEL, - ), - ): - with self.subTest(name): - replica = _replica(engines=[engine]) - out = backend.build(replica, engine, _PC, base.serving_label(replica), stack) - serving = out["model-serving-main"].spec.readiness - assert serving is not None - self.assertEqual(serving.policy, "DeriveFromCelQuery") - self.assertEqual(serving.celQuery, want_cel) - for key, obj in out.items(): - if key.startswith("resource-claim"): - readiness = obj.spec.readiness - assert readiness is not None - self.assertEqual(readiness.policy, "SuccessfulCreate") - - def test_multiple_device_requests_single_container_claim(self) -> None: - # resources.claims is a list-map keyed on name alone, so N device - # requests must NOT produce N container claims all named "devices". The - # container references the whole pod claim once; the template carries all - # requests. - engine = _standalone_engine( - device_requests=[ - v1alpha1.DeviceRequest(name="gpu", deviceClassName="gpu.nvidia.com", count=8), - v1alpha1.DeviceRequest(name="nic", deviceClassName="nic.nvidia.com", count=8), - ], - ) - replica = _replica(engines=[engine]) - out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - pod = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"] - claims = pod["containers"][0]["resources"]["claims"] - self.assertEqual(claims, [{"name": "devices"}]) - self.assertEqual(pod["resourceClaims"][0]["name"], "devices") - template = out["resource-claim-main-standalone"].spec.forProvider.manifest - template_requests = template["spec"]["spec"]["devices"]["requests"] - self.assertEqual([r["name"] for r in template_requests], ["gpu", "nic"]) - claim_readiness = out["resource-claim-main-standalone"].spec.readiness - assert claim_readiness is not None - self.assertEqual(claim_readiness.policy, "SuccessfulCreate") - - def test_claimless_leader_gets_no_claim(self) -> None: - # A coordinator-only leader (e.g. a vLLM DP head running - # --data-parallel-size-local=0) carries no deviceRequests. Its pod must - # get no resourceClaims, its container no resources.claims, and no - # leader ResourceClaimTemplate must be composed - only the worker's. - # It still pins to its pool and tolerates the GPU taint. - engine = _gang_engine( - leader_command=_LEADER_CMD, - worker_command=_WORKER_CMD, - leader_device_requests=[], - ) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - - self.assertNotIn("resource-claim-main-leader", out) - self.assertIn("resource-claim-main-worker", out) - - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader")["spec"]["podSpec"] - self.assertNotIn("resourceClaims", leader) - self.assertNotIn("resources", leader["containers"][0]) - self.assertEqual(leader["nodeSelector"], {"modelplane.ai/pool": "frontier"}) - self.assertEqual( - leader["tolerations"], [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}] - ) - worker = _clique(manifest, "worker")["spec"]["podSpec"] - self.assertEqual(worker["resourceClaims"], _claims("worker")) - self.assertEqual(worker["containers"][0]["resources"], {"claims": [{"name": "devices"}]}) - - def test_members_pin_to_their_own_pools(self) -> None: - # The scheduler may split a gang across pools when no single pool - # satisfies every member. Each member's pods must pin to that member's - # pool, not a shared engine-wide one. - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD, leader_pool="head") - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - self.assertEqual(_clique(manifest, "leader")["spec"]["podSpec"]["nodeSelector"], {"modelplane.ai/pool": "head"}) - self.assertEqual( - _clique(manifest, "worker")["spec"]["podSpec"]["nodeSelector"], {"modelplane.ai/pool": "frontier"} - ) +def test_member_metadata_propagates_to_native_pod_template() -> None: + """A Standalone member's template metadata lands on the Deployment's pod template.""" + # A Standalone member's template.metadata labels and annotations land + # on the Deployment's pod template, merged with the managed labels + # (#378). + engine = _standalone_engine() + engine.members[0].template.metadata = v1alpha1.Metadata( + labels={"example.com/role": "standalone"}, + annotations={"example.com/config": "standalone"}, + ) + replica = _replica(engines=[engine]) + out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") + meta = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["metadata"] + assert meta["labels"] == {"example.com/role": "standalone", _SERVING: "r", _WORKLOAD: _WORKLOAD_NAME} + assert meta["annotations"] == {"example.com/config": "standalone"} + + +def test_member_metadata_propagates_to_cliques_independently() -> None: + """Each Grove member's template metadata lands on its own clique only.""" + # Leader metadata lands on the leader clique and worker metadata on the + # worker clique; neither leaks into the other. Grove propagates a + # clique's labels and annotations to its pods (#378). + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + engine.members[0].template.metadata = v1alpha1.Metadata( + labels={"example.com/role": "leader"}, annotations={"example.com/config": "leader"} + ) + engine.members[1].template.metadata = v1alpha1.Metadata( + labels={"example.com/role": "worker"}, annotations={"example.com/config": "worker"} + ) + replica = _replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + leader = _clique(manifest, "leader") + assert leader["labels"] == { + "example.com/role": "leader", + _SERVING: "r", + _QUEUE_LABEL: _QUEUE, + _CLIQUE_ROLE: "leader", + } + assert leader["annotations"] == {"example.com/config": "leader"} + worker = _clique(manifest, "worker") + assert worker["labels"] == {"example.com/role": "worker", _QUEUE_LABEL: _QUEUE} + assert worker["annotations"] == {"example.com/config": "worker"} + + +def test_worker_without_metadata_composes_only_managed_labels() -> None: + """A Grove worker with no template metadata carries only the queue label.""" + # A worker member with no template.metadata composes a worker clique + # carrying only the managed queue label and no annotations key. + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + replica = _replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + worker = _clique(manifest, "worker") + assert worker["labels"] == {_QUEUE_LABEL: _QUEUE} + assert "annotations" not in worker + + +def _names(out: dict[str, k8sobjv1alpha1.Object]) -> set[str]: + """The names of the manifests a backend composed.""" + return {o.spec.forProvider.manifest["metadata"]["name"] for o in out.values()} + + +def test_co_located_replicas_get_distinct_names() -> None: + """Two replicas on one cluster compose distinct resource names.""" + # Two replicas of one deployment on the same cluster must produce + # distinct resource names on the remote cluster. + a = _replica("dep-clusterA") + b = _replica("dep-clusterB") + out_a = native.NativeBackend().build(a, a.spec.engines[0], _PC, base.serving_label(a), "Standard") + out_b = native.NativeBackend().build(b, b.spec.engines[0], _PC, base.serving_label(b), "Standard") + assert _names(out_a) & _names(out_b) == set() + + +def test_multi_engine_qualifies_workload_names() -> None: + """Each engine of a multi-engine replica composes distinctly named resources.""" + # A replica with two engines names each engine's workload distinctly so + # they don't collide on the remote cluster. + engines = [_standalone_engine("prefill"), _standalone_engine("decode")] + replica = _replica(engines=engines) + names = set() + for g in engines: + out = native.NativeBackend().build(replica, g, _PC, base.serving_label(replica), "Standard") + names |= _names(out) + assert len(names) == 4 # 2 deployments + 2 claim templates + + +@pytest.mark.parametrize( + ("backend", "engine", "stack", "want_cel"), + [ + pytest.param(native.NativeBackend(), _standalone_engine(), "Standard", base.AVAILABLE_CEL, id="native"), + pytest.param( + grove.GroveBackend(), + _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD), + "Dynamo", + base.GROVE_AVAILABLE_CEL, + id="grove", + ), + ], +) +def test_workload_readiness_policies(backend: base.Backend, engine: v1alpha1.Engine, stack: str, want_cel: str) -> None: + """A workload's readiness derives from its status, and a claim template's from its creation.""" + # A Deployment reports readiness from its Available condition; a + # PodCliqueSet publishes no such condition, so it's derived from its + # replica counters instead (base.GROVE_AVAILABLE_CEL). Either way the + # claim templates are ready on create. + replica = _replica(engines=[engine]) + out = backend.build(replica, engine, _PC, base.serving_label(replica), stack) + serving = out["model-serving-main"].spec.readiness + assert serving is not None + assert serving.policy == "DeriveFromCelQuery" + assert serving.celQuery == want_cel + for key, obj in out.items(): + if key.startswith("resource-claim"): + readiness = obj.spec.readiness + assert readiness is not None + assert readiness.policy == "SuccessfulCreate" + + +def test_multiple_device_requests_single_container_claim() -> None: + """Several device requests compose one container claim and one template carrying them all.""" + # resources.claims is a list-map keyed on name alone, so N device + # requests must NOT produce N container claims all named "devices". The + # container references the whole pod claim once; the template carries all + # requests. + engine = _standalone_engine( + device_requests=[ + v1alpha1.DeviceRequest(name="gpu", deviceClassName="gpu.nvidia.com", count=8), + v1alpha1.DeviceRequest(name="nic", deviceClassName="nic.nvidia.com", count=8), + ], + ) + replica = _replica(engines=[engine]) + out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") + pod = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"] + claims = pod["containers"][0]["resources"]["claims"] + assert claims == [{"name": "devices"}] + assert pod["resourceClaims"][0]["name"] == "devices" + template = out["resource-claim-main-standalone"].spec.forProvider.manifest + template_requests = template["spec"]["spec"]["devices"]["requests"] + assert [r["name"] for r in template_requests] == ["gpu", "nic"] + claim_readiness = out["resource-claim-main-standalone"].spec.readiness + assert claim_readiness is not None + assert claim_readiness.policy == "SuccessfulCreate" + + +def test_claimless_leader_gets_no_claim() -> None: + """A Grove leader with no device requests composes no claim, but still pins and tolerates.""" + # A coordinator-only leader (e.g. a vLLM DP head running + # --data-parallel-size-local=0) carries no deviceRequests. Its pod must + # get no resourceClaims, its container no resources.claims, and no + # leader ResourceClaimTemplate must be composed - only the worker's. + # It still pins to its pool and tolerates the GPU taint. + engine = _gang_engine( + leader_command=_LEADER_CMD, + worker_command=_WORKER_CMD, + leader_device_requests=[], + ) + replica = _replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + + assert "resource-claim-main-leader" not in out + assert "resource-claim-main-worker" in out + + manifest = out["model-serving-main"].spec.forProvider.manifest + leader = _clique(manifest, "leader")["spec"]["podSpec"] + assert "resourceClaims" not in leader + assert "resources" not in leader["containers"][0] + assert leader["nodeSelector"] == {"modelplane.ai/pool": "frontier"} + assert leader["tolerations"] == [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}] + + worker = _clique(manifest, "worker")["spec"]["podSpec"] + assert worker["resourceClaims"] == _claims("worker") + assert worker["containers"][0]["resources"] == {"claims": [{"name": "devices"}]} + + +def test_members_pin_to_their_own_pools() -> None: + """Each Grove member's pods pin to that member's own pool.""" + # The scheduler may split a gang across pools when no single pool + # satisfies every member. Each member's pods must pin to that member's + # pool, not a shared engine-wide one. + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD, leader_pool="head") + replica = _replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + assert _clique(manifest, "leader")["spec"]["podSpec"]["nodeSelector"] == {"modelplane.ai/pool": "head"} + assert _clique(manifest, "worker")["spec"]["podSpec"]["nodeSelector"] == {"modelplane.ai/pool": "frontier"} + + +# The LeaderWorkerSet backend for a Leader/Worker gang engine. + +_LWS_ROLE = "modelplane.ai/lws-role" + + +def _llmd_lws(engine: v1alpha1.Engine, replica: v1alpha1.ModelReplica) -> dict: + """The LeaderWorkerSet manifest the llm-d backend composes for engine.""" + out = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") + return out["model-serving-main"].spec.forProvider.manifest + + +def test_llmd_leader_worker_set_shape() -> None: + """The llm-d backend composes a LeaderWorkerSet of copies gangs, each the leader plus its workers.""" + engine = _gang_engine(nodes=3, copies=2) + replica = _replica(engines=[engine]) + manifest = _llmd_lws(engine, replica) + assert manifest["apiVersion"] == "leaderworkerset.x-k8s.io/v1" + assert manifest["kind"] == "LeaderWorkerSet" + assert manifest["metadata"] == {"name": _WORKLOAD_NAME, "namespace": "mp-ml-team-51733"} + assert manifest["spec"]["replicas"] == 2 + # Gang size is the leader plus the worker's node count. + assert manifest["spec"]["leaderWorkerTemplate"]["size"] == 4 + + +def test_llmd_only_leader_carries_serving_label() -> None: + """Only the LeaderWorkerSet's leader carries the serving label.""" + engine = _gang_engine() + replica = _replica(engines=[engine]) + lwt = _llmd_lws(engine, replica)["spec"]["leaderWorkerTemplate"] + leader_labels = lwt["leaderTemplate"]["metadata"]["labels"] + assert leader_labels[_SERVING] == "r" + assert leader_labels[_LWS_ROLE] == "leader" + # The worker followers never serve, so they carry no metadata at all. + assert "metadata" not in lwt["workerTemplate"] + + +def test_llmd_leader_address_and_rank_env_injected() -> None: + """Every LeaderWorkerSet container leads with the leader address and rank aliases.""" + # Every gang container leads with the backend-neutral coordination vars + # aliasing LWS_LEADER_ADDRESS / LWS_WORKER_INDEX. + engine = _gang_engine() + replica = _replica(engines=[engine]) + lwt = _llmd_lws(engine, replica)["spec"]["leaderWorkerTemplate"] + for tmpl in (lwt["leaderTemplate"], lwt["workerTemplate"]): + env = tmpl["spec"]["containers"][0]["env"] + assert env[0] == {"name": "MODELPLANE_LEADER_ADDRESS", "value": "$(LWS_LEADER_ADDRESS)"} + assert env[1] == {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"} + + +def test_llmd_no_modelexpress_env_even_with_a_cache() -> None: + """The llm-d backend injects no ModelExpress env, even for a replica with a cache.""" + # The llm-d (Standard) backend never injects ModelExpress env: that P2P + # wiring is the Grove (Dynamo) backend's, gated on the cluster stack. + engine = _gang_engine() + replica = v1alpha1.ModelReplica( + metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), + spec=v1alpha1.SpecModel( + clusterName="cluster-a", + modelCacheRef=v1alpha1.ModelCacheRef(name="c"), + engines=[engine], + ), + ) + lwt = _llmd_lws(engine, replica)["spec"]["leaderWorkerTemplate"] + for tmpl in (lwt["leaderTemplate"], lwt["workerTemplate"]): + container = tmpl["spec"]["containers"][0] + env_names = [e["name"] for e in container["env"]] + # HF_HUB_CACHE is the cache's own env (every stack); the MX bundle + # is not. + assert env_names == ["MODELPLANE_LEADER_ADDRESS", "MODELPLANE_RANK", "HF_HUB_CACHE"] + assert "MX_SERVER_ADDRESS" not in env_names + assert "securityContext" not in container -class TestLLMDBackend(unittest.TestCase): - """The LeaderWorkerSet backend for a Leader/Worker gang engine.""" - - _LWS_ROLE = "modelplane.ai/lws-role" - - @staticmethod - def _lws(engine: v1alpha1.Engine, replica: v1alpha1.ModelReplica) -> dict: - out = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - return out["model-serving-main"].spec.forProvider.manifest - - def test_leader_worker_set_shape(self) -> None: - engine = _gang_engine(nodes=3, copies=2) - replica = _replica(engines=[engine]) - manifest = self._lws(engine, replica) - self.assertEqual(manifest["apiVersion"], "leaderworkerset.x-k8s.io/v1") - self.assertEqual(manifest["kind"], "LeaderWorkerSet") - self.assertEqual(manifest["metadata"], {"name": _WORKLOAD_NAME, "namespace": "mp-ml-team-51733"}) - self.assertEqual(manifest["spec"]["replicas"], 2) - # Gang size is the leader plus the worker's node count. - self.assertEqual(manifest["spec"]["leaderWorkerTemplate"]["size"], 4) - - def test_only_leader_carries_serving_label(self) -> None: - engine = _gang_engine() - replica = _replica(engines=[engine]) - lwt = self._lws(engine, replica)["spec"]["leaderWorkerTemplate"] - leader_labels = lwt["leaderTemplate"]["metadata"]["labels"] - self.assertEqual(leader_labels[_SERVING], "r") - self.assertEqual(leader_labels[self._LWS_ROLE], "leader") - # The worker followers never serve, so they carry no metadata at all. - self.assertNotIn("metadata", lwt["workerTemplate"]) - - def test_leader_address_and_rank_env_injected(self) -> None: - # Every gang container leads with the backend-neutral coordination vars - # aliasing LWS_LEADER_ADDRESS / LWS_WORKER_INDEX. - engine = _gang_engine() - replica = _replica(engines=[engine]) - lwt = self._lws(engine, replica)["spec"]["leaderWorkerTemplate"] - for tmpl in (lwt["leaderTemplate"], lwt["workerTemplate"]): - env = tmpl["spec"]["containers"][0]["env"] - self.assertEqual(env[0], {"name": "MODELPLANE_LEADER_ADDRESS", "value": "$(LWS_LEADER_ADDRESS)"}) - self.assertEqual(env[1], {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}) - - def test_no_modelexpress_env_even_with_a_cache(self) -> None: - # The llm-d (Standard) backend never injects ModelExpress env: that P2P - # wiring is the Grove (Dynamo) backend's, gated on the cluster stack. - engine = _gang_engine() - replica = v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="c"), - engines=[engine], - ), - ) - lwt = self._lws(engine, replica)["spec"]["leaderWorkerTemplate"] - for tmpl in (lwt["leaderTemplate"], lwt["workerTemplate"]): - container = tmpl["spec"]["containers"][0] - env_names = [e["name"] for e in container["env"]] - # HF_HUB_CACHE is the cache's own env (every stack); the MX bundle - # is not. - self.assertEqual(env_names, ["MODELPLANE_LEADER_ADDRESS", "MODELPLANE_RANK", "HF_HUB_CACHE"]) - self.assertNotIn("MX_SERVER_ADDRESS", env_names) - self.assertNotIn("securityContext", container) - - def test_workload_readiness_uses_available_cel(self) -> None: - engine = _gang_engine() - replica = _replica(engines=[engine]) - out = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - readiness = out["model-serving-main"].spec.readiness - assert readiness is not None - self.assertEqual(readiness.policy, "DeriveFromCelQuery") - self.assertEqual(readiness.celQuery, base.AVAILABLE_CEL) - - -class TestBackendSelection(unittest.TestCase): - def test_standalone_engine_is_native(self) -> None: - # A Standalone engine is native regardless of the cluster's stack. - self.assertEqual(base.select_backend(_standalone_engine(), "Standard"), base.NATIVE) - self.assertEqual(base.select_backend(_standalone_engine(), "Dynamo"), base.NATIVE) - - def test_leader_worker_engine_is_llmd(self) -> None: - self.assertEqual(base.select_backend(_gang_engine(), "Standard"), base.LLMD) - - def test_leader_worker_engine_is_grove(self) -> None: - self.assertEqual(base.select_backend(_gang_engine(), "Dynamo"), base.GROVE) - - -class TestCacheMounts(unittest.TestCase): - def _replica( - self, *, cache: str | None = None, args: list[str] | None = None, command: list[str] | None = None - ) -> v1alpha1.ModelReplica: - engine = _standalone_engine(args=args or [], command=command) - modelcache = v1alpha1.ModelCacheRef(name=cache) if cache else None - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(namespace="ml-team"), - spec=v1alpha1.SpecModel(clusterName="c", modelCacheRef=modelcache, engines=[engine]), - ) +def test_llmd_workload_readiness_uses_available_cel() -> None: + """The LeaderWorkerSet's readiness derives from its Available condition.""" + engine = _gang_engine() + replica = _replica(engines=[engine]) + out = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") + readiness = out["model-serving-main"].spec.readiness + assert readiness is not None + assert readiness.policy == "DeriveFromCelQuery" + assert readiness.celQuery == base.AVAILABLE_CEL - @staticmethod - def _engine(replica: v1alpha1.ModelReplica) -> v1alpha1.Container: - spec = replica.spec.engines[0].members[0].template.spec - assert spec is not None - return spec.containers[0] - - def test_no_cache_no_mounts(self) -> None: - volumes, mounts = base.cache_mounts(self._replica()) - self.assertEqual((volumes, mounts), ([], [])) - - def test_cache_adds_volume_and_mount(self) -> None: - volumes, mounts = base.cache_mounts(self._replica(cache="qwen")) - self.assertEqual( - volumes, - [{"name": "model-cache", "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}}], - ) - self.assertEqual(mounts, [{"name": "model-cache", "mountPath": "/mnt/models"}]) - - def test_cache_env_points_huggingface_at_the_mount(self) -> None: - # The cache is staged in HuggingFace's cache layout, so pointing - # HF_HUB_CACHE at the mount is what lets an engine's own --model= - # resolve against it instead of pulling from HuggingFace (#407). - self.assertEqual( - base.cache_env(self._replica(cache="qwen")), - [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}], - ) - def test_cache_env_empty_without_cache(self) -> None: - self.assertEqual(base.cache_env(self._replica()), []) - - def test_cache_env_sets_no_offline_flag(self) -> None: - # HF_HUB_OFFLINE would break an engine that fetches a *different* repo - # at startup (kimi-k2's separately-gated tokenizer), and resolution - # doesn't need it. - names = {e["name"] for e in base.cache_env(self._replica(cache="qwen"))} - self.assertNotIn("HF_HUB_OFFLINE", names) - - -class TestNativeBackendCache(unittest.TestCase): - def _replica(self) -> v1alpha1.ModelReplica: - engine = _standalone_engine(args=[]) - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), - engines=[engine], - ), - ) +def test_select_backend_standalone_engine_is_native() -> None: + """A Standalone engine selects the native backend.""" + # A Standalone engine is native regardless of the cluster's stack. + assert base.select_backend(_standalone_engine(), "Standard") == base.NATIVE + assert base.select_backend(_standalone_engine(), "Dynamo") == base.NATIVE - def test_mounts_pvc_and_sets_cache_env(self) -> None: - # A cache contributes a volume, a mount, and the HF_HUB_CACHE that makes - # the engine's own --model= resolve against it. Modelplane injects - # no --model of its own: naming the model is the command's job. - replica = self._replica() - out = native.NativeBackend().build( - replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Standard" - ) - dep = out["model-serving-main"].spec.forProvider.manifest - pod = dep["spec"]["template"]["spec"] - vol_names = {v["name"] for v in pod["volumes"]} - self.assertIn("model-cache", vol_names) - container = pod["containers"][0] - self.assertIn({"name": "model-cache", "mountPath": "/mnt/models"}, container["volumeMounts"]) - self.assertIn({"name": "HF_HUB_CACHE", "value": "/mnt/models"}, container["env"]) - self.assertEqual(container["args"], []) - - def test_user_env_comes_after_cache_env(self) -> None: - # Kubernetes expands $(VAR) left to right, so Modelplane's own entries - # must precede the user's for a user entry to reference them. - replica = self._replica() - engine = replica.spec.engines[0] - spec = engine.members[0].template.spec - assert spec is not None - spec.containers[0].env = [v1alpha1.EnvItem(name="HF_TOKEN", value="x")] - out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] - self.assertEqual( - container["env"], - [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}, {"name": "HF_TOKEN", "value": "x"}], - ) +def test_select_backend_leader_worker_engine_is_llmd() -> None: + """A Leader/Worker engine on a Standard cluster selects the llm-d backend.""" + assert base.select_backend(_gang_engine(), "Standard") == base.LLMD -class TestGroveBackendCache(unittest.TestCase): - def _replica( - self, - *, - leader_command: list[str] | None = None, - worker_command: list[str] | None = None, - leader_args: list[str] | None = None, - worker_args: list[str] | None = None, - ) -> v1alpha1.ModelReplica: - engine = _gang_engine( - leader_command=leader_command, - worker_command=worker_command, - leader_args=leader_args, - worker_args=worker_args, - ) - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="kimi"), - engines=[engine], - ), - ) - def test_both_grove_cliques_mount_cache(self) -> None: - replica = self._replica(leader_args=[], worker_command=["/bin/sh", "-c", "join"]) - manifest = ( - grove.GroveBackend() - .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] - .spec.forProvider.manifest - ) - for clique_name in ("leader", "worker"): - pod = _clique(manifest, clique_name)["spec"]["podSpec"] - self.assertIn("model-cache", {v["name"] for v in pod["volumes"]}) - self.assertIn( - {"name": "model-cache", "mountPath": "/mnt/models"}, - pod["containers"][0]["volumeMounts"], - ) - - def test_sets_cache_env_on_every_clique_and_injects_no_model(self) -> None: - # A cache gives both cliques HF_HUB_CACHE so their own --model= - # resolves against the mount; Modelplane adds no --model itself. - replica = self._replica(leader_args=[], worker_command=["/bin/sh", "-c", "join"]) - manifest = ( - grove.GroveBackend() - .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] - .spec.forProvider.manifest - ) - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - self.assertIn({"name": "HF_HUB_CACHE", "value": "/mnt/models"}, container["env"]) - self.assertNotIn("--model=/mnt/models", container.get("args", [])) - - def test_command_engine_mounts_cache_without_injecting_model(self) -> None: - # A member with its own command keeps it verbatim and gets no injected - # --model (it points at the cache with its own flag). - leader_cmd = [ - "/bin/sh", - "-c", - "python3 -m sglang.launch_server --model-path /mnt/models --tp 16", - ] - replica = self._replica(leader_command=leader_cmd, worker_command=["/bin/sh", "-c", "join"]) - manifest = ( - grove.GroveBackend() - .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] - .spec.forProvider.manifest - ) - leader = _clique(manifest, "leader")["spec"]["podSpec"]["containers"][0] - self.assertIn( - {"name": "model-cache", "mountPath": "/mnt/models"}, - leader["volumeMounts"], - ) - self.assertEqual(leader["command"], leader_cmd) - - -class TestDisaggregated(unittest.TestCase): - """serving.mode: PrefillDecode routing layers an InferencePool + endpoint - picker over two engines, role-labels them, and sidecars decode — no unified - Service. Mirrors how fn.py composes engines then calls routing.apply.""" - - def _apply(self) -> dict[str, k8sobjv1alpha1.Object]: - prefill = _standalone_engine(name="prefill") - prefill.phase = "Prefill" - decode = _standalone_engine(name="decode") - decode.phase = "Decode" - replica = _replica(engines=[prefill, decode]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = {} - for engine in replica.spec.engines: - composed.update(native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard")) - return routing.apply(composed, replica, _PC) - - def _serving_pod(self, out: dict[str, k8sobjv1alpha1.Object], engine_name: str) -> dict: - return out[f"model-serving-{engine_name}"].spec.forProvider.manifest["spec"]["template"] - - def test_replaces_unified_service_with_pool_and_epp(self) -> None: - out = self._apply() - self.assertIn("inference-pool", out) - self.assertIn("epp", out) - self.assertIn("epp-config", out) - pool = out["inference-pool"].spec.forProvider.manifest - self.assertEqual(pool["kind"], "InferencePool") - self.assertEqual(pool["spec"]["endpointPickerRef"]["name"], "r-epp") - - def test_injects_nixl_plumbing(self) -> None: - """Both disagg engines get the NIXL plumbing the schema can't express: - a Memory /dev/shm and VLLM_NIXL_SIDE_CHANNEL_HOST = pod IP.""" - out = self._apply() - for role in ("prefill", "decode"): - pod = self._serving_pod(out, role)["spec"] - self.assertTrue( - any(v.get("emptyDir", {}).get("medium") == "Memory" for v in pod["volumes"]), - f"{role} missing Memory /dev/shm volume", - ) - engine = next(c for c in pod["containers"] if c["name"] == "engine") - self.assertIn("/dev/shm", [m["mountPath"] for m in engine["volumeMounts"]]) - host = next((e for e in engine["env"] if e["name"] == "VLLM_NIXL_SIDE_CHANNEL_HOST"), None) - assert host is not None, f"{role} missing VLLM_NIXL_SIDE_CHANNEL_HOST" - self.assertEqual(host["valueFrom"]["fieldRef"]["fieldPath"], "status.podIP") - self.assertIn("VLLM_NIXL_SIDE_CHANNEL_PORT", [e["name"] for e in engine["env"]]) - - def test_epp_config_arms_the_pd_decider(self) -> None: - """PrefillDecode silently serves decode-only unless the PD decider is armed. - - Selective prefix-based-pd-decider needs all of: nonCachedTokens > 0 (0 = - disabled), the approx-prefix-cache-producer plugin that populates the - attribute it reads, and that producer pinned to autoTune: false (the - true default never populates). And it must NOT carry the prepareDataPlugins - feature gate, which the v0.8.0 EPP image rejects and crashloops on. - """ - cfg = self._apply()["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - self.assertIn("prefix-based-pd-decider", cfg) - self.assertIn("nonCachedTokens: 16", cfg) - self.assertIn("approx-prefix-cache-producer", cfg) - self.assertIn("autoTune: false", cfg) - self.assertNotIn("nonCachedTokens: 0", cfg) - self.assertNotIn("prepareDataPlugins", cfg) - - def test_epp_and_sidecar_images_and_config_group_are_pinned(self) -> None: - """Lock the picker + sidecar images and the EndpointPickerConfig API group. - - Nothing else asserts these, so a wrong tag/registry path or a stale config - group passes CI and only surfaces as an EPP/sidecar crashloop at deploy. - These are deliberate literals, not routing._* constants: comparing to the - constant would be tautological (it can't catch a typo in the constant), and - a literal forces a bump to show up here and be reviewed. - """ - out = self._apply() - epp = out["epp"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"] - self.assertEqual( - next(c["image"] for c in epp if c["name"] == "epp"), - "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0", - ) - sidecar = next(c for c in self._serving_pod(out, "decode")["spec"]["containers"] if c["name"] == "pd-sidecar") - self.assertEqual(sidecar["image"], "ghcr.io/llm-d/llm-d-router-disagg-sidecar:v0.9.0") - cfg = out["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - self.assertIn("apiVersion: llm-d.ai/v1alpha1", cfg) - - def test_epp_role_watches_inferenceobjectives(self) -> None: - """The picker watches InferenceObjectives (GIE x-k8s.io group); the Role must allow it.""" - rules = self._apply()["epp-role"].spec.forProvider.manifest["rules"] - self.assertTrue( - any( - "inference.networking.x-k8s.io" in r["apiGroups"] and "inferenceobjectives" in r["resources"] - for r in rules - ), - f"EPP Role missing inferenceobjectives watch: {rules}", - ) +def test_select_backend_leader_worker_engine_is_grove() -> None: + """A Leader/Worker engine on a Dynamo cluster selects the Grove backend.""" + assert base.select_backend(_gang_engine(), "Dynamo") == base.GROVE - def test_decode_port_follows_user_arg(self) -> None: - """The sidecar and the decode container port track the user's --port, not a hardcoded one.""" - prefill = _standalone_engine(name="prefill") - prefill.phase = "Prefill" - decode = _standalone_engine(name="decode", args=["--model=m", "--port=9000"]) - decode.phase = "Decode" - replica = _replica(engines=[prefill, decode]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = {} - for e in replica.spec.engines: - composed.update(native.NativeBackend().build(replica, e, _PC, base.serving_label(replica), "Standard")) - out = routing.apply(composed, replica, _PC) - containers = self._serving_pod(out, "decode")["spec"]["containers"] - engine = next(c for c in containers if c["name"] == "engine") - sidecar = next(c for c in containers if c["name"] == "pd-sidecar") - self.assertEqual(engine["ports"][0]["containerPort"], 9000) - self.assertIn("--vllm-port=9000", sidecar["args"]) - self.assertEqual(sidecar["ports"][0]["containerPort"], 8000) - - def test_engines_role_labeled(self) -> None: - out = self._apply() - self.assertEqual(self._serving_pod(out, "prefill")["metadata"]["labels"]["llm-d.ai/role"], "prefill") - decode_labels = self._serving_pod(out, "decode")["metadata"]["labels"] - self.assertEqual(decode_labels["llm-d.ai/role"], "decode") - self.assertEqual(decode_labels["app"], "r") - - def test_decode_gets_sidecar_and_moves_engine_port(self) -> None: - out = self._apply() - containers = self._serving_pod(out, "decode")["spec"]["containers"] - names = [c["name"] for c in containers] - self.assertEqual(names, ["engine", "pd-sidecar"]) - engine = next(c for c in containers if c["name"] == "engine") - self.assertEqual(engine["ports"][0]["containerPort"], 8001) - self.assertEqual(engine["readinessProbe"]["timeoutSeconds"], 5) - sidecar = next(c for c in containers if c["name"] == "pd-sidecar") - self.assertEqual(sidecar["ports"][0]["containerPort"], 8000) - self.assertEqual(sidecar["readinessProbe"]["timeoutSeconds"], 5) - self.assertIn("--secure-proxy=false", sidecar["args"]) - - def test_prefill_has_no_sidecar(self) -> None: - containers = self._serving_pod(self._apply(), "prefill")["spec"]["containers"] - self.assertEqual([c["name"] for c in containers], ["engine"]) - - def test_route_targets_inference_pool(self) -> None: - route = self._apply()[base.ROUTE_KEY].spec.forProvider.manifest - rule = route["spec"]["rules"][0] - ref = rule["backendRefs"][0] - self.assertEqual(ref["kind"], "InferencePool") - self.assertEqual(ref["name"], "r-pool") - # Disable the request timeout so long token streams aren't severed. - self.assertEqual(rule["timeouts"]["request"], "0s") - - def test_selects_engines_by_phase_not_name(self) -> None: - """Roles come from each engine's phase, not its name.""" - decode = _standalone_engine(name="alpha") - decode.phase = "Decode" - prefill = _standalone_engine(name="beta") - prefill.phase = "Prefill" - replica = _replica(engines=[decode, prefill]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = {} - for e in replica.spec.engines: - composed.update(native.NativeBackend().build(replica, e, _PC, base.serving_label(replica), "Standard")) - out = routing.apply(composed, replica, _PC) - # alpha is Decode -> sidecar; beta is Prefill -> none, despite their names. - self.assertEqual( - [c["name"] for c in self._serving_pod(out, "alpha")["spec"]["containers"]], ["engine", "pd-sidecar"] - ) - self.assertEqual([c["name"] for c in self._serving_pod(out, "beta")["spec"]["containers"]], ["engine"]) - self.assertEqual(self._serving_pod(out, "alpha")["metadata"]["labels"]["llm-d.ai/role"], "decode") - self.assertEqual(self._serving_pod(out, "beta")["metadata"]["labels"]["llm-d.ai/role"], "prefill") - - def test_decode_can_be_a_grove_gang(self) -> None: - """A PrefillDecode engine can itself be a Leader/Worker gang, so routing - must decorate a Grove PodCliqueSet's leader clique - role label, serving - label, pd-sidecar, NIXL plumbing - exactly like a Deployment's pod - template. Exercises the _serving_pod_templates normalization that lets - one routing layer decorate both workload shapes.""" - prefill = _standalone_engine(name="prefill") - prefill.phase = "Prefill" - decode = _gang_engine(name="decode", leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - decode.phase = "Decode" - replica = _replica(engines=[prefill, decode]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = { - **native.NativeBackend().build(replica, prefill, _PC, base.serving_label(replica), "Standard"), - **grove.GroveBackend().build(replica, decode, _PC, base.serving_label(replica), "Dynamo"), - } - out = routing.apply(composed, replica, _PC) - - manifest = out["model-serving-decode"].spec.forProvider.manifest - leader_clique = _clique(manifest, "leader") - self.assertEqual(leader_clique["labels"]["llm-d.ai/role"], "decode") - self.assertEqual(leader_clique["labels"]["app"], "r") - leader = leader_clique["spec"]["podSpec"] - self.assertEqual([c["name"] for c in leader["containers"]], ["engine", "pd-sidecar"]) - self.assertTrue( - any(v.get("emptyDir", {}).get("medium") == "Memory" for v in leader["volumes"]), - "leader clique missing Memory /dev/shm volume for NIXL", - ) - engine = next(c for c in leader["containers"] if c["name"] == "engine") - self.assertIn("VLLM_NIXL_SIDE_CHANNEL_HOST", [e["name"] for e in engine["env"]]) - - # The worker clique never serves; routing must not touch it at all - - # its labels stay exactly what the Grove backend composed (just the - # queue label), with no role or serving label added. - worker_clique = _clique(manifest, "worker") - worker = worker_clique["spec"]["podSpec"] - self.assertEqual([c["name"] for c in worker["containers"]], ["engine"]) - self.assertEqual(worker_clique["labels"], {_QUEUE_LABEL: _QUEUE}) - - -class TestUnifiedRouting(unittest.TestCase): - """Unified serving (or no serving block) fronts the pods with an - InferencePool + endpoint picker in place of a plain Service, so requests - route by prefix cache and load rather than round-robin - one pod or many. - Mirrors how fn.py composes engines then calls routing.apply.""" - - def _apply(self, copies: int = 1) -> dict[str, k8sobjv1alpha1.Object]: - engine = _standalone_engine(copies=copies) - replica = _replica(engines=[engine]) - composed = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - return routing.apply(composed, replica, _PC) - - def test_fronts_with_pool_and_epp(self) -> None: - out = self._apply() - self.assertIn("inference-pool", out) - self.assertIn("epp", out) - self.assertIn("epp-config", out) - pool = out["inference-pool"].spec.forProvider.manifest - self.assertEqual(pool["kind"], "InferencePool") - self.assertEqual(pool["spec"]["endpointPickerRef"]["name"], "r-epp") - - def test_single_pod_also_pools(self) -> None: - """A single serving pod has nothing to pick between, but still gets the - pool. Always fronting with one avoids swapping a Service for a pool when a - second pod appears - a swap that would drop in-flight requests.""" - for copies in (1, 2): - with self.subTest(copies=copies): - out = self._apply(copies=copies) - self.assertIn("inference-pool", out) - self.assertIn("epp", out) - - def test_fronts_a_leader_worker_set(self) -> None: - """A Standard multi-node engine composes a LeaderWorkerSet, and unified - routing must handle that shape too: it reads the engine args for the KV - block size through _serving_pod_templates, which has to normalize a - LeaderWorkerSet's leaderTemplate alongside a Deployment's pod template - and a Grove PodCliqueSet's leader clique. Regression for a shape - normalization that only knew Deployment and PodCliqueSet and raised - KeyError on a LeaderWorkerSet.""" - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _replica(engines=[engine]) - composed = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - out = routing.apply(composed, replica, _PC) - self.assertIn("inference-pool", out) - self.assertEqual(out["model-serving-main"].spec.forProvider.manifest["kind"], "LeaderWorkerSet") - - def test_pool_selects_pods_by_the_serving_label(self) -> None: - """The pool selects the pods by the serving label they already carry, so - no relabeling is needed.""" - pool = self._apply()["inference-pool"].spec.forProvider.manifest - self.assertEqual(pool["spec"]["selector"]["matchLabels"], {base.LABEL_SERVING: "r"}) - - def test_route_targets_inference_pool(self) -> None: - route = self._apply()[base.ROUTE_KEY].spec.forProvider.manifest - ref = route["spec"]["rules"][0]["backendRefs"][0] - self.assertEqual(ref["kind"], "InferencePool") - self.assertEqual(ref["name"], "r-pool") - - def test_epp_config_is_unified_not_disaggregated(self) -> None: - """The unified picker scores by prefix cache and queue depth in a single - profile, with no prefill/decode split, and still needs the - approx-prefix-cache-producer that feeds the prefix-cache scorer.""" - cfg = self._apply()["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - self.assertIn("prefix-cache-scorer", cfg) - self.assertIn("queue-scorer", cfg) - self.assertIn("approx-prefix-cache-producer", cfg) - self.assertNotIn("prefill", cfg) - self.assertNotIn("decider", cfg) - - def test_epp_image_and_config_group_are_pinned(self) -> None: - """Lock the picker image and the EndpointPickerConfig API group for the - unified path too. A deliberate literal (not routing._EPP_IMAGE) so a wrong - tag/registry or a stale config group is caught in review, not as a - deploy-time crashloop. Unified has no sidecar, so only the EPP is checked. - """ - out = self._apply() - epp = out["epp"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"] - self.assertEqual( - next(c["image"] for c in epp if c["name"] == "epp"), - "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0", + +def _cache_replica( + *, cache: str | None = None, args: list[str] | None = None, command: list[str] | None = None +) -> v1alpha1.ModelReplica: + """A replica with one Standalone engine, referencing cache if one's given.""" + engine = _standalone_engine(args=args or [], command=command) + modelcache = v1alpha1.ModelCacheRef(name=cache) if cache else None + return v1alpha1.ModelReplica( + metadata=metav1.ObjectMeta(namespace="ml-team"), + spec=v1alpha1.SpecModel(clusterName="c", modelCacheRef=modelcache, engines=[engine]), + ) + + +def test_no_cache_no_mounts() -> None: + """A replica with no cache mounts nothing.""" + volumes, mounts = base.cache_mounts(_cache_replica()) + assert (volumes, mounts) == ([], []) + + +def test_cache_adds_volume_and_mount() -> None: + """A replica with a cache mounts the cache's PVC.""" + volumes, mounts = base.cache_mounts(_cache_replica(cache="qwen")) + assert volumes == [{"name": "model-cache", "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}}] + assert mounts == [{"name": "model-cache", "mountPath": "/mnt/models"}] + + +def test_cache_env_points_huggingface_at_the_mount() -> None: + """A replica with a cache points HF_HUB_CACHE at the mount.""" + # The cache is staged in HuggingFace's cache layout, so pointing + # HF_HUB_CACHE at the mount is what lets an engine's own --model= + # resolve against it instead of pulling from HuggingFace (#407). + assert base.cache_env(_cache_replica(cache="qwen")) == [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}] + + +def test_cache_env_empty_without_cache() -> None: + """A replica with no cache gets no cache env.""" + assert base.cache_env(_cache_replica()) == [] + + +def test_cache_env_sets_no_offline_flag() -> None: + """A replica with a cache doesn't set HF_HUB_OFFLINE.""" + # HF_HUB_OFFLINE would break an engine that fetches a *different* repo + # at startup (kimi-k2's separately-gated tokenizer), and resolution + # doesn't need it. + names = {e["name"] for e in base.cache_env(_cache_replica(cache="qwen"))} + assert "HF_HUB_OFFLINE" not in names + + +def _native_cache_replica() -> v1alpha1.ModelReplica: + """A replica with one Standalone engine, referencing the qwen cache.""" + engine = _standalone_engine(args=[]) + return v1alpha1.ModelReplica( + metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), + spec=v1alpha1.SpecModel( + clusterName="cluster-a", + modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), + engines=[engine], + ), + ) + + +def test_native_cache_mounts_pvc_and_sets_cache_env() -> None: + """The native backend mounts a cache and points HF_HUB_CACHE at it, injecting no --model.""" + # A cache contributes a volume, a mount, and the HF_HUB_CACHE that makes + # the engine's own --model= resolve against it. Modelplane injects + # no --model of its own: naming the model is the command's job. + replica = _native_cache_replica() + out = native.NativeBackend().build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Standard") + dep = out["model-serving-main"].spec.forProvider.manifest + pod = dep["spec"]["template"]["spec"] + vol_names = {v["name"] for v in pod["volumes"]} + assert "model-cache" in vol_names + container = pod["containers"][0] + assert {"name": "model-cache", "mountPath": "/mnt/models"} in container["volumeMounts"] + assert {"name": "HF_HUB_CACHE", "value": "/mnt/models"} in container["env"] + assert container["args"] == [] + + +def test_native_cache_user_env_comes_after_cache_env() -> None: + """A Standalone member's own env follows the cache env.""" + # Kubernetes expands $(VAR) left to right, so Modelplane's own entries + # must precede the user's for a user entry to reference them. + replica = _native_cache_replica() + engine = replica.spec.engines[0] + spec = engine.members[0].template.spec + assert spec is not None + spec.containers[0].env = [v1alpha1.EnvItem(name="HF_TOKEN", value="x")] + out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") + container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] + assert container["env"] == [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}, {"name": "HF_TOKEN", "value": "x"}] + + +def _grove_cache_replica( + *, + leader_command: list[str] | None = None, + worker_command: list[str] | None = None, + leader_args: list[str] | None = None, + worker_args: list[str] | None = None, +) -> v1alpha1.ModelReplica: + """A replica with one Leader/Worker engine, referencing the kimi cache.""" + engine = _gang_engine( + leader_command=leader_command, + worker_command=worker_command, + leader_args=leader_args, + worker_args=worker_args, + ) + return v1alpha1.ModelReplica( + metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), + spec=v1alpha1.SpecModel( + clusterName="cluster-a", + modelCacheRef=v1alpha1.ModelCacheRef(name="kimi"), + engines=[engine], + ), + ) + + +def test_grove_cache_both_cliques_mount_cache() -> None: + """The Grove backend mounts a cache on both cliques.""" + replica = _grove_cache_replica(leader_args=[], worker_command=["/bin/sh", "-c", "join"]) + manifest = ( + grove.GroveBackend() + .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] + .spec.forProvider.manifest + ) + for clique_name in ("leader", "worker"): + pod = _clique(manifest, clique_name)["spec"]["podSpec"] + assert "model-cache" in {v["name"] for v in pod["volumes"]} + assert {"name": "model-cache", "mountPath": "/mnt/models"} in pod["containers"][0]["volumeMounts"] + + +def test_grove_cache_sets_cache_env_on_every_clique_and_injects_no_model() -> None: + """The Grove backend points both cliques' HF_HUB_CACHE at a cache, injecting no --model.""" + # A cache gives both cliques HF_HUB_CACHE so their own --model= + # resolves against the mount; Modelplane adds no --model itself. + replica = _grove_cache_replica(leader_args=[], worker_command=["/bin/sh", "-c", "join"]) + manifest = ( + grove.GroveBackend() + .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] + .spec.forProvider.manifest + ) + for clique_name in ("leader", "worker"): + container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] + assert {"name": "HF_HUB_CACHE", "value": "/mnt/models"} in container["env"] + assert "--model=/mnt/models" not in container.get("args", []) + + +def test_grove_cache_command_engine_mounts_cache_without_injecting_model() -> None: + """A Grove member with its own command mounts a cache and keeps its command verbatim.""" + # A member with its own command keeps it verbatim and gets no injected + # --model (it points at the cache with its own flag). + leader_cmd = [ + "/bin/sh", + "-c", + "python3 -m sglang.launch_server --model-path /mnt/models --tp 16", + ] + replica = _grove_cache_replica(leader_command=leader_cmd, worker_command=["/bin/sh", "-c", "join"]) + manifest = ( + grove.GroveBackend() + .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] + .spec.forProvider.manifest + ) + leader = _clique(manifest, "leader")["spec"]["podSpec"]["containers"][0] + assert {"name": "model-cache", "mountPath": "/mnt/models"} in leader["volumeMounts"] + assert leader["command"] == leader_cmd + + +# serving.mode: PrefillDecode routing layers an InferencePool + endpoint +# picker over two engines, role-labels them, and sidecars decode — no unified +# Service. Mirrors how fn.py composes engines then calls routing.apply. + + +def _disaggregated_apply() -> dict[str, k8sobjv1alpha1.Object]: + """Routing for a PrefillDecode replica of two native engines.""" + prefill = _standalone_engine(name="prefill") + prefill.phase = "Prefill" + decode = _standalone_engine(name="decode") + decode.phase = "Decode" + replica = _replica(engines=[prefill, decode]) + replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") + composed = {} + for engine in replica.spec.engines: + composed.update(native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard")) + return routing.apply(composed, replica, _PC) + + +def _serving_pod(out: dict[str, k8sobjv1alpha1.Object], engine_name: str) -> dict: + """The pod template of an engine's Deployment.""" + return out[f"model-serving-{engine_name}"].spec.forProvider.manifest["spec"]["template"] + + +def test_disaggregated_replaces_unified_service_with_pool_and_epp() -> None: + """PrefillDecode routing fronts the engines with an InferencePool and endpoint picker.""" + out = _disaggregated_apply() + assert "inference-pool" in out + assert "epp" in out + assert "epp-config" in out + pool = out["inference-pool"].spec.forProvider.manifest + assert pool["kind"] == "InferencePool" + assert pool["spec"]["endpointPickerRef"]["name"] == "r-epp" + + +def test_disaggregated_injects_nixl_plumbing() -> None: + """Both PrefillDecode engines get the NIXL plumbing the schema can't express.""" + # The plumbing is a Memory /dev/shm and VLLM_NIXL_SIDE_CHANNEL_HOST = pod IP. + out = _disaggregated_apply() + for role in ("prefill", "decode"): + pod = _serving_pod(out, role)["spec"] + assert any(v.get("emptyDir", {}).get("medium") == "Memory" for v in pod["volumes"]), ( + f"{role} missing Memory /dev/shm volume" ) - cfg = out["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - self.assertIn("apiVersion: llm-d.ai/v1alpha1", cfg) - - def test_epp_pod_carries_config_checksum(self) -> None: - """The EPP reads its config once at startup, so a config change must roll - the pod. The pod template carries a sha256 of the rendered config to drive - that rollout.""" - template = self._apply()["epp"].spec.forProvider.manifest["spec"]["template"] - checksum = template["metadata"]["annotations"]["modelplane.ai/epp-config-checksum"] - self.assertEqual(len(checksum), 64) - - -class TestModelExpressEnv(unittest.TestCase): - """On a Dynamo cluster the native (Standalone) and Grove (Leader/Worker) - backends inject the ModelExpress P2P env (MX_SERVER_ADDRESS/MODEL_EXPRESS_URL/ - MX_MODEL_REVISION/MX_P2P_METADATA/POD_*) and the IPC_LOCK security context - into every engine container of a replica that references a cache. The env is - inert unless the engine command opts in with --load-format modelexpress. It's - gated on the cluster's Dynamo stack: on Standard neither backend injects it - (the portable engine command falls back), and the llm-d backend never does. - - HF_HUB_CACHE is deliberately NOT in this set: it's the cache's own env, on - every stack (see base.cache_env), and ModelExpress reads it only as a - fallback for its cache root. Keeping it out here is what makes these - assertions fail if it ever leaks back into modelexpress_env as a duplicate.""" - - _MODELEXPRESS_ENV_NAMES: ClassVar[set[str]] = { - "MX_SERVER_ADDRESS", - "MODEL_EXPRESS_URL", - "MX_MODEL_REVISION", - "MX_P2P_METADATA", - "POD_NAME", - "POD_UID", - "POD_NAMESPACE", + engine = next(c for c in pod["containers"] if c["name"] == "engine") + assert "/dev/shm" in [m["mountPath"] for m in engine["volumeMounts"]] + host = next((e for e in engine["env"] if e["name"] == "VLLM_NIXL_SIDE_CHANNEL_HOST"), None) + assert host is not None, f"{role} missing VLLM_NIXL_SIDE_CHANNEL_HOST" + assert host["valueFrom"]["fieldRef"]["fieldPath"] == "status.podIP" + assert "VLLM_NIXL_SIDE_CHANNEL_PORT" in [e["name"] for e in engine["env"]] + + +def test_disaggregated_epp_config_arms_the_pd_decider() -> None: + """PrefillDecode silently serves decode-only unless the PD decider is armed.""" + # Selective prefix-based-pd-decider needs all of: nonCachedTokens > 0 (0 = + # disabled), the approx-prefix-cache-producer plugin that populates the + # attribute it reads, and that producer pinned to autoTune: false (the + # true default never populates). And it must NOT carry the prepareDataPlugins + # feature gate, which the v0.8.0 EPP image rejects and crashloops on. + cfg = _disaggregated_apply()["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] + assert "prefix-based-pd-decider" in cfg + assert "nonCachedTokens: 16" in cfg + assert "approx-prefix-cache-producer" in cfg + assert "autoTune: false" in cfg + assert "nonCachedTokens: 0" not in cfg + assert "prepareDataPlugins" not in cfg + + +def test_disaggregated_epp_and_sidecar_images_and_config_group_are_pinned() -> None: + """Lock the picker and sidecar images and the EndpointPickerConfig API group.""" + # Nothing else asserts these, so a wrong tag/registry path or a stale config + # group passes CI and only surfaces as an EPP/sidecar crashloop at deploy. + # These are deliberate literals, not routing._* constants: comparing to the + # constant would be tautological (it can't catch a typo in the constant), and + # a literal forces a bump to show up here and be reviewed. + out = _disaggregated_apply() + epp = out["epp"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"] + assert next(c["image"] for c in epp if c["name"] == "epp") == "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0" + sidecar = next(c for c in _serving_pod(out, "decode")["spec"]["containers"] if c["name"] == "pd-sidecar") + assert sidecar["image"] == "ghcr.io/llm-d/llm-d-router-disagg-sidecar:v0.9.0" + cfg = out["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] + assert "apiVersion: llm-d.ai/v1alpha1" in cfg + + +def test_disaggregated_epp_role_watches_inferenceobjectives() -> None: + """The picker watches InferenceObjectives (GIE x-k8s.io group); the Role must allow it.""" + rules = _disaggregated_apply()["epp-role"].spec.forProvider.manifest["rules"] + assert any( + "inference.networking.x-k8s.io" in r["apiGroups"] and "inferenceobjectives" in r["resources"] for r in rules + ), f"EPP Role missing inferenceobjectives watch: {rules}" + + +def test_disaggregated_decode_port_follows_user_arg() -> None: + """The sidecar and the decode container port track the user's --port, not a hardcoded one.""" + prefill = _standalone_engine(name="prefill") + prefill.phase = "Prefill" + decode = _standalone_engine(name="decode", args=["--model=m", "--port=9000"]) + decode.phase = "Decode" + replica = _replica(engines=[prefill, decode]) + replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") + composed = {} + for e in replica.spec.engines: + composed.update(native.NativeBackend().build(replica, e, _PC, base.serving_label(replica), "Standard")) + out = routing.apply(composed, replica, _PC) + containers = _serving_pod(out, "decode")["spec"]["containers"] + engine = next(c for c in containers if c["name"] == "engine") + sidecar = next(c for c in containers if c["name"] == "pd-sidecar") + assert engine["ports"][0]["containerPort"] == 9000 + assert "--vllm-port=9000" in sidecar["args"] + assert sidecar["ports"][0]["containerPort"] == 8000 + + +def test_disaggregated_engines_role_labeled() -> None: + """PrefillDecode routing labels each engine's pods with its role.""" + out = _disaggregated_apply() + assert _serving_pod(out, "prefill")["metadata"]["labels"]["llm-d.ai/role"] == "prefill" + decode_labels = _serving_pod(out, "decode")["metadata"]["labels"] + assert decode_labels["llm-d.ai/role"] == "decode" + assert decode_labels["app"] == "r" + + +def test_disaggregated_decode_gets_sidecar_and_moves_engine_port() -> None: + """The decode engine gets the pd-sidecar on the serving port, and moves to another.""" + out = _disaggregated_apply() + containers = _serving_pod(out, "decode")["spec"]["containers"] + names = [c["name"] for c in containers] + assert names == ["engine", "pd-sidecar"] + engine = next(c for c in containers if c["name"] == "engine") + assert engine["ports"][0]["containerPort"] == 8001 + assert engine["readinessProbe"]["timeoutSeconds"] == 5 + sidecar = next(c for c in containers if c["name"] == "pd-sidecar") + assert sidecar["ports"][0]["containerPort"] == 8000 + assert sidecar["readinessProbe"]["timeoutSeconds"] == 5 + assert "--secure-proxy=false" in sidecar["args"] + + +def test_disaggregated_prefill_has_no_sidecar() -> None: + """The prefill engine gets no sidecar.""" + containers = _serving_pod(_disaggregated_apply(), "prefill")["spec"]["containers"] + assert [c["name"] for c in containers] == ["engine"] + + +def test_disaggregated_route_targets_inference_pool() -> None: + """PrefillDecode routing points the HTTPRoute at the InferencePool, with no request timeout.""" + route = _disaggregated_apply()[base.ROUTE_KEY].spec.forProvider.manifest + rule = route["spec"]["rules"][0] + ref = rule["backendRefs"][0] + assert ref["kind"] == "InferencePool" + assert ref["name"] == "r-pool" + # Disable the request timeout so long token streams aren't severed. + assert rule["timeouts"]["request"] == "0s" + + +def test_disaggregated_selects_engines_by_phase_not_name() -> None: + """Roles come from each engine's phase, not its name.""" + decode = _standalone_engine(name="alpha") + decode.phase = "Decode" + prefill = _standalone_engine(name="beta") + prefill.phase = "Prefill" + replica = _replica(engines=[decode, prefill]) + replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") + composed = {} + for e in replica.spec.engines: + composed.update(native.NativeBackend().build(replica, e, _PC, base.serving_label(replica), "Standard")) + out = routing.apply(composed, replica, _PC) + # alpha is Decode -> sidecar; beta is Prefill -> none, despite their names. + assert [c["name"] for c in _serving_pod(out, "alpha")["spec"]["containers"]] == ["engine", "pd-sidecar"] + assert [c["name"] for c in _serving_pod(out, "beta")["spec"]["containers"]] == ["engine"] + assert _serving_pod(out, "alpha")["metadata"]["labels"]["llm-d.ai/role"] == "decode" + assert _serving_pod(out, "beta")["metadata"]["labels"]["llm-d.ai/role"] == "prefill" + + +def test_disaggregated_decode_can_be_a_grove_gang() -> None: + """PrefillDecode routing decorates a Grove decode gang's leader clique, and leaves its worker alone.""" + # A PrefillDecode engine can itself be a Leader/Worker gang, so routing + # must decorate a Grove PodCliqueSet's leader clique - role label, serving + # label, pd-sidecar, NIXL plumbing - exactly like a Deployment's pod + # template. Exercises the _serving_pod_templates normalization that lets + # one routing layer decorate both workload shapes. + prefill = _standalone_engine(name="prefill") + prefill.phase = "Prefill" + decode = _gang_engine(name="decode", leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + decode.phase = "Decode" + replica = _replica(engines=[prefill, decode]) + replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") + composed = { + **native.NativeBackend().build(replica, prefill, _PC, base.serving_label(replica), "Standard"), + **grove.GroveBackend().build(replica, decode, _PC, base.serving_label(replica), "Dynamo"), } - # What a cache-referencing engine carries on Dynamo: the cache's env plus - # the MX bundle, and nothing else. - _CACHE_ENV_NAME: ClassVar[str] = "HF_HUB_CACHE" - - def _replica(self, *, cache: bool = True, engines: list[v1alpha1.Engine] | None = None) -> v1alpha1.ModelReplica: - engines = engines if engines is not None else [_standalone_engine(args=[])] - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen") if cache else None, - engines=engines, - ), - ) + out = routing.apply(composed, replica, _PC) + + manifest = out["model-serving-decode"].spec.forProvider.manifest + leader_clique = _clique(manifest, "leader") + assert leader_clique["labels"]["llm-d.ai/role"] == "decode" + assert leader_clique["labels"]["app"] == "r" + leader = leader_clique["spec"]["podSpec"] + assert [c["name"] for c in leader["containers"]] == ["engine", "pd-sidecar"] + assert any(v.get("emptyDir", {}).get("medium") == "Memory" for v in leader["volumes"]), ( + "leader clique missing Memory /dev/shm volume for NIXL" + ) + engine = next(c for c in leader["containers"] if c["name"] == "engine") + assert "VLLM_NIXL_SIDE_CHANNEL_HOST" in [e["name"] for e in engine["env"]] + + # The worker clique never serves; routing must not touch it at all - + # its labels stay exactly what the Grove backend composed (just the + # queue label), with no role or serving label added. + worker_clique = _clique(manifest, "worker") + worker = worker_clique["spec"]["podSpec"] + assert [c["name"] for c in worker["containers"]] == ["engine"] + assert worker_clique["labels"] == {_QUEUE_LABEL: _QUEUE} + + +# Unified serving (or no serving block) fronts the pods with an +# InferencePool + endpoint picker in place of a plain Service, so requests +# route by prefix cache and load rather than round-robin - one pod or many. +# Mirrors how fn.py composes engines then calls routing.apply. + + +def _unified_apply(copies: int = 1) -> dict[str, k8sobjv1alpha1.Object]: + """Routing for a Unified replica of one native engine.""" + engine = _standalone_engine(copies=copies) + replica = _replica(engines=[engine]) + composed = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") + return routing.apply(composed, replica, _PC) + + +def test_unified_fronts_with_pool_and_epp() -> None: + """Unified routing fronts the engine with an InferencePool and endpoint picker.""" + out = _unified_apply() + assert "inference-pool" in out + assert "epp" in out + assert "epp-config" in out + pool = out["inference-pool"].spec.forProvider.manifest + assert pool["kind"] == "InferencePool" + assert pool["spec"]["endpointPickerRef"]["name"] == "r-epp" + + +@pytest.mark.parametrize("copies", [1, 2]) +def test_unified_single_pod_also_pools(copies: int) -> None: + """A single serving pod gets a pool too, as several do.""" + # A single serving pod has nothing to pick between, but still gets the + # pool. Always fronting with one avoids swapping a Service for a pool when a + # second pod appears - a swap that would drop in-flight requests. + out = _unified_apply(copies=copies) + assert "inference-pool" in out + assert "epp" in out + + +def test_unified_fronts_a_leader_worker_set() -> None: + """Unified routing fronts a LeaderWorkerSet.""" + # A Standard multi-node engine composes a LeaderWorkerSet, and unified + # routing must handle that shape too: it reads the engine args for the KV + # block size through _serving_pod_templates, which has to normalize a + # LeaderWorkerSet's leaderTemplate alongside a Deployment's pod template + # and a Grove PodCliqueSet's leader clique. Regression for a shape + # normalization that only knew Deployment and PodCliqueSet and raised + # KeyError on a LeaderWorkerSet. + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + replica = _replica(engines=[engine]) + composed = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") + out = routing.apply(composed, replica, _PC) + assert "inference-pool" in out + assert out["model-serving-main"].spec.forProvider.manifest["kind"] == "LeaderWorkerSet" + + +def test_unified_pool_selects_pods_by_the_serving_label() -> None: + """The pool selects the pods by the serving label they already carry, so no relabeling is needed.""" + pool = _unified_apply()["inference-pool"].spec.forProvider.manifest + assert pool["spec"]["selector"]["matchLabels"] == {base.LABEL_SERVING: "r"} + + +def test_unified_route_targets_inference_pool() -> None: + """Unified routing points the HTTPRoute at the InferencePool.""" + route = _unified_apply()[base.ROUTE_KEY].spec.forProvider.manifest + ref = route["spec"]["rules"][0]["backendRefs"][0] + assert ref["kind"] == "InferencePool" + assert ref["name"] == "r-pool" + + +def test_unified_epp_config_is_unified_not_disaggregated() -> None: + """The unified picker scores by prefix cache and queue depth, with no prefill/decode split.""" + # It scores in a single profile, and still needs the + # approx-prefix-cache-producer that feeds the prefix-cache scorer. + cfg = _unified_apply()["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] + assert "prefix-cache-scorer" in cfg + assert "queue-scorer" in cfg + assert "approx-prefix-cache-producer" in cfg + assert "prefill" not in cfg + assert "decider" not in cfg + + +def test_unified_epp_image_and_config_group_are_pinned() -> None: + """Lock the picker image and the EndpointPickerConfig API group for the unified path too.""" + # A deliberate literal (not routing._EPP_IMAGE) so a wrong + # tag/registry or a stale config group is caught in review, not as a + # deploy-time crashloop. Unified has no sidecar, so only the EPP is checked. + out = _unified_apply() + epp = out["epp"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"] + assert next(c["image"] for c in epp if c["name"] == "epp") == "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0" + cfg = out["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] + assert "apiVersion: llm-d.ai/v1alpha1" in cfg + + +def test_unified_epp_pod_carries_config_checksum() -> None: + """The EPP pod template carries a sha256 of its config, so a config change rolls the pod.""" + # The EPP reads its config once at startup, so a config change must roll + # the pod. The pod template carries a sha256 of the rendered config to drive + # that rollout. + template = _unified_apply()["epp"].spec.forProvider.manifest["spec"]["template"] + checksum = template["metadata"]["annotations"]["modelplane.ai/epp-config-checksum"] + assert len(checksum) == 64 + + +# On a Dynamo cluster the native (Standalone) and Grove (Leader/Worker) +# backends inject the ModelExpress P2P env (MX_SERVER_ADDRESS/MODEL_EXPRESS_URL/ +# MX_MODEL_REVISION/MX_P2P_METADATA/POD_*) and the IPC_LOCK security context +# into every engine container of a replica that references a cache. The env is +# inert unless the engine command opts in with --load-format modelexpress. It's +# gated on the cluster's Dynamo stack: on Standard neither backend injects it +# (the portable engine command falls back), and the llm-d backend never does. +# +# HF_HUB_CACHE is deliberately NOT in this set: it's the cache's own env, on +# every stack (see base.cache_env), and ModelExpress reads it only as a +# fallback for its cache root. Keeping it out here is what makes these +# assertions fail if it ever leaks back into modelexpress_env as a duplicate. + +_MODELEXPRESS_ENV_NAMES = { + "MX_SERVER_ADDRESS", + "MODEL_EXPRESS_URL", + "MX_MODEL_REVISION", + "MX_P2P_METADATA", + "POD_NAME", + "POD_UID", + "POD_NAMESPACE", +} +# What a cache-referencing engine carries on Dynamo: the cache's env plus +# the MX bundle, and nothing else. +_CACHE_ENV_NAME = "HF_HUB_CACHE" - def test_grove_gang_gets_modelexpress_env_on_both_cliques(self) -> None: - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = self._replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - # Grove also gets the leader-address alias, unconditional on a cache, - # ahead of the cache env and the ModelExpress bundle. - want_env_names = self._MODELEXPRESS_ENV_NAMES | {base.LEADER_ADDRESS_ENV, self._CACHE_ENV_NAME} - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - env_names = {e["name"] for e in container["env"]} - self.assertEqual(env_names, want_env_names, f"{clique_name}: {env_names}") - self.assertEqual(container["env"][0], base.grove_leader_address_env()) - server_env = next(e for e in container["env"] if e["name"] == "MX_SERVER_ADDRESS") - # The per-cluster shared server's well-known Service, qualified by - # its namespace because the engine runs in its team's namespace. - self.assertEqual(server_env["value"], "modelexpress-server.default.svc:8001") - mxurl_env = next(e for e in container["env"] if e["name"] == "MODEL_EXPRESS_URL") - self.assertEqual(mxurl_env["value"], server_env["value"]) - self.assertEqual(container["securityContext"], {"capabilities": {"add": ["IPC_LOCK"]}}) - - def test_grove_gang_without_cache_gets_no_modelexpress_env(self) -> None: - # No cache means no ModelExpress env or security context, but the - # leader-address alias is unconditional (it doesn't depend on a cache). - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = self._replica(cache=False, engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - self.assertEqual(container["env"], [base.grove_leader_address_env()]) - self.assertNotIn("securityContext", container) - - def test_native_engine_gets_modelexpress_env_on_dynamo(self) -> None: - # A Standalone engine on a Dynamo cluster with a cache is as valid a P2P - # peer set as a gang, so it gets the full ModelExpress env and the - # IPC_LOCK security context on its engine container. - replica = self._replica() - out = native.NativeBackend().build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo") - container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] - env = {e["name"]: e for e in container["env"]} - self.assertEqual(set(env), self._MODELEXPRESS_ENV_NAMES | {self._CACHE_ENV_NAME}) - self.assertEqual(env["MX_SERVER_ADDRESS"]["value"], "modelexpress-server.default.svc:8001") - self.assertEqual(env["MX_P2P_METADATA"]["value"], "1") - self.assertEqual(env["HF_HUB_CACHE"]["value"], "/mnt/models") - # Isolates this cache's P2P source identity, qualified by the - # Modelplane namespace (like cache_pvc_name) so two namespaces' caches - # of the same name can't collide at the cluster's one shared server. - self.assertEqual(env["MX_MODEL_REVISION"]["value"], base.cache_pvc_name("ml-team", "qwen")) - for name, field in ( - ("POD_NAME", "metadata.name"), - ("POD_UID", "metadata.uid"), - ("POD_NAMESPACE", "metadata.namespace"), - ): - self.assertEqual(env[name]["valueFrom"]["fieldRef"]["fieldPath"], field) - self.assertEqual(container["securityContext"], {"capabilities": {"add": ["IPC_LOCK"]}}) - - def test_native_engine_gets_no_modelexpress_env_on_standard(self) -> None: - # The same cached Standalone engine on a Standard cluster gets no - # ModelExpress env and no security context: the portable engine command - # falls back. It keeps the cache's own HF_HUB_CACHE, which is not part - # of the ModelExpress bundle and applies on every stack. - replica = self._replica() - out = native.NativeBackend().build( - replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Standard" - ) - container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] - self.assertEqual(container["env"], [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}]) - self.assertNotIn("securityContext", container) - self.assertEqual(container["args"], []) - - -class TestKvBlockSize(unittest.TestCase): - """The EPP prefix-cache producer's blockSizeTokens is derived best-effort - from the engine flags (#179) so it matches the engine's KV block size.""" - - def test_defaults_to_16_when_absent(self) -> None: - self.assertEqual(routing._kv_block_size([]), 16) - self.assertEqual(routing._kv_block_size(["--model=/mnt/models"]), 16) - - def test_reads_vllm_block_size(self) -> None: - self.assertEqual(routing._kv_block_size(["--block-size", "32"]), 32) - self.assertEqual(routing._kv_block_size(["--model=/m", "--block-size=8"]), 8) - - def test_reads_sglang_page_size(self) -> None: - self.assertEqual(routing._kv_block_size(["--page-size=64"]), 64) - - def test_non_integer_falls_back_to_default(self) -> None: - self.assertEqual(routing._kv_block_size(["--block-size", "auto"]), 16) - - def test_rendered_config_uses_block_size(self) -> None: - cfg = routing._disaggregated_epp_config_yaml(32) - self.assertIn("blockSizeTokens: 32", cfg) - self.assertNotIn("BLOCK_SIZE_TOKENS", cfg) - - -class TestRemoteNamespace(unittest.TestCase): - """The mirrored namespace a replica's objects land in. The expected names are - spelled out, because compose-inference-cluster creates the namespace and - compose-model-route and compose-model-cache land objects in it by the same - derivation, and all four must agree.""" - - def test_remote_namespace(self) -> None: - cases = [ - ("a short namespace keeps its name, prefixed and hashed", "ml-team", "mp-ml-team-51733"), - ( - # 63 is the longest a namespace can be, so mp- plus it can't be - # used as is. It's truncated to leave room for the hash. - "the longest valid namespace still yields a valid one", - "a" * 63, - "mp-" + "a" * 54 + "-38bfb", - ), - ] - for name, namespace, want in cases: - with self.subTest(name): - got = base.remote_namespace(_replica(namespace=namespace)) - self.assertEqual(got, want) - self.assertLessEqual(len(got), 63) + +def _modelexpress_replica(*, cache: bool = True, engines: list[v1alpha1.Engine] | None = None) -> v1alpha1.ModelReplica: + """A replica of engines, referencing the qwen cache unless cache is False.""" + engines = engines if engines is not None else [_standalone_engine(args=[])] + return v1alpha1.ModelReplica( + metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), + spec=v1alpha1.SpecModel( + clusterName="cluster-a", + modelCacheRef=v1alpha1.ModelCacheRef(name="qwen") if cache else None, + engines=engines, + ), + ) + + +def test_grove_gang_gets_modelexpress_env_on_both_cliques() -> None: + """A cached Grove gang on Dynamo gets the ModelExpress env and IPC_LOCK on both cliques.""" + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + replica = _modelexpress_replica(engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + # Grove also gets the leader-address alias, unconditional on a cache, + # ahead of the cache env and the ModelExpress bundle. + want_env_names = _MODELEXPRESS_ENV_NAMES | {base.LEADER_ADDRESS_ENV, _CACHE_ENV_NAME} + for clique_name in ("leader", "worker"): + container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] + env_names = {e["name"] for e in container["env"]} + assert env_names == want_env_names, f"{clique_name}: {env_names}" + assert container["env"][0] == base.grove_leader_address_env() + server_env = next(e for e in container["env"] if e["name"] == "MX_SERVER_ADDRESS") + # The per-cluster shared server's well-known Service, qualified by + # its namespace because the engine runs in its team's namespace. + assert server_env["value"] == "modelexpress-server.default.svc:8001" + mxurl_env = next(e for e in container["env"] if e["name"] == "MODEL_EXPRESS_URL") + assert mxurl_env["value"] == server_env["value"] + assert container["securityContext"] == {"capabilities": {"add": ["IPC_LOCK"]}} + + +def test_grove_gang_without_cache_gets_no_modelexpress_env() -> None: + """A Grove gang with no cache gets only the leader address alias, and no security context.""" + # No cache means no ModelExpress env or security context, but the + # leader-address alias is unconditional (it doesn't depend on a cache). + engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) + replica = _modelexpress_replica(cache=False, engines=[engine]) + out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") + manifest = out["model-serving-main"].spec.forProvider.manifest + for clique_name in ("leader", "worker"): + container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] + assert container["env"] == [base.grove_leader_address_env()] + assert "securityContext" not in container + + +def test_native_engine_gets_modelexpress_env_on_dynamo() -> None: + """A cached Standalone engine on Dynamo gets the ModelExpress env and IPC_LOCK.""" + # A Standalone engine on a Dynamo cluster with a cache is as valid a P2P + # peer set as a gang, so it gets the full ModelExpress env and the + # IPC_LOCK security context on its engine container. + replica = _modelexpress_replica() + out = native.NativeBackend().build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo") + container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] + env = {e["name"]: e for e in container["env"]} + assert set(env) == _MODELEXPRESS_ENV_NAMES | {_CACHE_ENV_NAME} + assert env["MX_SERVER_ADDRESS"]["value"] == "modelexpress-server.default.svc:8001" + assert env["MX_P2P_METADATA"]["value"] == "1" + assert env["HF_HUB_CACHE"]["value"] == "/mnt/models" + # Isolates this cache's P2P source identity, qualified by the + # Modelplane namespace (like cache_pvc_name) so two namespaces' caches + # of the same name can't collide at the cluster's one shared server. + assert env["MX_MODEL_REVISION"]["value"] == base.cache_pvc_name("ml-team", "qwen") + for name, field in ( + ("POD_NAME", "metadata.name"), + ("POD_UID", "metadata.uid"), + ("POD_NAMESPACE", "metadata.namespace"), + ): + assert env[name]["valueFrom"]["fieldRef"]["fieldPath"] == field + assert container["securityContext"] == {"capabilities": {"add": ["IPC_LOCK"]}} + + +def test_native_engine_gets_no_modelexpress_env_on_standard() -> None: + """A cached Standalone engine on Standard gets only the cache env, and no security context.""" + # The same cached Standalone engine on a Standard cluster gets no + # ModelExpress env and no security context: the portable engine command + # falls back. It keeps the cache's own HF_HUB_CACHE, which is not part + # of the ModelExpress bundle and applies on every stack. + replica = _modelexpress_replica() + out = native.NativeBackend().build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Standard") + container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] + assert container["env"] == [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}] + assert "securityContext" not in container + assert container["args"] == [] + + +# The EPP prefix-cache producer's blockSizeTokens is derived best-effort +# from the engine flags (#179) so it matches the engine's KV block size. + + +def test_kv_block_size_defaults_to_16_when_absent() -> None: + """The KV block size defaults to 16 when no flag sets it.""" + assert routing._kv_block_size([]) == 16 + assert routing._kv_block_size(["--model=/mnt/models"]) == 16 + + +def test_kv_block_size_reads_vllm_block_size() -> None: + """The KV block size comes from vLLM's --block-size.""" + assert routing._kv_block_size(["--block-size", "32"]) == 32 + assert routing._kv_block_size(["--model=/m", "--block-size=8"]) == 8 + + +def test_kv_block_size_reads_sglang_page_size() -> None: + """The KV block size comes from SGLang's --page-size.""" + assert routing._kv_block_size(["--page-size=64"]) == 64 + + +def test_kv_block_size_non_integer_falls_back_to_default() -> None: + """A non-integer block size falls back to 16.""" + assert routing._kv_block_size(["--block-size", "auto"]) == 16 + + +def test_kv_block_size_rendered_config_uses_block_size() -> None: + """The rendered EPP config carries the block size in place of its placeholder.""" + cfg = routing._disaggregated_epp_config_yaml(32) + assert "blockSizeTokens: 32" in cfg + assert "BLOCK_SIZE_TOKENS" not in cfg + + +# The mirrored namespace a replica's objects land in. The expected names are +# spelled out, because compose-inference-cluster creates the namespace and +# compose-model-route and compose-model-cache land objects in it by the same +# derivation, and all four must agree. + +REMOTE_NAMESPACE_CASES = [ + pytest.param("ml-team", "mp-ml-team-51733", id="a short namespace keeps its name, prefixed and hashed"), + pytest.param( + # 63 is the longest a namespace can be, so mp- plus it can't be + # used as is. It's truncated to leave room for the hash. + "a" * 63, + "mp-" + "a" * 54 + "-38bfb", + id="the longest valid namespace still yields a valid one", + ), +] + + +@pytest.mark.parametrize(("namespace", "want"), REMOTE_NAMESPACE_CASES) +def test_remote_namespace(namespace: str, want: str) -> None: + """A replica's objects land in a namespace mirroring its own.""" + got = base.remote_namespace(_replica(namespace=namespace)) + assert got == want + assert len(got) <= 63 diff --git a/functions/compose-model-replica/tests/test_fn.py b/functions/compose-model-replica/tests/test_fn.py index 74ab85151..4d2aedd80 100644 --- a/functions/compose-model-replica/tests/test_fn.py +++ b/functions/compose-model-replica/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-model-replica function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.modelreplica import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -29,6 +31,21 @@ # A GPU device request CEL selector, as compose-model-deployment stamps it. _GPU_CEL = 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' +# Unified routing fronts the serving pods with an InferencePool + endpoint +# picker; their manifests are asserted in detail in test_backends. Here we +# only check the function wired the whole set in (and dropped the plain +# Service), then drop their manifests so the golden covers the dispatch, +# wiring and readiness the function itself owns. +_ROUTING_KEYS = { + "inference-pool", + "epp", + "epp-config", + "epp-role", + "epp-rolebinding", + "epp-serviceaccount", + "epp-service", +} + @dataclasses.dataclass class Case: @@ -39,10 +56,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - def _observed_object(*, ready: bool) -> fnv1.Resource: """A composed provider-kubernetes Object as observed back, with the Ready condition its readiness policy derives.""" @@ -66,529 +79,512 @@ def _observed_object(*, ready: bool) -> fnv1.Resource: ) -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function dispatches to a backend to compose serving resources on a remote cluster.""" - - xr = v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta( - name="test-replica", - namespace="ml-team", - labels={ - "modelplane.ai/deployment": "my-deployment", - "modelplane.ai/cluster": "cluster-a", - }, - ), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - engines=[ - v1alpha1.Engine( - name="main", - copies=1, - members=[ - v1alpha1.Member( - role="Standalone", - nodePoolName="frontier", - deviceRequests=[ - v1alpha1.DeviceRequest( - name="gpu", - deviceClassName="gpu.nvidia.com", - count=1, - selectors=[v1alpha1.Selector(cel=_GPU_CEL)], - ), - ], - template=v1alpha1.Template( - spec=v1alpha1.Spec( - containers=[ - v1alpha1.Container( - name="engine", - image="vllm/vllm-openai:latest", - args=["--model=Qwen/Qwen3-0.6B"], - ), - ], - ), +def _compose_cases() -> list[Case]: + """The compose cases. Later cases are built from earlier ones.""" + xr = v1alpha1.ModelReplica( + metadata=metav1.ObjectMeta( + name="test-replica", + namespace="ml-team", + labels={ + "modelplane.ai/deployment": "my-deployment", + "modelplane.ai/cluster": "cluster-a", + }, + ), + spec=v1alpha1.SpecModel( + clusterName="cluster-a", + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[v1alpha1.Selector(cel=_GPU_CEL)], + ), + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ), + ], ), ), - ], - ), - ], - ), - ).model_dump(exclude_none=True, mode="json") + ), + ], + ), + ], + ), + ).model_dump(exclude_none=True, mode="json") - cluster_requirement = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceCluster", - match_name="cluster-a", - ) + cluster_requirement = fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_name="cluster-a", + ) - # Case 1: cluster resolved with providerConfigRef — composes native - # Deployment. First reconcile: none of the composed resources are in - # observed yet, so none are marked ready (the function only asserts - # readiness for a resource it can see in observed state). - req1 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), - ), - ) - req1.required_resources["cluster"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceCluster", - "metadata": {"name": "cluster-a"}, - "spec": { - "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k"}}}, - }, - "status": { - "providerConfigRef": {"name": "cluster-a-pc"}, - "gateway": {"address": "10.0.0.1"}, - }, - } - ) + # Case 1: cluster resolved with providerConfigRef — composes native + # Deployment. First reconcile: none of the composed resources are in + # observed yet, so none are marked ready (the function only asserts + # readiness for a resource it can see in observed state). + req1 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), + ), + ) + req1.required_resources["cluster"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "cluster-a"}, + "spec": { + "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k"}}}, + }, + "status": { + "providerConfigRef": {"name": "cluster-a-pc"}, + "gateway": {"address": "10.0.0.1"}, + }, + } ) ) + ) - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - resources={ - "model-serving-main": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": { - "providerConfigRef": { - "kind": "ClusterProviderConfig", - "name": "cluster-a-pc", - }, - "readiness": { - "policy": "DeriveFromCelQuery", - "celQuery": ( - "has(object.status.conditions) && " - "object.status.conditions.exists(" - 'c, c.type == "Available" && c.status == "True")' - ), - }, - "forProvider": { - "manifest": { - "apiVersion": "apps/v1", - "kind": "Deployment", - "metadata": { - "name": resource.child_name("test-replica", "main"), - "namespace": "mp-ml-team-51733", + want1 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + resources={ + "model-serving-main": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "cluster-a-pc", + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status.conditions) && " + "object.status.conditions.exists(" + 'c, c.type == "Available" && c.status == "True")' + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": { + "name": resource.child_name("test-replica", "main"), + "namespace": "mp-ml-team-51733", + }, + "spec": { + "replicas": 1, + "selector": { + "matchLabels": { + "modelplane.ai/workload": resource.child_name( + "test-replica", "main" + ), + }, }, - "spec": { - "replicas": 1, - "selector": { - "matchLabels": { + "template": { + "metadata": { + "labels": { + "modelplane.ai/serving": "test-replica", "modelplane.ai/workload": resource.child_name( "test-replica", "main" ), }, }, - "template": { - "metadata": { - "labels": { - "modelplane.ai/serving": "test-replica", - "modelplane.ai/workload": resource.child_name( - "test-replica", "main" + "spec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "resources": {"claims": [{"name": "devices"}]}, + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + ], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + }, + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + ], + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": resource.child_name( + "test-replica", "main", "standalone", "devices" ), }, - }, - "spec": { - "containers": [ - { - "name": "engine", - "image": "vllm/vllm-openai:latest", - "args": ["--model=Qwen/Qwen3-0.6B"], - "ports": [{"containerPort": 8000}], - "resources": {"claims": [{"name": "devices"}]}, - "volumeMounts": [ - {"name": "dshm", "mountPath": "/dev/shm"}, - ], - "readinessProbe": { - "httpGet": {"path": "/health", "port": 8000}, - "initialDelaySeconds": 30, - "periodSeconds": 10, - "timeoutSeconds": 5, - }, - }, - ], - "volumes": [ - {"name": "dshm", "emptyDir": {"medium": "Memory"}}, - ], - "nodeSelector": {"modelplane.ai/pool": "frontier"}, - "resourceClaims": [ - { - "name": "devices", - "resourceClaimTemplateName": resource.child_name( - "test-replica", "main", "standalone", "devices" - ), - }, - ], - "tolerations": [ - { - "key": "nvidia.com/gpu", - "operator": "Exists", - "effect": "NoSchedule", - }, - ], - }, + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + }, + ], }, }, }, }, }, - } - ), + }, + } ), - "model-route": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": { - "providerConfigRef": { - "kind": "ClusterProviderConfig", - "name": "cluster-a-pc", - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "gateway.networking.k8s.io/v1", - "kind": "HTTPRoute", - "metadata": { - "name": "test-replica", - "namespace": "mp-ml-team-51733", - }, - "spec": { - "parentRefs": [ - { - "name": "cluster-gateway", - "namespace": "modelplane-system", - }, - ], - "rules": [ - { - "matches": [ - { - "path": { - "type": "PathPrefix", - "value": "/ml-team/test-replica/", - }, + ), + "model-route": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "cluster-a-pc", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "HTTPRoute", + "metadata": { + "name": "test-replica", + "namespace": "mp-ml-team-51733", + }, + "spec": { + "parentRefs": [ + { + "name": "cluster-gateway", + "namespace": "modelplane-system", + }, + ], + "rules": [ + { + "matches": [ + { + "path": { + "type": "PathPrefix", + "value": "/ml-team/test-replica/", }, - ], - "timeouts": {"request": "0s"}, - "filters": [ - { - "type": "URLRewrite", - "urlRewrite": { - "path": { - "type": "ReplacePrefixMatch", - "replacePrefixMatch": "/", - }, + }, + ], + "timeouts": {"request": "0s"}, + "filters": [ + { + "type": "URLRewrite", + "urlRewrite": { + "path": { + "type": "ReplacePrefixMatch", + "replacePrefixMatch": "/", }, }, - ], - "backendRefs": [ - { - "group": "inference.networking.k8s.io", - "kind": "InferencePool", - "name": "test-replica-pool", - }, - ], - }, - ], - }, + }, + ], + "backendRefs": [ + { + "group": "inference.networking.k8s.io", + "kind": "InferencePool", + "name": "test-replica-pool", + }, + ], + }, + ], }, }, }, - } - ), + }, + } ), - "resource-claim-main-standalone": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": { - "providerConfigRef": { - "kind": "ClusterProviderConfig", - "name": "cluster-a-pc", - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "resource.k8s.io/v1", - "kind": "ResourceClaimTemplate", - "metadata": { - "name": resource.child_name( - "test-replica", "main", "standalone", "devices" - ), - "namespace": "mp-ml-team-51733", - }, + ), + "resource-claim-main-standalone": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "cluster-a-pc", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": { + "name": resource.child_name( + "test-replica", "main", "standalone", "devices" + ), + "namespace": "mp-ml-team-51733", + }, + "spec": { "spec": { - "spec": { - "devices": { - "requests": [ - { - "name": "gpu", - "exactly": { - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "selectors": [ - {"cel": {"expression": _GPU_CEL}}, - ], - }, + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + {"cel": {"expression": _GPU_CEL}}, + ], }, - ], - }, + }, + ], }, }, }, }, }, - } - ), + }, + } ), - }, - ), - conditions=[ - fnv1.Condition( - type="ModelAccepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Deploying", - ), - fnv1.Condition( - type="ModelReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForModel", ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Composing vllm/vllm-openai:latest on cluster-a", - ), - ], - context=structpb.Struct(), - ) - want1.requirements.resources["cluster"].CopyFrom(cluster_requirement) - - # Case 2: cluster not resolved — early return with conditions. - req2 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelAccepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Deploying", ), - ) - - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - # Nothing is composed while waiting, so the XR is marked not ready - # rather than left to aggregate to trivially ready. - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - conditions=[ - fnv1.Condition( - type="ModelAccepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - fnv1.Condition( - type="ModelReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForModel", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for cluster to be resolved", - ), - ], - context=structpb.Struct(), - ) - want2.requirements.resources["cluster"].CopyFrom(cluster_requirement) - - # Case 3: cluster resolved but no providerConfigRef — early return. - req3 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), + fnv1.Condition( + type="ModelReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForModel", ), - ) - req3.required_resources["cluster"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceCluster", - "metadata": {"name": "cluster-a"}, - "spec": { - "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k"}}}, - }, - } - ) - ) - ) + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Composing vllm/vllm-openai:latest on cluster-a", + ), + ], + context=structpb.Struct(), + ) + want1.requirements.resources["cluster"].CopyFrom(cluster_requirement) - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - conditions=[ - fnv1.Condition( - type="ModelAccepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - fnv1.Condition( - type="ModelReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForModel", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for cluster providerConfigRef", - ), - ], - context=structpb.Struct(), - ) - want3.requirements.resources["cluster"].CopyFrom(cluster_requirement) + # Case 2: cluster not resolved — early return with conditions. + req2 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), + ), + ) - # Unified routing fronts the serving pods with an InferencePool + endpoint - # picker; their manifests are asserted in detail in test_backends. Here we - # only check the function wired the whole set in (and dropped the plain - # Service), then drop their manifests so the golden covers the dispatch, - # wiring and readiness the function itself owns. - routing_keys = { - "inference-pool", - "epp", - "epp-config", - "epp-role", - "epp-rolebinding", - "epp-serviceaccount", - "epp-service", - } - for key in routing_keys: - want1.desired.resources[key].CopyFrom(fnv1.Resource()) + want2 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + # Nothing is composed while waiting, so the XR is marked not ready + # rather than left to aggregate to trivially ready. + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + conditions=[ + fnv1.Condition( + type="ModelAccepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + fnv1.Condition( + type="ModelReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForModel", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for cluster to be resolved", + ), + ], + context=structpb.Struct(), + ) + want2.requirements.resources["cluster"].CopyFrom(cluster_requirement) - # Case 4: the resources from case 1 now exist in observed, and the - # workload Object reports Available (so its derived Ready is True). The - # function marks each observed resource ready once its Object reports - # Ready: the workload because it's serving and the rest because existing - # is being ready for them. Built from case 1, mutating only what the - # observed-ready transition changes: the three ready flags, the - # acceptance/readiness conditions, and the dropped first-reconcile event. - req4 = fnv1.RunFunctionRequest() - req4.CopyFrom(req1) - # The workload Object as provider-kubernetes observes it back: applied - # (atProvider.manifest populated) and Available (its derived Ready=True). - req4.observed.resources["model-serving-main"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": {"forProvider": {"manifest": {"kind": "Deployment"}}}, - "status": { - "atProvider": {"manifest": {"kind": "Deployment"}}, - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2025-01-01T00:00:00Z", - }, - ], - }, - } - ), + # Case 3: cluster resolved but no providerConfigRef — early return. + req3 = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), + ), + ) + req3.required_resources["cluster"].items.append( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "cluster-a"}, + "spec": { + "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k"}}}, + }, + } ) ) - # The other two have no runtime readiness to wait on, so under their - # SuccessfulCreate policy provider-kubernetes reports them Ready once - # applied. (The InferencePool + endpoint picker resources aren't - # observed here, so they stay unready.) - for key in ("model-route", "resource-claim-main-standalone"): - req4.observed.resources[key].CopyFrom(_observed_object(ready=True)) + ) - want4 = fnv1.RunFunctionResponse() - want4.CopyFrom(want1) - for key in ("model-serving-main", "model-route", "resource-claim-main-standalone"): - want4.desired.resources[key].ready = fnv1.READY_TRUE - del want4.conditions[:] - want4.conditions.extend( - [ - fnv1.Condition(type="ModelAccepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Accepted"), - fnv1.Condition(type="ModelReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Serving"), - ] + want3 = fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + conditions=[ + fnv1.Condition( + type="ModelAccepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + fnv1.Condition( + type="ModelReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForModel", + ), + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for cluster providerConfigRef", + ), + ], + context=structpb.Struct(), + ) + want3.requirements.resources["cluster"].CopyFrom(cluster_requirement) + + # The routing objects' manifests are dropped from the golden (see + # _ROUTING_KEYS). + for key in _ROUTING_KEYS: + want1.desired.resources[key].CopyFrom(fnv1.Resource()) + + # Case 4: the resources from case 1 now exist in observed, and the + # workload Object reports Available (so its derived Ready is True). The + # function marks each observed resource ready once its Object reports + # Ready: the workload because it's serving and the rest because existing + # is being ready for them. Built from case 1, mutating only what the + # observed-ready transition changes: the three ready flags, the + # acceptance/readiness conditions, and the dropped first-reconcile event. + req4 = fnv1.RunFunctionRequest() + req4.CopyFrom(req1) + # The workload Object as provider-kubernetes observes it back: applied + # (atProvider.manifest populated) and Available (its derived Ready=True). + req4.observed.resources["model-serving-main"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": {"forProvider": {"manifest": {"kind": "Deployment"}}}, + "status": { + "atProvider": {"manifest": {"kind": "Deployment"}}, + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2025-01-01T00:00:00Z", + }, + ], + }, + } + ), ) - # The "Composing ..." event fires only the first reconcile (model-serving - # not yet observed), so it's gone now. - del want4.results[:] - - # Case 5: everything from case 4 plus the routing objects is observed, - # but the endpoint picker's Service failed to apply, say because its - # name was invalid, so its Object isn't Ready. Being observed isn't - # being applied, so it stays unready and holds the replica unready with - # it. Built from case 4, mutating only the routing objects' ready flags. - req5 = fnv1.RunFunctionRequest() - req5.CopyFrom(req4) - for key in routing_keys: - req5.observed.resources[key].CopyFrom(_observed_object(ready=key != "epp-service")) - - want5 = fnv1.RunFunctionResponse() - want5.CopyFrom(want4) - for key in routing_keys - {"epp-service"}: - want5.desired.resources[key].ready = fnv1.READY_TRUE - - # Case 6: as case 5, but everything applied and the endpoint picker's - # Deployment isn't Available yet, so its Object's CEL-derived Ready is - # False. The gateway fails closed without a picker, so the replica stays - # unready with it. - req6 = fnv1.RunFunctionRequest() - req6.CopyFrom(req4) - for key in routing_keys: - req6.observed.resources[key].CopyFrom(_observed_object(ready=key != "epp")) - - want6 = fnv1.RunFunctionResponse() - want6.CopyFrom(want4) - for key in routing_keys - {"epp"}: - want6.desired.resources[key].ready = fnv1.READY_TRUE - - cases = [ - Case(name="cluster ready composes native Deployment", req=req1, want=want1), - Case(name="cluster not resolved returns waiting conditions", req=req2, want=want2), - Case(name="cluster without providerConfigRef returns waiting conditions", req=req3, want=want3), - Case(name="observed resources are marked ready", req=req4, want=want4), - Case(name="an object that failed to apply stays unready", req=req5, want=want5), - Case(name="an unavailable endpoint picker stays unready", req=req6, want=want6), + ) + # The other two have no runtime readiness to wait on, so under their + # SuccessfulCreate policy provider-kubernetes reports them Ready once + # applied. (The InferencePool + endpoint picker resources aren't + # observed here, so they stay unready.) + for key in ("model-route", "resource-claim-main-standalone"): + req4.observed.resources[key].CopyFrom(_observed_object(ready=True)) + + want4 = fnv1.RunFunctionResponse() + want4.CopyFrom(want1) + for key in ("model-serving-main", "model-route", "resource-claim-main-standalone"): + want4.desired.resources[key].ready = fnv1.READY_TRUE + del want4.conditions[:] + want4.conditions.extend( + [ + fnv1.Condition(type="ModelAccepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Accepted"), + fnv1.Condition(type="ModelReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Serving"), ] - - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - got_dict = json_format.MessageToDict(got) - resources = got_dict.get("desired", {}).get("resources", {}) - if "model-serving-main" in resources: - self.assertLessEqual(routing_keys, set(resources)) - self.assertNotIn("model-service", resources) - # The routing objects land in the mirrored namespace too, - # before they're dropped from the golden below. - for key in routing_keys: - manifest = resources[key]["resource"]["spec"]["forProvider"]["manifest"] - self.assertEqual(manifest["metadata"]["namespace"], "mp-ml-team-51733", key) - del resources[key]["resource"] - self.assertEqual( - json_format.MessageToDict(case.want), - got_dict, - "-want, +got", - ) + ) + # The "Composing ..." event fires only the first reconcile (model-serving + # not yet observed), so it's gone now. + del want4.results[:] + + # Case 5: everything from case 4 plus the routing objects is observed, + # but the endpoint picker's Service failed to apply, say because its + # name was invalid, so its Object isn't Ready. Being observed isn't + # being applied, so it stays unready and holds the replica unready with + # it. Built from case 4, mutating only the routing objects' ready flags. + req5 = fnv1.RunFunctionRequest() + req5.CopyFrom(req4) + for key in _ROUTING_KEYS: + req5.observed.resources[key].CopyFrom(_observed_object(ready=key != "epp-service")) + + want5 = fnv1.RunFunctionResponse() + want5.CopyFrom(want4) + for key in _ROUTING_KEYS - {"epp-service"}: + want5.desired.resources[key].ready = fnv1.READY_TRUE + + # Case 6: as case 5, but everything applied and the endpoint picker's + # Deployment isn't Available yet, so its Object's CEL-derived Ready is + # False. The gateway fails closed without a picker, so the replica stays + # unready with it. + req6 = fnv1.RunFunctionRequest() + req6.CopyFrom(req4) + for key in _ROUTING_KEYS: + req6.observed.resources[key].CopyFrom(_observed_object(ready=key != "epp")) + + want6 = fnv1.RunFunctionResponse() + want6.CopyFrom(want4) + for key in _ROUTING_KEYS - {"epp"}: + want6.desired.resources[key].ready = fnv1.READY_TRUE + + return [ + Case(name="cluster ready composes native Deployment", req=req1, want=want1), + Case(name="cluster not resolved returns waiting conditions", req=req2, want=want2), + Case(name="cluster without providerConfigRef returns waiting conditions", req=req3, want=want3), + Case(name="observed resources are marked ready", req=req4, want=want4), + Case(name="an object that failed to apply stays unready", req=req5, want=want5), + Case(name="an unavailable endpoint picker stays unready", req=req6, want=want6), + ] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """The function dispatches to a backend to compose serving resources on a remote cluster.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + got_dict = _to_dict(got) + resources = got_dict.get("desired", {}).get("resources", {}) + if "model-serving-main" in resources: + assert set(resources) >= _ROUTING_KEYS + assert "model-service" not in resources + # The routing objects land in the mirrored namespace too, + # before they're dropped from the golden below. + for key in _ROUTING_KEYS: + manifest = resources[key]["resource"]["spec"]["forProvider"]["manifest"] + assert manifest["metadata"]["namespace"] == "mp-ml-team-51733", key + del resources[key]["resource"] + assert got_dict == _to_dict(case.want) diff --git a/functions/compose-model-route/tests/__init__.py b/functions/compose-model-route/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-model-route/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-model-route/tests/test_fn.py b/functions/compose-model-route/tests/test_fn.py index 0a8755aa2..c19c64d14 100644 --- a/functions/compose-model-route/tests/test_fn.py +++ b/functions/compose-model-route/tests/test_fn.py @@ -14,15 +14,17 @@ """Tests for the compose-model-route function.""" +import asyncio import base64 import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 from models.ai.modelplane.inferencegateway import v1alpha1 as igv1alpha1 @@ -248,374 +250,335 @@ def _manifest(rsp: fnv1.RunFunctionResponse, key: str) -> dict: return resource.struct_to_dict(rsp.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) -class TestGates(unittest.IsolatedAsyncioTestCase): - """Passes where a route can't be composed compose nothing and say why. - Asserting the whole response proves nothing is composed against a cluster the - route can't yet reach, rather than a subset being applied.""" - - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_gates(self) -> None: - composed = _endpoint("self", origin="https://gw-eu.example.com", composed=True) - cases = [ - Case( - name="the gateway's client PKI hasn't issued, so nothing can name its certificate", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) - ), - required_resources=_required( - gateway=[_gateway(client_ca=None)], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [composed]}, - ), +# Passes where a route can't be composed compose nothing and say why. Asserting +# the whole response proves nothing is composed against a cluster the route +# can't yet reach, rather than a subset being applied. +def _gates_cases() -> list[Case]: + composed = _endpoint("self", origin="https://gw-eu.example.com", composed=True) + return [ + Case( + name="the gateway's client PKI hasn't issued, so nothing can name its certificate", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) ), - want=_not_ready( - {"model": _MODEL, "endpoints": {"total": 0, "ready": 0}}, - fn.CONDITION_REASON_WAITING_FOR_GATEWAY, - "InferenceGateway eu has not published its client CA", - _requirements([_entry("d")]), + required_resources=_required( + gateway=[_gateway(client_ca=None)], + clusters=[_cluster("gw-eu")], + **{"endpoints-d": [composed]}, ), ), - Case( - name="no selected endpoint is ready", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) - ), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", ready=False)]}, - ), + want=_not_ready( + {"model": _MODEL, "endpoints": {"total": 0, "ready": 0}}, + fn.CONDITION_REASON_WAITING_FOR_GATEWAY, + "InferenceGateway eu has not published its client CA", + _requirements([_entry("d")]), + ), + ), + Case( + name="no selected endpoint is ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) ), - want=_not_ready( - { - "model": _MODEL, - "address": "203.0.113.1", - "endpoints": {"total": 1, "ready": 0}, - }, - fn.CONDITION_REASON_NO_ENDPOINTS, - "None of the 1 selected ModelEndpoints is ready to carry traffic", - _requirements([_entry("d")]), + required_resources=_required( + gateway=[_gateway()], + clusters=[_cluster("gw-eu")], + **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", ready=False)]}, ), ), - Case( - name="a composed endpoint whose cluster withdrew its CA is dropped", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) - ), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu", ca=None)], - **{"endpoints-d": [composed]}, - ), + want=_not_ready( + { + "model": _MODEL, + "address": "203.0.113.1", + "endpoints": {"total": 1, "ready": 0}, + }, + fn.CONDITION_REASON_NO_ENDPOINTS, + "None of the 1 selected ModelEndpoints is ready to carry traffic", + _requirements([_entry("d")]), + ), + ), + Case( + name="a composed endpoint whose cluster withdrew its CA is dropped", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) ), - want=_not_ready( - { - "model": _MODEL, - "address": "203.0.113.1", - "endpoints": {"total": 1, "ready": 0}, - }, - fn.CONDITION_REASON_NO_ENDPOINTS, - "None of the 1 selected ModelEndpoints is ready to carry traffic", - _requirements([_entry("d")]), - warning="Endpoints left out of the route, their cluster has published no gateway CA: self", + required_resources=_required( + gateway=[_gateway()], + clusters=[_cluster("gw-eu", ca=None)], + **{"endpoints-d": [composed]}, ), ), - Case( - name="a credential Secret missing its key drops the endpoint", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("a")]))) - ), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-a": [_endpoint("wrongkey", origin="https://a.example.com", credential="k")], - "credential-wrongkey": [_secret("k", {"token": "sk-1"})], - }, - ), + want=_not_ready( + { + "model": _MODEL, + "address": "203.0.113.1", + "endpoints": {"total": 1, "ready": 0}, + }, + fn.CONDITION_REASON_NO_ENDPOINTS, + "None of the 1 selected ModelEndpoints is ready to carry traffic", + _requirements([_entry("d")]), + warning="Endpoints left out of the route, their cluster has published no gateway CA: self", + ), + ), + Case( + name="a credential Secret missing its key drops the endpoint", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("a")]))) ), - want=_not_ready( - { - "model": _MODEL, - "address": "203.0.113.1", - "endpoints": {"total": 1, "ready": 0}, + required_resources=_required( + gateway=[_gateway()], + clusters=[_cluster("gw-eu")], + **{ + "endpoints-a": [_endpoint("wrongkey", origin="https://a.example.com", credential="k")], + "credential-wrongkey": [_secret("k", {"token": "sk-1"})], }, - fn.CONDITION_REASON_NO_ENDPOINTS, - "None of the 1 selected ModelEndpoints is ready to carry traffic", - _requirements([_entry("a")], credentials={"wrongkey": "k"}), - warning=( - "Endpoints left out of the route, their credential Secret missing or missing its key: wrongkey" - ), ), ), - ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) - - -class TestCompose(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """A composed self-hosted endpoint at priority 0 and a third-party - provider at priority 1: backends, credential, cluster CA and route.""" - entries = [_entry("d", priority=0), _entry("together", priority=1)] - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-d": [ - _endpoint("self", origin="https://gw-eu.example.com", model="d", composed=True), - ], - "endpoints-together": [ - _endpoint( - "together", - origin="https://api.together.xyz", - model="Qwen/Qwen2.5", - credential="together-key", - ), - ], - "credential-together": [_secret("together-key", {"apiKey": "sk-tog"})], + want=_not_ready( + { + "model": _MODEL, + "address": "203.0.113.1", + "endpoints": {"total": 1, "ready": 0}, }, + fn.CONDITION_REASON_NO_ENDPOINTS, + "None of the 1 selected ModelEndpoints is ready to carry traffic", + _requirements([_entry("a")], credentials={"wrongkey": "k"}), + warning=( + "Endpoints left out of the route, their credential Secret missing or missing its key: wrongkey" + ), ), - ) - got = await self.runner.RunFunction(req, None) - - # The exact set, so an unexpected extra object fails the test. The - # mirrored namespace isn't here: compose-inference-cluster composes it. - self.assertEqual( - set(got.desired.resources), - { - "client-certificate", - "backend-self", - "aibackend-self", - "backend-together", - "aibackend-together", - "credential-together", - "credpolicy-together", - "cluster-ca-gw-eu", - "route", + ), + ] + + +@pytest.mark.parametrize("case", _gates_cases(), ids=lambda case: case.name) +def test_gates(case: Case) -> None: + """A pass where a route can't be composed composes nothing and says why.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) + + +def test_compose() -> None: + """A composed endpoint and a third-party provider get backends, a credential, a cluster CA and a route.""" + # A composed self-hosted endpoint at priority 0 and a third-party provider + # at priority 1. + entries = [_entry("d", priority=0), _entry("together", priority=1)] + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), + required_resources=_required( + gateway=[_gateway()], + clusters=[_cluster("gw-eu")], + **{ + "endpoints-d": [ + _endpoint("self", origin="https://gw-eu.example.com", model="d", composed=True), + ], + "endpoints-together": [ + _endpoint( + "together", + origin="https://api.together.xyz", + model="Qwen/Qwen2.5", + credential="together-key", + ), + ], + "credential-together": [_secret("together-key", {"apiKey": "sk-tog"})], }, - ) + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + # The exact set, so an unexpected extra object fails the test. The + # mirrored namespace isn't here: compose-inference-cluster composes it. + assert set(got.desired.resources) == { + "client-certificate", + "backend-self", + "aibackend-self", + "backend-together", + "aibackend-together", + "credential-together", + "credpolicy-together", + "cluster-ca-gw-eu", + "route", + } - route = _manifest(got, "route") - # The timeouts are the ModelRoute's, which _route_xr sets to values - # other than the ModelService's defaults. - self.assertEqual( - route["spec"]["rules"], - [ + route = _manifest(got, "route") + # The timeouts are the ModelRoute's, which _route_xr sets to values + # other than the ModelService's defaults. + assert route["spec"]["rules"] == [ + { + "matches": [{"headers": [{"type": "Exact", "name": "x-ai-eg-model", "value": _MODEL}]}], + "backendRefs": [ + {"name": _be("self"), "weight": 1, "priority": 0, "modelNameOverride": "d"}, { - "matches": [{"headers": [{"type": "Exact", "name": "x-ai-eg-model", "value": _MODEL}]}], - "backendRefs": [ - {"name": _be("self"), "weight": 1, "priority": 0, "modelNameOverride": "d"}, - { - "name": _be("together"), - "weight": 1, - "priority": 1, - "modelNameOverride": "Qwen/Qwen2.5", - }, - ], - "timeouts": {"request": "600s"}, - "streamIdleTimeout": "0s", - "modelsOwnedBy": _NS, - } - ], - ) - # Declaring the token costs is what makes the ext-proc ask a backend for - # usage on a streamed response, which otherwise reports none, and is - # where the metered counts in the access log come from. - self.assertEqual( - route["spec"]["llmRequestCosts"], - [ - {"metadataKey": "llm_input_token", "type": "InputToken"}, - {"metadataKey": "llm_output_token", "type": "OutputToken"}, - {"metadataKey": "llm_total_token", "type": "TotalToken"}, + "name": _be("together"), + "weight": 1, + "priority": 1, + "modelNameOverride": "Qwen/Qwen2.5", + }, ], - ) + "timeouts": {"request": "600s"}, + "streamIdleTimeout": "0s", + "modelsOwnedBy": _NS, + } + ] + # Declaring the token costs is what makes the ext-proc ask a backend for + # usage on a streamed response, which otherwise reports none, and is + # where the metered counts in the access log come from. + assert route["spec"]["llmRequestCosts"] == [ + {"metadataKey": "llm_input_token", "type": "InputToken"}, + {"metadataKey": "llm_output_token", "type": "OutputToken"}, + {"metadataKey": "llm_total_token", "type": "TotalToken"}, + ] + + # The composed backend pins its cluster's CA and presents the client + # certificate, both this route's own; the third-party one uses the + # system trust store. + assert _manifest(got, "backend-self")["spec"]["tls"] == { + "caCertificateRefs": [{"kind": "ConfigMap", "group": "", "name": "assistant-eu-gw-eu-ca-3dd16"}], + "sni": "gw-eu.example.com", + "clientCertificateRef": {"kind": "Secret", "group": "", "name": "assistant-eu-client-08324"}, + } + assert _manifest(got, "backend-together")["spec"]["tls"] == { + "wellKnownCACertificates": "System", + "sni": "api.together.xyz", + } - # The composed backend pins its cluster's CA and presents the client - # certificate, both this route's own; the third-party one uses the - # system trust store. - self.assertEqual( - _manifest(got, "backend-self")["spec"]["tls"], - { - "caCertificateRefs": [{"kind": "ConfigMap", "group": "", "name": "assistant-eu-gw-eu-ca-3dd16"}], - "sni": "gw-eu.example.com", - "clientCertificateRef": {"kind": "Secret", "group": "", "name": "assistant-eu-client-08324"}, - }, - ) - self.assertEqual( - _manifest(got, "backend-together")["spec"]["tls"], - {"wellKnownCACertificates": "System", "sni": "api.together.xyz"}, - ) + # The caller header is stripped only for the backend we don't operate. + assert "headerMutation" not in _manifest(got, "aibackend-self")["spec"] + assert _manifest(got, "aibackend-together")["spec"]["headerMutation"] == {"remove": ["x-modelplane-caller"]} - # The caller header is stripped only for the backend we don't operate. - self.assertNotIn("headerMutation", _manifest(got, "aibackend-self")["spec"]) - self.assertEqual( - _manifest(got, "aibackend-together")["spec"]["headerMutation"], - {"remove": ["x-modelplane-caller"]}, - ) + # The credential is republished under the fixed apiKey key. + assert _manifest(got, "credential-together")["data"] == {"apiKey": base64.b64encode(b"sk-tog").decode()} + # Named for this route, so no other route in the namespace composes it. + assert _manifest(got, "cluster-ca-gw-eu") == { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "assistant-eu-gw-eu-ca-3dd16", "namespace": "mp-ml-team-51733"}, + "data": {"ca.crt": _CLUSTER_CA}, + } - # The credential is republished under the fixed apiKey key. - self.assertEqual( - _manifest(got, "credential-together")["data"], - {"apiKey": base64.b64encode(b"sk-tog").decode()}, - ) - # Named for this route, so no other route in the namespace composes it. - self.assertEqual( - _manifest(got, "cluster-ca-gw-eu"), - { - "apiVersion": "v1", - "kind": "ConfigMap", - "metadata": {"name": "assistant-eu-gw-eu-ca-3dd16", "namespace": "mp-ml-team-51733"}, - "data": {"ca.crt": _CLUSTER_CA}, - }, - ) + # Every composed object lands in the namespace mirroring the route's own, + # which compose-inference-cluster composes. + for key in ("backend-self", "backend-together", "credential-together", "cluster-ca-gw-eu", "route"): + assert _manifest(got, key)["metadata"]["namespace"] == "mp-ml-team-51733", key - # Every composed object lands in the namespace mirroring the route's own, - # which compose-inference-cluster composes. - for key in ("backend-self", "backend-together", "credential-together", "cluster-ca-gw-eu", "route"): - self.assertEqual(_manifest(got, key)["metadata"]["namespace"], "mp-ml-team-51733", key) + # The route lives in the team namespace but attaches across to the gateway. + assert route["spec"]["parentRefs"][0]["namespace"] == "modelplane-system" - # The route lives in the team namespace but attaches across to the gateway. - self.assertEqual(route["spec"]["parentRefs"][0]["namespace"], "modelplane-system") + # The client certificate the backends present, issued from the gateway's + # CA ClusterIssuer into this namespace. Named for this route, so no other + # route in the namespace composes it, and deleted with the route. + assert ( + "managementPolicies" + not in resource.struct_to_dict(got.desired.resources["client-certificate"].resource)["spec"] + ) + assert _manifest(got, "client-certificate") == { + "apiVersion": "cert-manager.io/v1", + "kind": "Certificate", + "metadata": {"name": "assistant-eu-client-08324", "namespace": "mp-ml-team-51733"}, + "spec": { + "secretName": "assistant-eu-client-08324", + "commonName": "inference-gateway-eu", + "usages": ["client auth", "digital signature", "key encipherment"], + "duration": "2160h", + "renewBefore": "720h", + "privateKey": {"algorithm": "ECDSA", "size": 256, "rotationPolicy": "Always"}, + "issuerRef": {"name": "inference-gateway-ca", "kind": "ClusterIssuer", "group": "cert-manager.io"}, + }, + } - # The client certificate the backends present, issued from the gateway's - # CA ClusterIssuer into this namespace. Named for this route, so no other - # route in the namespace composes it, and deleted with the route. - self.assertNotIn( - "managementPolicies", - resource.struct_to_dict(got.desired.resources["client-certificate"].resource)["spec"], - ) - self.assertEqual( - _manifest(got, "client-certificate"), - { - "apiVersion": "cert-manager.io/v1", - "kind": "Certificate", - "metadata": {"name": "assistant-eu-client-08324", "namespace": "mp-ml-team-51733"}, - "spec": { - "secretName": "assistant-eu-client-08324", - "commonName": "inference-gateway-eu", - "usages": ["client auth", "digital signature", "key encipherment"], - "duration": "2160h", - "renewBefore": "720h", - "privateKey": {"algorithm": "ECDSA", "size": 256, "rotationPolicy": "Always"}, - "issuerRef": {"name": "inference-gateway-ca", "kind": "ClusterIssuer", "group": "cert-manager.io"}, - }, - }, - ) - async def test_route_binds_to_the_listener_matching_the_gateways_tls(self) -> None: - """A TLS gateway serves inference on its HTTPS listener alone, so the - route binds there; without TLS there's only the HTTP listener. Binding to - :80 on a TLS gateway would carry credentials in the clear.""" - for tls, want in ((False, "http"), (True, "https")): - with self.subTest(tls=tls): - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) - ), - required_resources=_required( - gateway=[_gateway(tls=tls)], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", composed=True)]}, - ), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual(_manifest(got, "route")["spec"]["parentRefs"][0]["sectionName"], want) - - async def test_status_reports_address_and_counts(self) -> None: - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")])))), - required_resources=_required( - gateway=[_gateway(address="203.0.113.9")], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", composed=True)]}, - ), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual( - resource.struct_to_dict(got.desired.composite.resource)["status"], - { - "model": _MODEL, - "address": "203.0.113.9", - "endpoints": {"total": 1, "ready": 1}, - }, - ) +@pytest.mark.parametrize(("tls", "want"), [(False, "http"), (True, "https")]) +def test_route_binds_to_the_listener_matching_the_gateways_tls(tls: bool, want: str) -> None: + """The route binds to the HTTPS listener on a TLS gateway, and to the HTTP listener otherwise.""" + # A TLS gateway serves inference on its HTTPS listener alone, so the route + # binds there; without TLS there's only the HTTP listener. Binding to :80 on + # a TLS gateway would carry credentials in the clear. + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")])))), + required_resources=_required( + gateway=[_gateway(tls=tls)], + clusters=[_cluster("gw-eu")], + **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", composed=True)]}, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert _manifest(got, "route")["spec"]["parentRefs"][0]["sectionName"] == want + + +def test_status_reports_address_and_counts() -> None: + """The status reports the model name, the gateway's address, and the endpoint counts.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")])))), + required_resources=_required( + gateway=[_gateway(address="203.0.113.9")], + clusters=[_cluster("gw-eu")], + **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", composed=True)]}, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert resource.struct_to_dict(got.desired.composite.resource)["status"] == { + "model": _MODEL, + "address": "203.0.113.9", + "endpoints": {"total": 1, "ready": 1}, + } - async def test_an_endpoint_matched_twice_belongs_to_the_first_entry(self) -> None: - """A canary entry and a catch-all entry must not both weight one - endpoint; the first that matches it wins.""" - entries = [_entry("kimi", name="canary", priority=0), _entry("kimi", name="catchall", priority=1)] - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-canary": [_endpoint("kimi-a", origin="https://a.example.com")], - "endpoints-catchall": [_endpoint("kimi-a", origin="https://a.example.com")], - }, - ), - ) - got = await self.runner.RunFunction(req, None) - # Neither endpoint is Modelplane-composed, so no client certificate is - # issued. - self.assertNotIn("client-certificate", got.desired.resources) - refs = _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] - self.assertEqual(refs, [{"name": _be("kimi-a"), "weight": 1, "priority": 0}]) - self.assertEqual( - resource.struct_to_dict(got.desired.composite.resource)["status"]["endpoints"], - {"total": 1, "ready": 1}, - ) - async def test_priorities_are_renumbered_without_gaps(self) -> None: - """A ModelService's priorities are an ordering; Envoy's are levels it - walks from 0. A user writing 0 and 5, or a tier gone unready during a - roll, would otherwise leave gaps in what Envoy gets.""" - entries = [_entry("a", priority=0), _entry("b", priority=5), _entry("c", priority=9)] - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - # The middle tier has no ready endpoint, so it drops out and - # must not leave a hole behind it. - "endpoints-a": [_endpoint("a-0", origin="https://a.example.com")], - "endpoints-b": [_endpoint("b-0", origin="https://b.example.com", ready=False)], - "endpoints-c": [_endpoint("c-0", origin="https://c.example.com")], - }, - ), - ) - got = await self.runner.RunFunction(req, None) - refs = _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] - self.assertEqual([r["priority"] for r in refs], [0, 1], "two tiers survive, renumbered 0 and 1") +def test_an_endpoint_matched_twice_belongs_to_the_first_entry() -> None: + """An endpoint two entries match belongs to the first of them.""" + # A canary entry and a catch-all entry must not both weight one endpoint; + # the first that matches it wins. + entries = [_entry("kimi", name="canary", priority=0), _entry("kimi", name="catchall", priority=1)] + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), + required_resources=_required( + gateway=[_gateway()], + clusters=[_cluster("gw-eu")], + **{ + "endpoints-canary": [_endpoint("kimi-a", origin="https://a.example.com")], + "endpoints-catchall": [_endpoint("kimi-a", origin="https://a.example.com")], + }, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + # Neither endpoint is Modelplane-composed, so no client certificate is + # issued. + assert "client-certificate" not in got.desired.resources + refs = _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] + assert refs == [{"name": _be("kimi-a"), "weight": 1, "priority": 0}] + assert resource.struct_to_dict(got.desired.composite.resource)["status"]["endpoints"] == {"total": 1, "ready": 1} + + +def test_priorities_are_renumbered_without_gaps() -> None: + """The priorities of the tiers that have a ready endpoint are renumbered from 0, without gaps.""" + # A ModelService's priorities are an ordering; Envoy's are levels it walks + # from 0. A user writing 0 and 5, or a tier gone unready during a roll, + # would otherwise leave gaps in what Envoy gets. + entries = [_entry("a", priority=0), _entry("b", priority=5), _entry("c", priority=9)] + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), + required_resources=_required( + gateway=[_gateway()], + clusters=[_cluster("gw-eu")], + **{ + # The middle tier has no ready endpoint, so it drops out and + # must not leave a hole behind it. + "endpoints-a": [_endpoint("a-0", origin="https://a.example.com")], + "endpoints-b": [_endpoint("b-0", origin="https://b.example.com", ready=False)], + "endpoints-c": [_endpoint("c-0", origin="https://c.example.com")], + }, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + refs = _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] + assert [r["priority"] for r in refs] == [0, 1], "two tiers survive, renumbered 0 and 1" @dataclasses.dataclass @@ -629,77 +592,71 @@ class WeightCase: want_refs: list[dict] -class TestWeights(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_weight_distribution(self) -> None: - def _origins(*names_: str) -> list[dict]: - return [_endpoint(n, origin=f"https://{n}.example.com") for n in names_] +def _weight_distribution_cases() -> list[WeightCase]: + def _origins(*names_: str) -> list[dict]: + return [_endpoint(n, origin=f"https://{n}.example.com") for n in names_] + + def _ref(ep: str, weight: int, priority: int = 0) -> dict: + return {"name": _be(ep), "weight": weight, "priority": priority} + + return [ + WeightCase( + # An entry's weight is written once but applied per backend, so it + # spreads over the endpoints it matched while the ratio between + # entries survives: 90 over three is 30 each, 10 over one is 10, + # reduced by the gcd to the smallest equivalent integers. + name="a weight spreads across a tier's endpoints, ratio preserved", + entries=[_entry("big", weight=90), _entry("small", weight=10)], + endpoints={ + "endpoints-big": _origins("big-0", "big-1", "big-2"), + "endpoints-small": _origins("small-0"), + }, + want_refs=[ + _ref("big-0", 3), + _ref("big-1", 3), + _ref("big-2", 3), + _ref("small-0", 1), + ], + ), + WeightCase( + # Weight 1 over five endpoints must floor none of them to 0, which + # would drop them from the load assignment rather than share. + name="a weight below its endpoint count floors no endpoint", + entries=[_entry("many", weight=1)], + endpoints={"endpoints-many": _origins("many-0", "many-1", "many-2", "many-3", "many-4")}, + want_refs=[_ref(f"many-{i}", 1) for i in range(5)], + ), + WeightCase( + # A max-weight entry beside a tiny one spread over two endpoints + # scales past the per-backendRef limit even though every weight is + # in bounds, so it rescales to the limit rather than composing a + # route the API server rejects. + name="an extreme but valid ratio is clamped to the limit", + entries=[_entry("big", weight=1000000, priority=0), _entry("small", weight=1, priority=0)], + endpoints={"endpoints-big": _origins("big-0"), "endpoints-small": _origins("small-0", "small-1")}, + want_refs=[_ref("big-0", 1000000), _ref("small-0", 1), _ref("small-1", 1)], + ), + WeightCase( + # The remainder is handed to the first endpoints of a tier, so the + # order must be the endpoints' names rather than the API server's + # unspecified list order, or the composed weights churn. + name="endpoints are ordered by name for a stable split", + entries=[_entry("d")], + endpoints={"endpoints-d": _origins("z", "a", "m")}, + want_refs=[_ref("a", 1), _ref("m", 1), _ref("z", 1)], + ), + ] - def _ref(ep: str, weight: int, priority: int = 0) -> dict: - return {"name": _be(ep), "weight": weight, "priority": priority} - cases = [ - WeightCase( - # An entry's weight is written once but applied per backend, so it - # spreads over the endpoints it matched while the ratio between - # entries survives: 90 over three is 30 each, 10 over one is 10, - # reduced by the gcd to the smallest equivalent integers. - name="a weight spreads across a tier's endpoints, ratio preserved", - entries=[_entry("big", weight=90), _entry("small", weight=10)], - endpoints={ - "endpoints-big": _origins("big-0", "big-1", "big-2"), - "endpoints-small": _origins("small-0"), - }, - want_refs=[ - _ref("big-0", 3), - _ref("big-1", 3), - _ref("big-2", 3), - _ref("small-0", 1), - ], - ), - WeightCase( - # Weight 1 over five endpoints must floor none of them to 0, which - # would drop them from the load assignment rather than share. - name="a weight below its endpoint count floors no endpoint", - entries=[_entry("many", weight=1)], - endpoints={"endpoints-many": _origins("many-0", "many-1", "many-2", "many-3", "many-4")}, - want_refs=[_ref(f"many-{i}", 1) for i in range(5)], - ), - WeightCase( - # A max-weight entry beside a tiny one spread over two endpoints - # scales past the per-backendRef limit even though every weight is - # in bounds, so it rescales to the limit rather than composing a - # route the API server rejects. - name="an extreme but valid ratio is clamped to the limit", - entries=[_entry("big", weight=1000000, priority=0), _entry("small", weight=1, priority=0)], - endpoints={"endpoints-big": _origins("big-0"), "endpoints-small": _origins("small-0", "small-1")}, - want_refs=[_ref("big-0", 1000000), _ref("small-0", 1), _ref("small-1", 1)], - ), - WeightCase( - # The remainder is handed to the first endpoints of a tier, so the - # order must be the endpoints' names rather than the API server's - # unspecified list order, or the composed weights churn. - name="endpoints are ordered by name for a stable split", - entries=[_entry("d")], - endpoints={"endpoints-d": _origins("z", "a", "m")}, - want_refs=[_ref("a", 1), _ref("m", 1), _ref("z", 1)], - ), - ] - for case in cases: - with self.subTest(case.name): - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(case.entries))) - ), - required_resources=_required(gateway=[_gateway()], clusters=[_cluster("gw-eu")], **case.endpoints), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual(_manifest(got, "route")["spec"]["rules"][0]["backendRefs"], case.want_refs) +@pytest.mark.parametrize("case", _weight_distribution_cases(), ids=lambda case: case.name) +def test_weight_distribution(case: WeightCase) -> None: + """Each entry's weight is distributed across the endpoints it matched.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(case.entries)))), + required_resources=_required(gateway=[_gateway()], clusters=[_cluster("gw-eu")], **case.endpoints), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] == case.want_refs @dataclasses.dataclass @@ -712,67 +669,61 @@ class CredentialCase: want: dict -class TestCredentials(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_credential_policy(self) -> None: - secret = resource.child_name(f"{_SVC}-{_GW}", "provider", "credential") - target = {"group": "aigateway.envoyproxy.io", "kind": "AIServiceBackend", "name": _be("provider")} - cases = [ - CredentialCase( - name="a backend speaking OpenAI's API gets the key as a bearer token", - api=None, - want={ - "apiVersion": "aigateway.envoyproxy.io/v1beta1", - "kind": "BackendSecurityPolicy", - "metadata": {"name": _be("provider"), "namespace": "mp-ml-team-51733"}, - "spec": { - "type": "APIKey", - "apiKey": {"secretRef": {"name": secret}}, - "targetRefs": [target], - }, +def _credential_policy_cases() -> list[CredentialCase]: + secret = resource.child_name(f"{_SVC}-{_GW}", "provider", "credential") + target = {"group": "aigateway.envoyproxy.io", "kind": "AIServiceBackend", "name": _be("provider")} + return [ + CredentialCase( + name="a backend speaking OpenAI's API gets the key as a bearer token", + api=None, + want={ + "apiVersion": "aigateway.envoyproxy.io/v1beta1", + "kind": "BackendSecurityPolicy", + "metadata": {"name": _be("provider"), "namespace": "mp-ml-team-51733"}, + "spec": { + "type": "APIKey", + "apiKey": {"secretRef": {"name": secret}}, + "targetRefs": [target], }, - ), - CredentialCase( - name="a backend speaking Anthropic's API gets the key in x-api-key", - api=mev1alpha1.Api(schema="Anthropic"), - want={ - "apiVersion": "aigateway.envoyproxy.io/v1beta1", - "kind": "BackendSecurityPolicy", - "metadata": {"name": _be("provider"), "namespace": "mp-ml-team-51733"}, - "spec": { - "type": "AnthropicAPIKey", - "anthropicAPIKey": {"secretRef": {"name": secret}}, - "targetRefs": [target], - }, + }, + ), + CredentialCase( + name="a backend speaking Anthropic's API gets the key in x-api-key", + api=mev1alpha1.Api(schema="Anthropic"), + want={ + "apiVersion": "aigateway.envoyproxy.io/v1beta1", + "kind": "BackendSecurityPolicy", + "metadata": {"name": _be("provider"), "namespace": "mp-ml-team-51733"}, + "spec": { + "type": "AnthropicAPIKey", + "anthropicAPIKey": {"secretRef": {"name": secret}}, + "targetRefs": [target], }, - ), - ] - for case in cases: - with self.subTest(case.name): - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) - ), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-d": [ - _endpoint( - "provider", - origin="https://api.example.com", - api=case.api, - credential="provider-key", - ) - ], - "credential-provider": [_secret("provider-key", {"apiKey": "sk-provider"})], - }, - ), - ) - got = await self.runner.RunFunction(req, None) - self.assertEqual(_manifest(got, "credpolicy-provider"), case.want) + }, + ), + ] + + +@pytest.mark.parametrize("case", _credential_policy_cases(), ids=lambda case: case.name) +def test_credential_policy(case: CredentialCase) -> None: + """A keyed backend's BackendSecurityPolicy sends the key the way its API expects.""" + req = fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")])))), + required_resources=_required( + gateway=[_gateway()], + clusters=[_cluster("gw-eu")], + **{ + "endpoints-d": [ + _endpoint( + "provider", + origin="https://api.example.com", + api=case.api, + credential="provider-key", + ) + ], + "credential-provider": [_secret("provider-key", {"apiKey": "sk-provider"})], + }, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert _manifest(got, "credpolicy-provider") == case.want diff --git a/functions/compose-model-service/tests/__init__.py b/functions/compose-model-service/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-model-service/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-model-service/tests/test_fn.py b/functions/compose-model-service/tests/test_fn.py index 7b10eab6b..4fb2975b9 100644 --- a/functions/compose-model-service/tests/test_fn.py +++ b/functions/compose-model-service/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-model-service function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.inferencegateway import v1alpha1 as igv1alpha1 from models.ai.modelplane.modelroute import v1alpha1 as mrtv1alpha1 @@ -146,277 +148,270 @@ def _required(**resources) -> dict: # noqa: ANN003 } -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - entries = [_entry("kimi-k2")] - cases = [ - Case( - name="gateways not resolved yet: require them and wait", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries)))), +def _compose_cases() -> list[Case]: + entries = [_entry("kimi-k2")] + return [ + Case( + name="gateways not resolved yet: require them and wait", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries)))), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"model": _MODEL, "routes": {"total": 0, "ready": 0}}} + ), + ready=fnv1.READY_FALSE, + ) ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 0, "ready": 0}}} - ), - ready=fnv1.READY_FALSE, - ) - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" - ), - } - ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_WAITING_FOR_GATEWAYS, - message="Waiting for the gateways to resolve", - ) - ], - results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the gateways to resolve")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" + ), + } ), + conditions=[ + fnv1.Condition( + type=fn.CONDITION_TYPE_ROUTING_READY, + status=fnv1.STATUS_CONDITION_FALSE, + reason=fn.CONDITION_REASON_WAITING_FOR_GATEWAYS, + message="Waiting for the gateways to resolve", + ) + ], + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the gateways to resolve")], ), - Case( - name="no gateway selects the service: unreachable, and say so", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_service(entries, labels={"region": "us"})) - ) - ), - required_resources=_required(gateways=[_gateway("eu", "gw-eu", selector={"region": "eu"})]), + ), + Case( + name="no gateway selects the service: unreachable, and say so", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct(_service(entries, labels={"region": "us"})) + ) ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 0, "ready": 0}}} - ), - ready=fnv1.READY_FALSE, - ) - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" - ), - } - ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_NO_GATEWAY, - message=( - "No InferenceGateway's serviceSelector matches this service's labels, " - "so no caller can reach it" - ), - ) - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message=( - "No InferenceGateway's serviceSelector matches this service's labels, " - "so no caller can reach it" - ), - ) - ], + required_resources=_required(gateways=[_gateway("eu", "gw-eu", selector={"region": "eu"})]), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"model": _MODEL, "routes": {"total": 0, "ready": 0}}} + ), + ready=fnv1.READY_FALSE, + ) ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" + ), + } + ), + conditions=[ + fnv1.Condition( + type=fn.CONDITION_TYPE_ROUTING_READY, + status=fnv1.STATUS_CONDITION_FALSE, + reason=fn.CONDITION_REASON_NO_GATEWAY, + message=( + "No InferenceGateway's serviceSelector matches this service's labels, " + "so no caller can reach it" + ), + ) + ], + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message=( + "No InferenceGateway's serviceSelector matches this service's labels, " + "so no caller can reach it" + ), + ) + ], ), - Case( - # A gateway with no address is left out of readiness, but with no - # other gateway there is nowhere a caller could reach the service. - name="its only gateway has no address yet: not RoutingReady", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - ), - required_resources=_required(gateways=[_gateway("eu", "gw-eu")]), + ), + Case( + # A gateway with no address is left out of readiness, but with no + # other gateway there is nowhere a caller could reach the service. + name="its only gateway has no address yet: not RoutingReady", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 1, "ready": 0}}} - ), - ready=fnv1.READY_FALSE, + required_resources=_required(gateways=[_gateway("eu", "gw-eu")]), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"model": _MODEL, "routes": {"total": 1, "ready": 0}}} ), - resources={ - "route-eu": fnv1.Resource(resource=resource.dict_to_struct(_route("eu", "gw-eu"))), - }, + ready=fnv1.READY_FALSE, ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" - ), - } - ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_WAITING_FOR_GATEWAYS, - message="Waiting for gateways to come up: eu", - ) + resources={ + "route-eu": fnv1.Resource(resource=resource.dict_to_struct(_route("eu", "gw-eu"))), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" + ), + } + ), + conditions=[ + fnv1.Condition( + type=fn.CONDITION_TYPE_ROUTING_READY, + status=fnv1.STATUS_CONDITION_FALSE, + reason=fn.CONDITION_REASON_WAITING_FOR_GATEWAYS, + message="Waiting for gateways to come up: eu", + ) + ], + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for gateways to come up: eu")], + ), + ), + Case( + name="two gateways serve it: a ModelRoute each, waiting for both routes", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), + ), + required_resources=_required( + gateways=[ + _gateway("eu", "gw-eu", address="203.0.113.1"), + _gateway("us", "gw-us", address="203.0.113.2"), ], - results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for gateways to come up: eu")], ), ), - Case( - name="two gateways serve it: a ModelRoute each, waiting for both routes", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - ), - required_resources=_required( - gateways=[ - _gateway("eu", "gw-eu", address="203.0.113.1"), - _gateway("us", "gw-us", address="203.0.113.2"), - ], + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"model": _MODEL, "routes": {"total": 2, "ready": 0}}} + ), + ready=fnv1.READY_FALSE, ), + resources={ + "route-eu": fnv1.Resource(resource=resource.dict_to_struct(_route("eu", "gw-eu"))), + "route-us": fnv1.Resource(resource=resource.dict_to_struct(_route("us", "gw-us"))), + }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 2, "ready": 0}}} - ), - ready=fnv1.READY_FALSE, + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" ), - resources={ - "route-eu": fnv1.Resource(resource=resource.dict_to_struct(_route("eu", "gw-eu"))), - "route-us": fnv1.Resource(resource=resource.dict_to_struct(_route("us", "gw-us"))), - }, - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" - ), - } - ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_WAITING_FOR_ROUTES, - message="Waiting for routes on gateways: eu, us", - ) - ], - results=[ - fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for routes on gateways: eu, us") + } + ), + conditions=[ + fnv1.Condition( + type=fn.CONDITION_TYPE_ROUTING_READY, + status=fnv1.STATUS_CONDITION_FALSE, + reason=fn.CONDITION_REASON_WAITING_FOR_ROUTES, + message="Waiting for routes on gateways: eu, us", + ) + ], + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for routes on gateways: eu, us")], + ), + ), + Case( + name="both routes accepted: ModelRoutes ready, service RoutingReady", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), + resources={ + "route-eu": _observed_route("eu", ready=True), + "route-us": _observed_route("us", ready=True), + }, + ), + required_resources=_required( + gateways=[ + _gateway("eu", "gw-eu", address="203.0.113.1"), + _gateway("us", "gw-us", address="203.0.113.2"), ], ), ), - Case( - name="both routes accepted: ModelRoutes ready, service RoutingReady", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - resources={ - "route-eu": _observed_route("eu", ready=True), - "route-us": _observed_route("us", ready=True), - }, - ), - required_resources=_required( - gateways=[ - _gateway("eu", "gw-eu", address="203.0.113.1"), - _gateway("us", "gw-us", address="203.0.113.2"), - ], + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"model": _MODEL, "routes": {"total": 2, "ready": 2}}} + ), + ready=fnv1.READY_TRUE, ), + resources={ + "route-eu": fnv1.Resource( + resource=resource.dict_to_struct(_route("eu", "gw-eu")), ready=fnv1.READY_TRUE + ), + "route-us": fnv1.Resource( + resource=resource.dict_to_struct(_route("us", "gw-us")), ready=fnv1.READY_TRUE + ), + }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 2, "ready": 2}}} - ), - ready=fnv1.READY_TRUE, + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" ), - resources={ - "route-eu": fnv1.Resource( - resource=resource.dict_to_struct(_route("eu", "gw-eu")), ready=fnv1.READY_TRUE - ), - "route-us": fnv1.Resource( - resource=resource.dict_to_struct(_route("us", "gw-us")), ready=fnv1.READY_TRUE - ), - }, - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" - ), - } - ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_TRUE, - reason=fn.CONDITION_REASON_ROUTES_ACCEPTED, - ) - ], + } ), - ), - ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) - - async def test_absent_selector_serves_every_service(self) -> None: - """A gateway with no serviceSelector serves the service, and one still - coming up (no address) is excluded from readiness rather than failing - it: both RoutingReady and the service's own Ready ignore its unready - ModelRoute.""" - entries = [_entry("kimi-k2")] - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - resources={"route-eu": _observed_route("eu", ready=True)}, - ), - required_resources=_required( - gateways=[ - _gateway("eu", "gw-eu", address="203.0.113.1"), - _gateway("us", "gw-us"), # no address: still coming up + conditions=[ + fnv1.Condition( + type=fn.CONDITION_TYPE_ROUTING_READY, + status=fnv1.STATUS_CONDITION_TRUE, + reason=fn.CONDITION_REASON_ROUTES_ACCEPTED, + ) ], ), - ) - got = await self.runner.RunFunction(req, None) - self.assertIn("route-eu", got.desired.resources) - self.assertIn("route-us", got.desired.resources) - cond = next(c for c in got.conditions if c.type == fn.CONDITION_TYPE_ROUTING_READY) - self.assertEqual(cond.status, fnv1.STATUS_CONDITION_TRUE, "us has no address, so it doesn't block") - self.assertEqual(got.desired.composite.ready, fnv1.READY_TRUE, "nor does its unready ModelRoute") + ), + ] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes a ModelRoute per serving gateway and reports routing readiness.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) + + +def test_absent_selector_serves_every_service() -> None: + """A gateway with no serviceSelector serves the service, and one with no address doesn't block readiness.""" + # A gateway with no serviceSelector serves the service, and one still + # coming up (no address) is excluded from readiness rather than failing + # it: both RoutingReady and the service's own Ready ignore its unready + # ModelRoute. + entries = [_entry("kimi-k2")] + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), + resources={"route-eu": _observed_route("eu", ready=True)}, + ), + required_resources=_required( + gateways=[ + _gateway("eu", "gw-eu", address="203.0.113.1"), + _gateway("us", "gw-us"), # no address: still coming up + ], + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + assert "route-eu" in got.desired.resources + assert "route-us" in got.desired.resources + cond = next(c for c in got.conditions if c.type == fn.CONDITION_TYPE_ROUTING_READY) + assert cond.status == fnv1.STATUS_CONDITION_TRUE, "us has no address, so it doesn't block" + assert got.desired.composite.ready == fnv1.READY_TRUE, "nor does its unready ModelRoute" diff --git a/functions/compose-nebius-cluster/tests/test_fn.py b/functions/compose-nebius-cluster/tests/test_fn.py index 0561ac006..74b131657 100644 --- a/functions/compose-nebius-cluster/tests/test_fn.py +++ b/functions/compose-nebius-cluster/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-nebius-cluster function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.nebiuscluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -36,10 +38,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - # The Nebius ClusterProviderConfig the function reads the credentials # Secret off. _NEBIUS_PROVIDER_CONFIG = { @@ -414,405 +412,402 @@ def _observed_ready(desired: dict) -> fnv1.Resource: ) -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function composes Nebius mk8s cluster infrastructure.""" - cases = [ - Case( - name="first pass composes infra resources; autoscaling from maxNodeCount", - req=_req([_GPU_POOL]), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - # The CSI driver release and StorageClass aren't - # composed yet: the cluster isn't observed, so the - # ProviderConfigs can't reach it. - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - # nodeCount defaults to 1 and minNodeCount is - # unset, so autoscaling starts at the node count. - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), +def _compose_cases() -> list[Case]: + """The cases for test_compose, with requirements patched onto their wants by position.""" + cases = [ + Case( + name="first pass composes infra resources; autoscaling from maxNodeCount", + req=_req([_GPU_POOL]), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), + "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), + "cloud-init": fnv1.Resource( + resource=resource.dict_to_struct(_cloud_init_secret()), + ready=fnv1.READY_TRUE, + ), + # The CSI driver release and StorageClass aren't + # composed yet: the cluster isn't observed, so the + # ProviderConfigs can't reach it. + "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), + # nodeCount defaults to 1 and minNodeCount is + # unset, so autoscaling starts at the node count. + "nodegroup-gpu-h100": fnv1.Resource( + resource=resource.dict_to_struct( + _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), ), - }, - ), - context=structpb.Struct(), + ready=fnv1.READY_TRUE, + ), + }, ), + context=structpb.Struct(), ), - Case( - name="provider config not yet fetched gates provider configs, not infra", - req=_req([_GPU_POOL], with_provider_config=False), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_status(with_credentials=False)), - ), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), - ), - }, + ), + Case( + name="provider config not yet fetched gates provider configs, not infra", + req=_req([_GPU_POOL], with_provider_config=False), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct(_status(with_credentials=False)), ), - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for Nebius ClusterProviderConfig default", + resources={ + "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), + "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), + "cloud-init": fnv1.Resource( + resource=resource.dict_to_struct(_cloud_init_secret()), + ready=fnv1.READY_TRUE, ), - ], - context=structpb.Struct(), + "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), + "nodegroup-gpu-h100": fnv1.Resource( + resource=resource.dict_to_struct( + _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), + ), + ), + }, ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for Nebius ClusterProviderConfig default", + ), + ], + context=structpb.Struct(), ), - Case( - name="deleted provider config keeps credentials from the observed ProviderConfig", - req=_req( - [_GPU_POOL], - observed_resources={ + ), + Case( + name="deleted provider config keeps credentials from the observed ProviderConfig", + req=_req( + [_GPU_POOL], + observed_resources={ + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + ), + ), + }, + with_provider_config=False, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), + "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), + "cloud-init": fnv1.Resource( + resource=resource.dict_to_struct(_cloud_init_secret()), + ready=fnv1.READY_TRUE, + ), + "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), + "nodegroup-gpu-h100": fnv1.Resource( + resource=resource.dict_to_struct( + _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), + ), + ), "provider-config-kubernetes": fnv1.Resource( resource=resource.dict_to_struct( _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), ), + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + ), + ready=fnv1.READY_TRUE, ), }, - with_provider_config=False, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Nebius ClusterProviderConfig default not found; keeping the " + "credentials the composed ProviderConfig already carries", ), - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Nebius ClusterProviderConfig default not found; keeping the " - "credentials the composed ProviderConfig already carries", + ], + context=structpb.Struct(), + ), + ), + Case( + name="fixed-size fabric pool composes a GPU cluster and fixedNodeCount", + req=_req( + [ + v1alpha1.NodePool( + name="gpu-h100", + role="GPU", + platform="gpu-h100-sxm", + preset="8gpu-128vcpu-1600gb", + diskSizeGb=200, + nodeCount=2, + fabric=v1alpha1.Fabric( + type="InfiniBand", + infiniband=v1alpha1.Infiniband(fabric="fabric-2"), ), - ], - context=structpb.Struct(), - ), + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + ), + ] ), - Case( - name="fixed-size fabric pool composes a GPU cluster and fixedNodeCount", - req=_req( - [ - v1alpha1.NodePool( - name="gpu-h100", - role="GPU", - platform="gpu-h100-sxm", - preset="8gpu-128vcpu-1600gb", - diskSizeGb=200, - nodeCount=2, - fabric=v1alpha1.Fabric( - type="InfiniBand", - infiniband=v1alpha1.Infiniband(fabric="fabric-2"), - ), - gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), + "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), + "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), + "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), + "cloud-init": fnv1.Resource( + resource=resource.dict_to_struct(_cloud_init_secret()), + ready=fnv1.READY_TRUE, ), - ] - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - "gpu-cluster-fabric-2": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "compute.nebius.m.upbound.io/v1beta1", - "kind": "GpuCluster", - "metadata": {"labels": {"modelplane.ai/fabric": "fabric-2"}}, - "spec": { - "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, - "forProvider": { - "name": "test-cluster-fabric-2", - "infinibandFabric": "fabric-2", - }, + "gpu-cluster-fabric-2": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.nebius.m.upbound.io/v1beta1", + "kind": "GpuCluster", + "metadata": {"labels": {"modelplane.ai/fabric": "fabric-2"}}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "name": "test-cluster-fabric-2", + "infinibandFabric": "fabric-2", }, - } - ), + }, + } ), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu( - { - "gpuCluster": { - "idSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/fabric": "fabric-2"}, - }, + ), + "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), + "nodegroup-gpu-h100": fnv1.Resource( + resource=resource.dict_to_struct( + _nodegroup_gpu( + { + "gpuCluster": { + "idSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/fabric": "fabric-2"}, }, }, - fixedNodeCount=2, - ), + }, + fixedNodeCount=2, ), ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), ), - }, - ), - context=structpb.Struct(), - ), - ), - Case( - name="marks managed resources ready from observed conditions", - req=_req( - [_GPU_POOL], - observed_resources={ - "network": _observed_ready(_network()), - "subnet": _observed_ready(_subnet()), - "cluster": _observed_ready(_cluster()), - "filesystem": _observed_ready(_filesystem()), - "release-csi-mounted-fs-path": _observed_ready(_csi_release()), - "nodegroup-system": _observed_ready(_nodegroup_system()), - "nodegroup-gpu-h100": _observed_ready( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), + ready=fnv1.READY_TRUE, ), }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ready=fnv1.READY_TRUE, - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ready=fnv1.READY_TRUE, - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "filesystem": fnv1.Resource( - resource=resource.dict_to_struct(_filesystem()), - ready=fnv1.READY_TRUE, - ), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - # The cluster is observed, so the CSI driver - # release and StorageClass are composed too. - "release-csi-mounted-fs-path": fnv1.Resource( - resource=resource.dict_to_struct(_csi_release()), - ready=fnv1.READY_TRUE, - ), - "storage-class-rwx-fs": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource( - resource=resource.dict_to_struct(_nodegroup_system()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), - ready=fnv1.READY_TRUE, + context=structpb.Struct(), + ), + ), + Case( + name="marks managed resources ready from observed conditions", + req=_req( + [_GPU_POOL], + observed_resources={ + "network": _observed_ready(_network()), + "subnet": _observed_ready(_subnet()), + "cluster": _observed_ready(_cluster()), + "filesystem": _observed_ready(_filesystem()), + "release-csi-mounted-fs-path": _observed_ready(_csi_release()), + "nodegroup-system": _observed_ready(_nodegroup_system()), + "nodegroup-gpu-h100": _observed_ready( + _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct(_network()), + ready=fnv1.READY_TRUE, + ), + "subnet": fnv1.Resource( + resource=resource.dict_to_struct(_subnet()), + ready=fnv1.READY_TRUE, + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, + ), + "filesystem": fnv1.Resource( + resource=resource.dict_to_struct(_filesystem()), + ready=fnv1.READY_TRUE, + ), + "cloud-init": fnv1.Resource( + resource=resource.dict_to_struct(_cloud_init_secret()), + ready=fnv1.READY_TRUE, + ), + # The cluster is observed, so the CSI driver + # release and StorageClass are composed too. + "release-csi-mounted-fs-path": fnv1.Resource( + resource=resource.dict_to_struct(_csi_release()), + ready=fnv1.READY_TRUE, + ), + "storage-class-rwx-fs": fnv1.Resource( + resource=resource.dict_to_struct(_storage_class()), + ready=fnv1.READY_TRUE, + ), + "nodegroup-system": fnv1.Resource( + resource=resource.dict_to_struct(_nodegroup_system()), + ready=fnv1.READY_TRUE, + ), + "nodegroup-gpu-h100": fnv1.Resource( + resource=resource.dict_to_struct( + _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), ), - }, - ), - context=structpb.Struct(), + ready=fnv1.READY_TRUE, + ), + }, ), + context=structpb.Struct(), ), - Case( - name="custom credentials flow through to all cloud MRs", - req=_req( - [_GPU_POOL], - credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-nebius-account"), - provider_config_resource={ - "apiVersion": "nebius.m.upbound.io/v1beta1", - "kind": "ProviderConfig", - "metadata": {"name": "my-nebius-account", "namespace": "crossplane-system"}, - "spec": { - "identity": {"type": "ServiceAccount"}, - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "crossplane-system", - "name": "nebius-credentials", - "key": "credentials.json", - }, + ), + Case( + name="custom credentials flow through to all cloud MRs", + req=_req( + [_GPU_POOL], + credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-nebius-account"), + provider_config_resource={ + "apiVersion": "nebius.m.upbound.io/v1beta1", + "kind": "ProviderConfig", + "metadata": {"name": "my-nebius-account", "namespace": "crossplane-system"}, + "spec": { + "identity": {"type": "ServiceAccount"}, + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "crossplane-system", + "name": "nebius-credentials", + "key": "credentials.json", }, - "projectID": "project-e00test", }, + "projectID": "project-e00test", }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network("ProviderConfig", "my-nebius-account")), - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet("ProviderConfig", "my-nebius-account")), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster("ProviderConfig", "my-nebius-account")), - ), - "filesystem": fnv1.Resource( - resource=resource.dict_to_struct(_filesystem("ProviderConfig", "my-nebius-account")), - ), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_system("ProviderConfig", "my-nebius-account"), - ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct(_network("ProviderConfig", "my-nebius-account")), + ), + "subnet": fnv1.Resource( + resource=resource.dict_to_struct(_subnet("ProviderConfig", "my-nebius-account")), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster("ProviderConfig", "my-nebius-account")), + ), + "filesystem": fnv1.Resource( + resource=resource.dict_to_struct(_filesystem("ProviderConfig", "my-nebius-account")), + ), + "cloud-init": fnv1.Resource( + resource=resource.dict_to_struct(_cloud_init_secret()), + ready=fnv1.READY_TRUE, + ), + "nodegroup-system": fnv1.Resource( + resource=resource.dict_to_struct( + _nodegroup_system("ProviderConfig", "my-nebius-account"), ), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu( - {}, - "ProviderConfig", - "my-nebius-account", - autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, - ), + ), + "nodegroup-gpu-h100": fnv1.Resource( + resource=resource.dict_to_struct( + _nodegroup_gpu( + {}, + "ProviderConfig", + "my-nebius-account", + autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, ), ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), ), - }, - ), - context=structpb.Struct(), + ready=fnv1.READY_TRUE, + ), + }, ), + context=structpb.Struct(), ), - ] - - # Every compose path declares the provider config requirement; the - # selector kind and name vary by credentials. - custom_creds_selector = fnv1.ResourceSelector( - api_version="nebius.m.upbound.io/v1beta1", - kind="ProviderConfig", - match_name="my-nebius-account", - namespace="modelplane-system", - ) - for case in cases[:-1]: - case.want.requirements.resources["nebius-provider-config"].CopyFrom(_PROVIDER_CONFIG_SELECTOR) - cases[-1].want.requirements.resources["nebius-provider-config"].CopyFrom(custom_creds_selector) - - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) + ), + ] + + # Every compose path declares the provider config requirement; the + # selector kind and name vary by credentials. + custom_creds_selector = fnv1.ResourceSelector( + api_version="nebius.m.upbound.io/v1beta1", + kind="ProviderConfig", + match_name="my-nebius-account", + namespace="modelplane-system", + ) + for case in cases[:-1]: + case.want.requirements.resources["nebius-provider-config"].CopyFrom(_PROVIDER_CONFIG_SELECTOR) + cases[-1].want.requirements.resources["nebius-provider-config"].CopyFrom(custom_creds_selector) + return cases + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes a NebiusCluster's network, mk8s cluster, node groups and ProviderConfigs.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-serving-stack/tests/__init__.py b/functions/compose-serving-stack/tests/__init__.py deleted file mode 100644 index 5d373016d..000000000 --- a/functions/compose-serving-stack/tests/__init__.py +++ /dev/null @@ -1,15 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - - diff --git a/functions/compose-serving-stack/tests/test_fn.py b/functions/compose-serving-stack/tests/test_fn.py index 30668f60e..9213ba2fb 100644 --- a/functions/compose-serving-stack/tests/test_fn.py +++ b/functions/compose-serving-stack/tests/test_fn.py @@ -23,17 +23,19 @@ resource - for every cloud and stack, as frozen literals. """ +import asyncio import copy import dataclasses +import json import pathlib -import unittest +import pytest import yaml -from crossplane.function import logging, resource +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.servingstack import v1alpha1 from models.io.crossplane.m.helm.providerconfig import v1beta1 as helmpcv1beta1 @@ -45,11 +47,6 @@ from models.io.crossplane.protection.usage import v1beta1 as usagev1beta1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 - -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - # Precomputed child_name value for test-backend. _PC_NAME = "test-backend-cluster-63fde" @@ -847,328 +844,315 @@ class Case: want: fnv1.RunFunctionResponse -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - full = _provider_configs() | _EXISTING_DYNAMO_USAGES | _existing_dynamo_stack() - - # Second pass: PCs observed. depends_on gates first creation, so - # only the dependency-free wave renders; each dependent waits for - # its dependency's Ready before it is first created. - dep_gated = { - "envoy-gateway", # -> cert-manager - "ai-gateway", # -> ai-gateway-crds - "gateway-proxy", # -> gateway-namespace - "kai-queue-root", # -> kai-scheduler - "kai-queue", # -> kai-scheduler - "modelexpress-server", # -> modelexpress-crds - "gateway-selfsigned-issuer", # -> cert-manager, gateway-namespace - "trust-manager", # -> gateway-selfsigned-issuer - } - first_wave = {k: v for k, v in full.items() if k not in dep_gated} - - # Third pass: every rendered resource observed Ready (the gateway - # with its address assigned), so everything is marked ready and - # the address lands in the XR status. - rendered = [k for k in _existing_dynamo_stack() if k != "gateway"] - observed_ready = _observed_pcs() - for key in rendered: - observed_ready[key] = fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ) - observed_ready["gateway"] = fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "conditions": [{"type": "Ready", "status": "True"}], - "atProvider": { - "manifest": {"status": {"addresses": [{"type": "IPAddress", "value": "203.0.113.7"}]}}, - }, +def _compose_cases() -> list[Case]: + """The test_compose cases, built from the Existing/Dynamo stack's resources.""" + full = _provider_configs() | _EXISTING_DYNAMO_USAGES | _existing_dynamo_stack() + + # Second pass: PCs observed. depends_on gates first creation, so + # only the dependency-free wave renders; each dependent waits for + # its dependency's Ready before it is first created. + dep_gated = { + "envoy-gateway", # -> cert-manager + "ai-gateway", # -> ai-gateway-crds + "gateway-proxy", # -> gateway-namespace + "kai-queue-root", # -> kai-scheduler + "kai-queue", # -> kai-scheduler + "modelexpress-server", # -> modelexpress-crds + "gateway-selfsigned-issuer", # -> cert-manager, gateway-namespace + "trust-manager", # -> gateway-selfsigned-issuer + } + first_wave = {k: v for k, v in full.items() if k not in dep_gated} + + # Third pass: every rendered resource observed Ready (the gateway + # with its address assigned), so everything is marked ready and + # the address lands in the XR status. + rendered = [k for k in _existing_dynamo_stack() if k != "gateway"] + observed_ready = _observed_pcs() + for key in rendered: + observed_ready[key] = fnv1.Resource( + resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) + ) + observed_ready["gateway"] = fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "conditions": [{"type": "Ready", "status": "True"}], + "atProvider": { + "manifest": {"status": {"addresses": [{"type": "IPAddress", "value": "203.0.113.7"}]}}, }, - } - ) + }, + } ) - # Every component observed Ready; PCs and Usages are ready on arrival. - all_ready = copy.deepcopy(full) - for res in all_ready.values(): - res.ready = fnv1.READY_TRUE - - cases = [ - Case( - name="first pass composes only the provider configs and usages", - req=_request("Existing", "Dynamo"), - # Everything targeting the remote cluster is gated on the - # ProviderConfigs having been observed; Usages reference - # nothing remote and compose immediately. The unready - # ProviderConfigs keep the composite unready until the - # stack actually renders. - want=_response(_provider_configs(ready=False) | _EXISTING_DYNAMO_USAGES), - ), - Case( - name="second pass renders the dependency-free wave", - req=_request("Existing", "Dynamo", observed=_observed_pcs()), - want=_response(first_wave), - ), - Case( - name="all dependencies ready renders the whole stack, marks it ready, and writes the gateway address", - req=_request("Existing", "Dynamo", observed=observed_ready), - want=_response(all_ready, status={"gateway": {"address": "203.0.113.7"}}), + ) + # Every component observed Ready; PCs and Usages are ready on arrival. + all_ready = copy.deepcopy(full) + for res in all_ready.values(): + res.ready = fnv1.READY_TRUE + + return [ + Case( + name="first pass composes only the provider configs and usages", + req=_request("Existing", "Dynamo"), + # Everything targeting the remote cluster is gated on the + # ProviderConfigs having been observed; Usages reference + # nothing remote and compose immediately. The unready + # ProviderConfigs keep the composite unready until the + # stack actually renders. + want=_response(_provider_configs(ready=False) | _EXISTING_DYNAMO_USAGES), + ), + Case( + name="second pass renders the dependency-free wave", + req=_request("Existing", "Dynamo", observed=_observed_pcs()), + want=_response(first_wave), + ), + Case( + name="all dependencies ready renders the whole stack, marks it ready, and writes the gateway address", + req=_request("Existing", "Dynamo", observed=observed_ready), + want=_response(all_ready, status={"gateway": {"address": "203.0.113.7"}}), + ), + ] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes the Existing/Dynamo stack across the reconcile passes.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) + + +def test_identity_secret_type_flows_to_provider_configs() -> None: + """A non-GCP identity secret's type and namespace reach both ProviderConfigs verbatim.""" + # The type is stamped as is rather than being forced to + # GoogleApplicationCredentials, and the secret's own namespace wins over + # the XR's. + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ServingStack( + metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), + spec=v1alpha1.Spec( + cloud="Nebius", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret( + type="NebiusServiceAccountCredentials", + name="nebius-secret", + key="credentials.json", + namespace="other-ns", + ), + ], + gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), + ), + ).model_dump(exclude_none=True, mode="json") + ), ), - ] - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) - - async def test_identity_secret_type_flows_to_provider_configs(self) -> None: - """A non-GCP identity secret's type is stamped verbatim on both - ProviderConfigs rather than being forced to GoogleApplicationCredentials, - and its own namespace wins over the XR's.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Nebius", - secrets=[ - v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), - v1alpha1.Secret( - type="NebiusServiceAccountCredentials", - name="nebius-secret", - key="credentials.json", - namespace="other-ns", - ), + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + pc = resource.struct_to_dict(got.desired.resources["provider-config-kubernetes"].resource) + assert pc["spec"]["identity"]["type"] == "NebiusServiceAccountCredentials" + assert pc["spec"]["identity"]["secretRef"]["namespace"] == "other-ns" + helm_pc = resource.struct_to_dict(got.desired.resources["provider-config-helm"].resource) + assert helm_pc["spec"]["identity"]["type"] == "NebiusServiceAccountCredentials" + + +def test_cluster_gateway_composes_mtls_with_ca() -> None: + """A cluster with an InferenceGateway CA composes its own PKI and serves mTLS.""" + # It issues its own PKI, republishes the CA without its key, demands a + # client certificate on its HTTPS listener, and publishes the CA in status. + # + # The hostname is a full Service FQDN, so the CA certificate's commonName + # overflows the 64-byte X.509 limit and is truncated. + hostname = "gateway-test-backend-12345.modelplane-system.svc.cluster.local" + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ServingStack( + metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), + spec=v1alpha1.Spec( + cloud="Existing", + stack="Standard", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway( + hostname=hostname, + # Deliberately out of name order, to prove the + # bundle sorts before concatenating. + clientCAs=[ + v1alpha1.ClientCA(name="fleet-b", certificate="BBB"), + v1alpha1.ClientCA(name="fleet-a", certificate="AAA"), ], - gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), ), - ).model_dump(exclude_none=True, mode="json") - ), + ), + ).model_dump(exclude_none=True, mode="json") ), ), - ) - got = await self.runner.RunFunction(req, None) - pc = resource.struct_to_dict(got.desired.resources["provider-config-kubernetes"].resource) - self.assertEqual("NebiusServiceAccountCredentials", pc["spec"]["identity"]["type"]) - self.assertEqual("other-ns", pc["spec"]["identity"]["secretRef"]["namespace"]) - helm_pc = resource.struct_to_dict(got.desired.resources["provider-config-helm"].resource) - self.assertEqual("NebiusServiceAccountCredentials", helm_pc["spec"]["identity"]["type"]) - - async def test_cluster_gateway_composes_mtls_with_ca(self) -> None: - """A cluster with an InferenceGateway CA serves mTLS: it issues its own - PKI, republishes the CA without its key, demands a client certificate on - its HTTPS listener, and publishes the CA in status. - - The hostname is a full Service FQDN, so the CA certificate's commonName - overflows the 64-byte X.509 limit and is truncated. - """ - hostname = "gateway-test-backend-12345.modelplane-system.svc.cluster.local" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( + # PCs observed, the self-signed Issuer Ready (so trust-manager and + # the CA chain proceed), and the CA ConfigMap trust-manager syncs + # carrying the certificate back for status. + resources=_observed_pcs() + | { + "gateway-selfsigned-issuer": fnv1.Resource( + resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) + ), + "gateway-ca-configmap": fnv1.Resource( resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Existing", - stack="Standard", - secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], - gateway=v1alpha1.Gateway( - hostname=hostname, - # Deliberately out of name order, to prove the - # bundle sorts before concatenating. - clientCAs=[ - v1alpha1.ClientCA(name="fleet-b", certificate="BBB"), - v1alpha1.ClientCA(name="fleet-a", certificate="AAA"), - ], - ), - ), - ).model_dump(exclude_none=True, mode="json") - ), + {"status": {"atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}}} + ) ), - # PCs observed, the self-signed Issuer Ready (so trust-manager and - # the CA chain proceed), and the CA ConfigMap trust-manager syncs - # carrying the certificate back for status. - resources=_observed_pcs() - | { - "gateway-selfsigned-issuer": fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"conditions": [{"type": "Ready", "status": "True"}]}} - ) - ), - "gateway-ca-configmap": fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}}} - ) - ), - }, - ), - ) - got = await self.runner.RunFunction(req, None) + }, + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + + def manifest(key: str) -> dict: + return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] + + ca_cert = manifest("gateway-ca-certificate") + assert ca_cert["spec"]["commonName"] == "modelplane cluster CA gateway-test-backend-12345.modelplane-syst" + assert len(ca_cert["spec"]["commonName"]) <= 64 + assert ca_cert["spec"]["isCA"] + assert ca_cert["spec"]["issuerRef"]["name"] == "modelplane-selfsigned" + + serving = manifest("gateway-serving-certificate") + assert serving["spec"]["dnsNames"] == [hostname] + assert serving["spec"]["issuerRef"]["name"] == "modelplane-cluster-ca" + + bundle = manifest("gateway-ca-bundle") + assert bundle["apiVersion"] == "trust.cert-manager.io/v1alpha1" + assert bundle["spec"]["sources"] == [{"secret": {"name": "modelplane-cluster-ca", "key": "ca.crt"}}] + + # Observed only, never managed: trust-manager owns the ConfigMap. + ca_cm = got.desired.resources["gateway-ca-configmap"] + assert resource.struct_to_dict(ca_cm.resource)["spec"]["managementPolicies"] == ["Observe"] + + # Every InferenceGateway's CA, sorted by name and concatenated. + client_bundle = manifest("gateway-client-ca-bundle") + assert client_bundle["data"]["ca.crt"] == "AAA\nBBB\n" + + client_auth = manifest("gateway-client-auth") + assert client_auth["kind"] == "ClientTrafficPolicy" + assert client_auth["spec"]["targetRefs"][0]["sectionName"] == "https" + assert ( + client_auth["spec"]["tls"]["clientValidation"]["caCertificateRefs"][0]["name"] + == "modelplane-inference-gateway-cas" + ) - def manifest(key: str) -> dict: - return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] + # One HTTPS listener, terminating TLS with the serving certificate. + gateway = manifest("gateway") + assert gateway["spec"]["listeners"] == [ + { + "name": "https", + "protocol": "HTTPS", + "port": 443, + "hostname": hostname, + "tls": {"mode": "Terminate", "certificateRefs": [{"name": "cluster-gateway-serving"}]}, + "allowedRoutes": { + "namespaces": { + "from": "Selector", + "selector": {"matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}]}, + } + }, + } + ] - ca_cert = manifest("gateway-ca-certificate") - self.assertEqual( - "modelplane cluster CA gateway-test-backend-12345.modelplane-syst", ca_cert["spec"]["commonName"] - ) - self.assertLessEqual(len(ca_cert["spec"]["commonName"]), 64) - self.assertTrue(ca_cert["spec"]["isCA"]) - self.assertEqual("modelplane-selfsigned", ca_cert["spec"]["issuerRef"]["name"]) - - serving = manifest("gateway-serving-certificate") - self.assertEqual([hostname], serving["spec"]["dnsNames"]) - self.assertEqual("modelplane-cluster-ca", serving["spec"]["issuerRef"]["name"]) - - bundle = manifest("gateway-ca-bundle") - self.assertEqual("trust.cert-manager.io/v1alpha1", bundle["apiVersion"]) - self.assertEqual([{"secret": {"name": "modelplane-cluster-ca", "key": "ca.crt"}}], bundle["spec"]["sources"]) - - # Observed only, never managed: trust-manager owns the ConfigMap. - ca_cm = got.desired.resources["gateway-ca-configmap"] - self.assertEqual(["Observe"], resource.struct_to_dict(ca_cm.resource)["spec"]["managementPolicies"]) - - # Every InferenceGateway's CA, sorted by name and concatenated. - client_bundle = manifest("gateway-client-ca-bundle") - self.assertEqual("AAA\nBBB\n", client_bundle["data"]["ca.crt"]) - - client_auth = manifest("gateway-client-auth") - self.assertEqual("ClientTrafficPolicy", client_auth["kind"]) - self.assertEqual("https", client_auth["spec"]["targetRefs"][0]["sectionName"]) - self.assertEqual( - "modelplane-inference-gateway-cas", - client_auth["spec"]["tls"]["clientValidation"]["caCertificateRefs"][0]["name"], - ) + status = resource.struct_to_dict(got.desired.composite.resource)["status"] + assert status["gateway"]["caCertificate"] == "CLUSTERCA" - # One HTTPS listener, terminating TLS with the serving certificate. - gateway = manifest("gateway") - self.assertEqual( - [ + # Every PKI resource must be tracked for readiness: + # compose_gateway_pki marks only the keys it returns, so one composed + # but not returned would silently hold the cluster un-Ready. Observe + # each Ready and assert it's marked ready, which fails if the key was + # dropped from the rendered list. (The self-signed Issuer and + # trust-manager are stack components, covered by the golden test.) + pki_keys = [ + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + ] + for key in pki_keys: + req.observed.resources[key].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) + ) + ) + # Preserve the CA ConfigMap's data alongside its Ready condition. + req.observed.resources["gateway-ca-configmap"].CopyFrom( + fnv1.Resource( + resource=resource.dict_to_struct( { - "name": "https", - "protocol": "HTTPS", - "port": 443, - "hostname": hostname, - "tls": {"mode": "Terminate", "certificateRefs": [{"name": "cluster-gateway-serving"}]}, - "allowedRoutes": { - "namespaces": { - "from": "Selector", - "selector": { - "matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}] - }, - } - }, + "status": { + "conditions": [{"type": "Ready", "status": "True"}], + "atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}, + } } - ], - gateway["spec"]["listeners"], + ) ) + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + for key in pki_keys: + assert got.desired.resources[key].ready == fnv1.READY_TRUE, f"{key} not marked ready" - status = resource.struct_to_dict(got.desired.composite.resource)["status"] - self.assertEqual("CLUSTERCA", status["gateway"]["caCertificate"]) - - # Every PKI resource must be tracked for readiness: - # compose_gateway_pki marks only the keys it returns, so one composed - # but not returned would silently hold the cluster un-Ready. Observe - # each Ready and assert it's marked ready, which fails if the key was - # dropped from the rendered list. (The self-signed Issuer and - # trust-manager are stack components, covered by the golden test.) - pki_keys = [ - "gateway-ca-certificate", - "gateway-ca-issuer", - "gateway-serving-certificate", - "gateway-ca-bundle", - "gateway-ca-configmap", - "gateway-client-ca-bundle", - "gateway-client-auth", - ] - for key in pki_keys: - req.observed.resources[key].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ) - ) - # Preserve the CA ConfigMap's data alongside its Ready condition. - req.observed.resources["gateway-ca-configmap"].CopyFrom( - fnv1.Resource( + +def test_cluster_gateway_without_ca_serves_nothing() -> None: + """A cluster with no InferenceGateway CA withholds its Gateway, and warns.""" + # Withholding the Gateway entirely, rather than serving the engines + # unauthenticated. + req = fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=fnv1.Resource( resource=resource.dict_to_struct( - { - "status": { - "conditions": [{"type": "Ready", "status": "True"}], - "atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}, - } - } - ) - ) - ) - got = await self.runner.RunFunction(req, None) - for key in pki_keys: - self.assertEqual(fnv1.READY_TRUE, got.desired.resources[key].ready, f"{key} not marked ready") - - async def test_cluster_gateway_without_ca_serves_nothing(self) -> None: - """A cluster with no InferenceGateway CA withholds the Gateway entirely - rather than serving the engines unauthenticated, and warns.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Existing", - stack="Standard", - secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], - gateway=v1alpha1.Gateway(hostname="gw.clusters.example.com"), - ), - ).model_dump(exclude_none=True, mode="json") - ), + v1alpha1.ServingStack( + metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), + spec=v1alpha1.Spec( + cloud="Existing", + stack="Standard", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway(hostname="gw.clusters.example.com"), + ), + ).model_dump(exclude_none=True, mode="json") ), - resources=_observed_pcs(), + ), + resources=_observed_pcs(), + ), + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + # The GatewayClass and the cluster's own PKI are composed, so the CA is + # ready to publish when the first InferenceGateway's CA arrives. The + # Gateway, the client CA bundle and the policy demanding a client + # certificate aren't, and nor is the Usage protecting the Gateway. + gateway_keys = {k for k in got.desired.resources if k.startswith(("gateway", "usage-gateway"))} + assert gateway_keys == { + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-namespace", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + } + assert list(got.results) == [ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message=( + "Gateway gw.clusters.example.com not served: no InferenceGateway has published a client " + "CA for this cluster to trust, and serving without one would accept unauthenticated callers" ), ) - got = await self.runner.RunFunction(req, None) - # The GatewayClass and the cluster's own PKI are composed, so the CA is - # ready to publish when the first InferenceGateway's CA arrives. The - # Gateway, the client CA bundle and the policy demanding a client - # certificate aren't, and nor is the Usage protecting the Gateway. - gateway_keys = {k for k in got.desired.resources if k.startswith(("gateway", "usage-gateway"))} - self.assertEqual( - { - "gateway-class", - "gateway-ca-certificate", - "gateway-ca-issuer", - "gateway-serving-certificate", - "gateway-ca-bundle", - "gateway-ca-configmap", - "gateway-namespace", - "usage-gateway-namespace-by-gateway-proxy", - "usage-gateway-namespace-by-gateway-selfsigned-issuer", - "usage-gateway-selfsigned-issuer-by-trust-manager", - }, - gateway_keys, - ) - self.assertEqual( - [ - fnv1.Result( - severity=fnv1.SEVERITY_WARNING, - message=( - "Gateway gw.clusters.example.com not served: no InferenceGateway has published a client " - "CA for this cluster to trust, and serving without one would accept unauthenticated callers" - ), - ) - ], - list(got.results), - ) + ] # The composed-resource key a component renders under is its identity: @@ -1316,28 +1300,21 @@ async def test_cluster_gateway_without_ca_serves_nothing(self) -> None: } -class TestKeyInventory(unittest.IsolatedAsyncioTestCase): - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_composed_resource_keys(self) -> None: - for cloud, cloud_keys in _INVENTORY.items(): - for stack, stack_keys in (("Standard", _STANDARD), ("Dynamo", _DYNAMO)): - with self.subTest(cloud=cloud, stack=stack): - expected = _ALWAYS | _COMMON | cloud_keys | stack_keys - # Observe every expected key Ready so the depends_on - # install gate opens and the full stack renders; a - # key the function doesn't render still fails the - # comparison. - observed = _observed_pcs() - for key in expected: - observed[key] = fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"conditions": [{"type": "Ready", "status": "True"}]}} - ) - ) - got = await self.runner.RunFunction(_request(cloud, stack, observed=observed), None) - self.assertEqual(expected, set(got.desired.resources.keys())) +@pytest.mark.parametrize( + ("stack", "stack_keys"), [("Standard", _STANDARD), ("Dynamo", _DYNAMO)], ids=["Standard", "Dynamo"] +) +@pytest.mark.parametrize(("cloud", "cloud_keys"), list(_INVENTORY.items()), ids=list(_INVENTORY)) +def test_composed_resource_keys(cloud: str, cloud_keys: frozenset[str], stack: str, stack_keys: frozenset[str]) -> None: + """Every cloud and stack composes exactly its inventoried resource keys.""" + expected = _ALWAYS | _COMMON | cloud_keys | stack_keys + # Observe every expected key Ready so the depends_on + # install gate opens and the full stack renders; a + # key the function doesn't render still fails the + # comparison. + observed = _observed_pcs() + for key in expected: + observed[key] = fnv1.Resource( + resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) + ) + got = asyncio.run(fn.FunctionRunner().RunFunction(_request(cloud, stack, observed=observed), None)) + assert set(got.desired.resources.keys()) == expected diff --git a/functions/compose-serving-stack/tests/test_stacks.py b/functions/compose-serving-stack/tests/test_stacks.py index a33684eab..2017a2264 100644 --- a/functions/compose-serving-stack/tests/test_stacks.py +++ b/functions/compose-serving-stack/tests/test_stacks.py @@ -20,114 +20,115 @@ generated ones, once mapped - without involving fn.py. """ -import unittest - +import pytest from function import stacks -class TestComponents(unittest.TestCase): - def test_every_cloud_and_stack_joins(self) -> None: - for cloud in stacks.clouds(): - for stack in stacks.stacks(): - with self.subTest(cloud=cloud, stack=stack): - got = stacks.join(cloud, stack) - self.assertTrue(got, "a joined stack can't be empty") - - def test_charts_have_reserved_release_names(self) -> None: - for cloud in stacks.clouds(): - for stack in stacks.stacks(): - for c in stacks.join(cloud, stack): - if isinstance(c, stacks.Chart): - with self.subTest(cloud=cloud, stack=stack, key=c.key): - self.assertEqual( - f"mp-{c.chart}", - c.release, - "release names are mp-: stable across upgrades, reserved to Modelplane", - ) - - def test_manifests_are_populated(self) -> None: - for cloud in stacks.clouds(): - for stack in stacks.stacks(): - for c in stacks.join(cloud, stack): - if isinstance(c, stacks.Manifests): - with self.subTest(cloud=cloud, stack=stack, key=c.key): - self.assertTrue(c.manifests, "a Manifests entry can't be empty") - - def test_multi_doc_manifests_derive_per_doc_keys(self) -> None: - for cloud in stacks.clouds(): - for stack in stacks.stacks(): - for c in stacks.join(cloud, stack): - keys = stacks.components.doc_keys(c) - if isinstance(c, stacks.Chart) or len(c.manifests) == 1: - self.assertEqual([c.key], keys) - continue - with self.subTest(cloud=cloud, stack=stack, key=c.key): - self.assertEqual( - [f"{c.key}-{doc['metadata']['name']}" for doc in c.manifests], - keys, - "a multi-doc bundle renders one Object per doc, keyed -", - ) - - def test_ready_entries_are_single_doc(self) -> None: - # A readiness CEL query applies to every doc in an entry, so an - # entry carrying one keeps to a single manifest - a Service or - # ServiceAccount has no status conditions to satisfy it. - for cloud in stacks.clouds(): - for stack in stacks.stacks(): - for c in stacks.join(cloud, stack): - if isinstance(c, stacks.Manifests) and c.ready is not None: - with self.subTest(cloud=cloud, stack=stack, key=c.key): - self.assertEqual(1, len(c.manifests)) - - def test_depended_on_charts_wait(self) -> None: - # A chart another component depends on renders with helm --wait, - # so its Ready means healthy and the install gate orders - # dependents on health rather than deploy. Without this, the - # gate would open the moment Helm accepted the manifests. - for cloud in stacks.clouds(): - for stack in stacks.stacks(): - joined = stacks.join(cloud, stack) - depended_on = {dep for c in joined for dep in c.depends_on} - for c in joined: - if isinstance(c, stacks.Chart) and c.key in depended_on: - with self.subTest(cloud=cloud, stack=stack, key=c.key): - self.assertTrue(c.wait, "a depended-on chart must set wait") - - def test_no_wildcard_tolerations(self) -> None: - # A keyless toleration tolerates every taint, so the pod lands - # on tainted GPU nodes: control-plane charts squat on - # accelerated capacity and their eviction stalls autoscaler - # scale-down. aicr's bundler stamps exactly that wildcard on - # every pod it renders; the generator scopes each one - # (TOLERATIONS in generate.py). This pins that no keyless - # toleration survives in any joined stack, chart values and - # manifests alike. - def check(node: object, where: str) -> None: - if isinstance(node, dict): - for key, val in node.items(): - if key == "tolerations" and isinstance(val, list): - for toleration in val: - self.assertTrue( - isinstance(toleration, dict) and "key" in toleration, - f"keyless (wildcard) toleration in {where}", - ) - else: - check(val, where) - elif isinstance(node, list): - for item in node: - check(item, where) - - for cloud in stacks.clouds(): - for stack in stacks.stacks(): - for c in stacks.join(cloud, stack): - with self.subTest(cloud=cloud, stack=stack, key=c.key): - check(c.values if isinstance(c, stacks.Chart) else c.manifests, c.key) - - def test_unknown_cloud_and_stack_fail_closed(self) -> None: - # The Literal types reject these at type-checking time; this - # exercises the runtime guard behind them, which catches the API - # and the stacks package disagreeing on a value. - with self.assertRaises(ValueError): - stacks.join("Mars", "Standard") # ty: ignore[invalid-argument-type] - with self.assertRaises(ValueError): - stacks.join("Nebius", "Turbo") # ty: ignore[invalid-argument-type] +@pytest.mark.parametrize("stack", stacks.stacks()) +@pytest.mark.parametrize("cloud", stacks.clouds()) +def test_every_cloud_and_stack_joins(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """Every cloud and stack pair joins into a non-empty stack.""" + got = stacks.join(cloud, stack) + assert got, "a joined stack can't be empty" + + +@pytest.mark.parametrize("stack", stacks.stacks()) +@pytest.mark.parametrize("cloud", stacks.clouds()) +def test_charts_have_reserved_release_names(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """Every Chart's Helm release is named mp-.""" + for c in stacks.join(cloud, stack): + if isinstance(c, stacks.Chart): + assert c.release == f"mp-{c.chart}", ( + f"{c.key}: release names are mp-: stable across upgrades, reserved to Modelplane" + ) + + +@pytest.mark.parametrize("stack", stacks.stacks()) +@pytest.mark.parametrize("cloud", stacks.clouds()) +def test_manifests_are_populated(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """Every Manifests entry carries at least one manifest.""" + for c in stacks.join(cloud, stack): + if isinstance(c, stacks.Manifests): + assert c.manifests, f"{c.key}: a Manifests entry can't be empty" + + +@pytest.mark.parametrize("stack", stacks.stacks()) +@pytest.mark.parametrize("cloud", stacks.clouds()) +def test_multi_doc_manifests_derive_per_doc_keys(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """A multi-doc Manifests entry renders a key per doc, and anything else renders its own key.""" + for c in stacks.join(cloud, stack): + keys = stacks.components.doc_keys(c) + if isinstance(c, stacks.Chart) or len(c.manifests) == 1: + assert keys == [c.key] + continue + assert keys == [f"{c.key}-{doc['metadata']['name']}" for doc in c.manifests], ( + f"{c.key}: a multi-doc bundle renders one Object per doc, keyed -" + ) + + +@pytest.mark.parametrize("stack", stacks.stacks()) +@pytest.mark.parametrize("cloud", stacks.clouds()) +def test_ready_entries_are_single_doc(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """A Manifests entry with a readiness query carries a single manifest.""" + # A readiness CEL query applies to every doc in an entry, so an + # entry carrying one keeps to a single manifest - a Service or + # ServiceAccount has no status conditions to satisfy it. + for c in stacks.join(cloud, stack): + if isinstance(c, stacks.Manifests) and c.ready is not None: + assert len(c.manifests) == 1, c.key + + +@pytest.mark.parametrize("stack", stacks.stacks()) +@pytest.mark.parametrize("cloud", stacks.clouds()) +def test_depended_on_charts_wait(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """Every Chart another component depends on sets wait.""" + # A chart another component depends on renders with helm --wait, + # so its Ready means healthy and the install gate orders + # dependents on health rather than deploy. Without this, the + # gate would open the moment Helm accepted the manifests. + joined = stacks.join(cloud, stack) + depended_on = {dep for c in joined for dep in c.depends_on} + for c in joined: + if isinstance(c, stacks.Chart) and c.key in depended_on: + assert c.wait, f"{c.key}: a depended-on chart must set wait" + + +@pytest.mark.parametrize("stack", stacks.stacks()) +@pytest.mark.parametrize("cloud", stacks.clouds()) +def test_no_wildcard_tolerations(cloud: stacks.Cloud, stack: stacks.Stack) -> None: + """No component of a joined stack carries a keyless toleration.""" + + # A keyless toleration tolerates every taint, so the pod lands + # on tainted GPU nodes: control-plane charts squat on + # accelerated capacity and their eviction stalls autoscaler + # scale-down. aicr's bundler stamps exactly that wildcard on + # every pod it renders; the generator scopes each one + # (TOLERATIONS in generate.py). This pins that no keyless + # toleration survives in any joined stack, chart values and + # manifests alike. + def check(node: object, where: str) -> None: + if isinstance(node, dict): + for key, val in node.items(): + if key == "tolerations" and isinstance(val, list): + for toleration in val: + assert isinstance(toleration, dict), f"keyless (wildcard) toleration in {where}" + assert "key" in toleration, f"keyless (wildcard) toleration in {where}" + else: + check(val, where) + elif isinstance(node, list): + for item in node: + check(item, where) + + for c in stacks.join(cloud, stack): + check(c.values if isinstance(c, stacks.Chart) else c.manifests, c.key) + + +def test_unknown_cloud_and_stack_fail_closed() -> None: + """join rejects an unknown cloud or stack.""" + # The Literal types reject these at type-checking time; this + # exercises the runtime guard behind them, which catches the API + # and the stacks package disagreeing on a value. + with pytest.raises(ValueError, match="unknown cloud 'Mars'"): + stacks.join("Mars", "Standard") # ty: ignore[invalid-argument-type] + with pytest.raises(ValueError, match="unknown stack 'Turbo'"): + stacks.join("Nebius", "Turbo") # ty: ignore[invalid-argument-type] diff --git a/functions/compose-usages/tests/__init__.py b/functions/compose-usages/tests/__init__.py deleted file mode 100644 index b53d39d12..000000000 --- a/functions/compose-usages/tests/__init__.py +++ /dev/null @@ -1,14 +0,0 @@ -# Copyright 2026 The Modelplane Authors. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - diff --git a/functions/compose-usages/tests/test_fn.py b/functions/compose-usages/tests/test_fn.py index 531ce2c7c..912f59a70 100644 --- a/functions/compose-usages/tests/test_fn.py +++ b/functions/compose-usages/tests/test_fn.py @@ -14,14 +14,16 @@ """Tests for the compose-usages function.""" +import asyncio import dataclasses -import unittest +import json -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb _NAMESPACE = "test-ns" @@ -131,132 +133,123 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - cases = [ - Case( - name="labels each consumer and composes a Usage per ProviderConfig reference", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=_composite())), - desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), - resources={ - "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), - "gateway-namespace": fnv1.Resource(resource=resource.dict_to_struct(_OBJECT)), - "prometheus": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE_WITH_LABEL)), - "config-map": fnv1.Resource(resource=resource.dict_to_struct(_OBJECT_NO_PC)), - "provider-config-helm": fnv1.Resource(resource=resource.dict_to_struct(_PROVIDER_CONFIG)), - }, +COMPOSE_CASES = [ + Case( + name="labels each consumer and composes a Usage per ProviderConfig reference", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=_composite())), + desired=fnv1.State( + composite=fnv1.Resource(resource=_composite()), + resources={ + "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), + "gateway-namespace": fnv1.Resource(resource=resource.dict_to_struct(_OBJECT)), + "prometheus": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE_WITH_LABEL)), + "config-map": fnv1.Resource(resource=resource.dict_to_struct(_OBJECT_NO_PC)), + "provider-config-helm": fnv1.Resource(resource=resource.dict_to_struct(_PROVIDER_CONFIG)), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=_composite()), + resources={ + "cert-manager": fnv1.Resource( + resource=resource.dict_to_struct(_labelled(_RELEASE, "cert-manager")), ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), - resources={ - "cert-manager": fnv1.Resource( - resource=resource.dict_to_struct(_labelled(_RELEASE, "cert-manager")), - ), - "gateway-namespace": fnv1.Resource( - resource=resource.dict_to_struct(_labelled(_OBJECT, "gateway-namespace")), - ), - # Existing labels are preserved when the consumer label is stamped. - "prometheus": fnv1.Resource( - resource=resource.dict_to_struct(_labelled(_RELEASE_WITH_LABEL, "prometheus")), - ), - # An Object with no providerConfigRef is left untouched, no Usage. - "config-map": fnv1.Resource( - resource=resource.dict_to_struct(_OBJECT_NO_PC), - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_PROVIDER_CONFIG), - ), - "usage-pc-cert-manager": fnv1.Resource( - resource=resource.dict_to_struct( - _usage("helm.m.crossplane.io/v1beta1", "Release", "cert-manager") - ), - ready=fnv1.READY_TRUE, - ), - "usage-pc-gateway-namespace": fnv1.Resource( - resource=resource.dict_to_struct( - _usage("kubernetes.m.crossplane.io/v1alpha1", "Object", "gateway-namespace") - ), - ready=fnv1.READY_TRUE, - ), - "usage-pc-prometheus": fnv1.Resource( - resource=resource.dict_to_struct( - _usage("helm.m.crossplane.io/v1beta1", "Release", "prometheus") - ), - ready=fnv1.READY_TRUE, - ), - }, + "gateway-namespace": fnv1.Resource( + resource=resource.dict_to_struct(_labelled(_OBJECT, "gateway-namespace")), ), - context=structpb.Struct(), - ), - ), - Case( - name="no Usages when the composite has no namespace", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=_composite(namespace=None))), - desired=fnv1.State( - composite=fnv1.Resource(resource=_composite(namespace=None)), - resources={ - "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), - }, + # Existing labels are preserved when the consumer label is stamped. + "prometheus": fnv1.Resource( + resource=resource.dict_to_struct(_labelled(_RELEASE_WITH_LABEL, "prometheus")), ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=_composite(namespace=None)), - resources={ - "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), - }, + # An Object with no providerConfigRef is left untouched, no Usage. + "config-map": fnv1.Resource( + resource=resource.dict_to_struct(_OBJECT_NO_PC), ), - context=structpb.Struct(), - ), - ), - Case( - name="no consumers means no Usages", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=_composite())), - desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), - resources={ - "provider-config-helm": fnv1.Resource(resource=resource.dict_to_struct(_PROVIDER_CONFIG)), - }, + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_PROVIDER_CONFIG), + ), + "usage-pc-cert-manager": fnv1.Resource( + resource=resource.dict_to_struct( + _usage("helm.m.crossplane.io/v1beta1", "Release", "cert-manager") + ), + ready=fnv1.READY_TRUE, ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), - resources={ - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_PROVIDER_CONFIG), - ), - }, + "usage-pc-gateway-namespace": fnv1.Resource( + resource=resource.dict_to_struct( + _usage("kubernetes.m.crossplane.io/v1alpha1", "Object", "gateway-namespace") + ), + ready=fnv1.READY_TRUE, ), - context=structpb.Struct(), - ), + "usage-pc-prometheus": fnv1.Resource( + resource=resource.dict_to_struct( + _usage("helm.m.crossplane.io/v1beta1", "Release", "prometheus") + ), + ready=fnv1.READY_TRUE, + ), + }, ), - ] + context=structpb.Struct(), + ), + ), + Case( + name="no Usages when the composite has no namespace", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=_composite(namespace=None))), + desired=fnv1.State( + composite=fnv1.Resource(resource=_composite(namespace=None)), + resources={ + "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=_composite(namespace=None)), + resources={ + "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), + }, + ), + context=structpb.Struct(), + ), + ), + Case( + name="no consumers means no Usages", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=fnv1.Resource(resource=_composite())), + desired=fnv1.State( + composite=fnv1.Resource(resource=_composite()), + resources={ + "provider-config-helm": fnv1.Resource(resource=resource.dict_to_struct(_PROVIDER_CONFIG)), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=_composite()), + resources={ + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct(_PROVIDER_CONFIG), + ), + }, + ), + context=structpb.Struct(), + ), + ), +] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction labels consumers and composes their Usages.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-vultr-cluster/tests/test_fn.py b/functions/compose-vultr-cluster/tests/test_fn.py index 436a356c0..608ed76f7 100644 --- a/functions/compose-vultr-cluster/tests/test_fn.py +++ b/functions/compose-vultr-cluster/tests/test_fn.py @@ -14,15 +14,17 @@ """Tests for the compose-vultr-cluster function.""" +import asyncio import dataclasses -import unittest +import json from typing import Any -from crossplane.function import logging, resource +import pytest +from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 from function import fn from google.protobuf import duration_pb2 as durationpb -from google.protobuf import json_format +from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.vultrcluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @@ -37,10 +39,6 @@ class Case: want: fnv1.RunFunctionResponse -def setUpModule() -> None: - logging.configure(level=logging.Level.DISABLED) - - # Name of the cluster's connection secret. Derived like the function derives # it - the hash suffix depends only on the parent and child names. _KUBECONFIG_SECRET_NAME = resource.child_name("test-cluster", "kubeconfig") @@ -277,277 +275,269 @@ def _observed_unready(desired: dict) -> fnv1.Resource: ) -class TestFunctionRunner(unittest.IsolatedAsyncioTestCase): - """Tests for FunctionRunner.RunFunction.""" - - maxDiff = None - - @classmethod - def setUpClass(cls) -> None: - cls.runner = fn.FunctionRunner() - - async def test_compose(self) -> None: - """The function composes VKE cluster infrastructure.""" - cases = [ - Case( - name="cluster composed first; node pools withheld until cluster Ready", - req=_req([_GPU_POOL]), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - }, +COMPOSE_CASES = [ + Case( + name="cluster composed first; node pools withheld until cluster Ready", + req=_req([_GPU_POOL]), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), ), - context=structpb.Struct(), - ), + }, ), - Case( - name="node pools and GPU observer composed once cluster is Ready; autoscaling from maxNodeCount", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), - ), - }, + context=structpb.Struct(), + ), + ), + Case( + name="node pools and GPU observer composed once cluster is Ready; autoscaling from maxNodeCount", + req=_req( + [_GPU_POOL], + observed_resources={ + "cluster": _observed_ready(_cluster()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, ), - context=structpb.Struct(), - ), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config()), + ready=fnv1.READY_TRUE, + ), + "gpu-observer": fnv1.Resource( + resource=resource.dict_to_struct(_gpu_observer()), + ), + }, ), - Case( - name="dependents kept when the cluster Ready condition transiently regresses", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_unready(_cluster()), - "provider-config-kubernetes": _observed_ready(_provider_config()), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), - ), - }, + context=structpb.Struct(), + ), + ), + Case( + name="dependents kept when the cluster Ready condition transiently regresses", + req=_req( + [_GPU_POOL], + observed_resources={ + "cluster": _observed_unready(_cluster()), + "provider-config-kubernetes": _observed_ready(_provider_config()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), ), - context=structpb.Struct(), - ), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config()), + ready=fnv1.READY_TRUE, + ), + "gpu-observer": fnv1.Resource( + resource=resource.dict_to_struct(_gpu_observer()), + ), + }, ), - Case( - name="observed node pool alone keeps dependents composed", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_unready(_cluster()), - "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ready=fnv1.READY_TRUE, - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), - ), - }, + context=structpb.Struct(), + ), + ), + Case( + name="observed node pool alone keeps dependents composed", + req=_req( + [_GPU_POOL], + observed_resources={ + "cluster": _observed_unready(_cluster()), + "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), ), - context=structpb.Struct(), - ), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), + ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config()), + ready=fnv1.READY_TRUE, + ), + "gpu-observer": fnv1.Resource( + resource=resource.dict_to_struct(_gpu_observer()), + ), + }, ), - Case( - name="fixed-size GPU pool", - req=_req( - [ - v1alpha1.NodePool( - name="gpu-l40s", - role="GPU", - plan="vcg-l40s-16c-180g-48vram", - nodeCount=2, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), - ), - ], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, + context=structpb.Struct(), + ), + ), + Case( + name="fixed-size GPU pool", + req=_req( + [ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + plan="vcg-l40s-16c-180g-48vram", + nodeCount=2, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct( - _node_pool( - label="gpu-l40s", - plan="vcg-l40s-16c-180g-48vram", - node_quantity=2, - labels=[ - {"key": "modelplane.ai/pool", "value": "gpu-l40s"}, - {"key": "modelplane.ai/gpu", "value": "nvidia-l40s"}, - {"key": "nvidia.com/gpu.deploy.device-plugin", "value": "false"}, - ], - taints=_GPU_TAINTS, - ), - ), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), + ], + observed_resources={ + "cluster": _observed_ready(_cluster()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, + ), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct( + _node_pool( + label="gpu-l40s", + plan="vcg-l40s-16c-180g-48vram", + node_quantity=2, + labels=[ + {"key": "modelplane.ai/pool", "value": "gpu-l40s"}, + {"key": "modelplane.ai/gpu", "value": "nvidia-l40s"}, + {"key": "nvidia.com/gpu.deploy.device-plugin", "value": "false"}, + ], + taints=_GPU_TAINTS, ), - }, + ), ), - context=structpb.Struct(), - ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config()), + ready=fnv1.READY_TRUE, + ), + "gpu-observer": fnv1.Resource( + resource=resource.dict_to_struct(_gpu_observer()), + ), + }, ), - Case( - name="minNodeCount sets the autoscaler floor; System pool carries no taint", - req=_req( - [ - v1alpha1.NodePool( - name="workers", - role="System", - plan="vc2-6c-16gb", - nodeCount=2, - minNodeCount=2, - maxNodeCount=5, - ), - ], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, + context=structpb.Struct(), + ), + ), + Case( + name="minNodeCount sets the autoscaler floor; System pool carries no taint", + req=_req( + [ + v1alpha1.NodePool( + name="workers", + role="System", + plan="vc2-6c-16gb", + nodeCount=2, + minNodeCount=2, + maxNodeCount=5, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-workers": fnv1.Resource( - resource=resource.dict_to_struct( - _node_pool( - label="workers", - plan="vc2-6c-16gb", - node_quantity=2, - labels=[{"key": "modelplane.ai/pool", "value": "workers"}], - auto_scaler=True, - min_nodes=2, - max_nodes=5, - ), - ), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), + ], + observed_resources={ + "cluster": _observed_ready(_cluster()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, + ), + "node-pool-workers": fnv1.Resource( + resource=resource.dict_to_struct( + _node_pool( + label="workers", + plan="vc2-6c-16gb", + node_quantity=2, + labels=[{"key": "modelplane.ai/pool", "value": "workers"}], + auto_scaler=True, + min_nodes=2, + max_nodes=5, ), - }, + ), ), - context=structpb.Struct(), - ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config()), + ready=fnv1.READY_TRUE, + ), + "gpu-observer": fnv1.Resource( + resource=resource.dict_to_struct(_gpu_observer()), + ), + }, ), - Case( - name="VultrCluster Ready only once the gpu-observer is Ready", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_ready(_cluster()), - "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN), - "gpu-observer": _observed_ready(_gpu_observer()), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ready=fnv1.READY_TRUE, - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), - ready=fnv1.READY_TRUE, - ), - }, + context=structpb.Struct(), + ), + ), + Case( + name="VultrCluster Ready only once the gpu-observer is Ready", + req=_req( + [_GPU_POOL], + observed_resources={ + "cluster": _observed_ready(_cluster()), + "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN), + "gpu-observer": _observed_ready(_gpu_observer()), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct(_cluster()), + ready=fnv1.READY_TRUE, ), - context=structpb.Struct(), - ), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), + ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct(_provider_config()), + ready=fnv1.READY_TRUE, + ), + "gpu-observer": fnv1.Resource( + resource=resource.dict_to_struct(_gpu_observer()), + ready=fnv1.READY_TRUE, + ), + }, ), - ] - - for case in cases: - with self.subTest(case.name): - got = await self.runner.RunFunction(case.req, None) - self.assertEqual( - json_format.MessageToDict(case.want), - json_format.MessageToDict(got), - "-want, +got", - ) + context=structpb.Struct(), + ), + ), +] + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes VKE cluster infrastructure.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/nix/checks.nix b/nix/checks.nix index 000ff36d9..6ced53cc6 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -48,8 +48,9 @@ let # Type-check each function with ty. Each function exports its own 'function' # module, so checking all functions at once would let ty resolve one # function's `function.fn` import to another's package. We check each in - # isolation against a venv that provides its dependencies, plus the protobuf - # type stubs ty needs to resolve the SDK's generated Struct and Duration. + # isolation against a venv that provides its dependencies, pytest, which the + # tests import, and the protobuf type stubs ty needs to resolve the SDK's + # generated Struct and Duration. # # Unlike mkFunctionTest, which runs the function module from the venv, ty # checks the source, so we copy function/ and tests/ from the tree. We also @@ -60,6 +61,7 @@ let let venv = pythonSet.mkVirtualEnv "${name}-ty-env" { ${name} = [ ]; + pytest = [ ]; types-protobuf = [ ]; }; in diff --git a/pyproject.toml b/pyproject.toml index 1cdafcfa1..350e59cc8 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -57,6 +57,7 @@ select = [ "TID", # flake8-tidy-imports (no relative imports) "BLE", # flake8-blind-except (no bare except Exception) "ANN", # flake8-annotations (require type annotations) + "PT", # flake8-pytest-style (pytest idioms, no unittest assertions) "RUF", # ruff-specific rules (incl. unused/malformed noqa) ] @@ -75,6 +76,8 @@ allow-star-arg-any = true # Tests use magic values, many parameters, long hardcoded resource dicts, and # boolean fixture toggles passed positionally. "**/tests/**" = ["E501", "PLR2004", "PLR0913", "FBT001"] +# The pytest style rules are for tests. Elsewhere an assert guards an invariant. +"!**/tests/**" = ["PT"] # fn.py uses gRPC's required PascalCase method name. "functions/*/function/fn.py" = ["N802"] # The docs manifest validator is a CLI script; print is its output. From 0bece70bf800cff29b27dbeba2d82fc771441fa6 Mon Sep 17 00:00:00 2001 From: Nic Cope Date: Wed, 30 Sep 2026 21:38:40 -0700 Subject: [PATCH 3/7] Write each unit test case out in full The tests built their cases in several ways. Some derived a case from another by copying and mutating it, or patched a request after building it. Others built whole requests and responses with helpers, asserted on individual fields, or computed expected values with the code under test or the SDK. A reader often couldn't tell what a case checked without tracing code. Some case names also claimed conditions their input didn't set up. This commit rewrites the tests to the rules CONTRIBUTING now states, and says why in a comment wherever a test departs from them. Helpers take everything that varies as a required keyword argument, because a defaulted argument had hidden that a case named for an unpinned replica was pinned. A field-level test became a case in its entry point's table, or was deleted where a case already sent the same request, and misleading case names are corrected. Every distinct RunFunction request and response is unchanged. Writing cases out in full takes the tests from about 19,000 lines to 39,000. Towards #473. Signed-off-by: Nic Cope --- CONTRIBUTING.md | 58 +- .../compose-aks-cluster/tests/test_fn.py | 1031 ++- .../compose-eks-cluster/tests/test_fn.py | 4290 ++++++--- .../compose-gke-cluster/tests/test_fn.py | 1211 +-- .../compose-inference-class/tests/test_fn.py | 14 +- .../tests/test_fn.py | 6923 ++++++++------ .../tests/test_fn.py | 3642 ++++++-- .../compose-model-cache/tests/test_fn.py | 1812 ++-- .../tests/test_cel.py | 337 +- .../compose-model-deployment/tests/test_fn.py | 3687 ++++---- .../tests/test_quantity.py | 175 +- .../tests/test_scheduling.py | 6190 ++++++++++--- .../tests/test_semver.py | 131 +- .../compose-model-endpoint/tests/test_fn.py | 298 +- .../tests/test_backends.py | 8165 ++++++++++++++--- .../compose-model-replica/tests/test_fn.py | 1273 ++- .../compose-model-route/tests/test_fn.py | 2580 ++++-- .../compose-model-service/tests/test_fn.py | 710 +- .../compose-nebius-cluster/tests/test_fn.py | 1650 ++-- .../compose-serving-stack/tests/test_fn.py | 5429 ++++++++--- .../tests/test_stacks.py | 105 +- functions/compose-usages/tests/test_fn.py | 412 +- .../compose-vultr-cluster/tests/test_fn.py | 868 +- 23 files changed, 35458 insertions(+), 15533 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 3cc00dc18..de878e32a 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -281,8 +281,8 @@ can have its own `test_.py` too. The canonical form is a table of `Case`s, each running the function on a `RunFunctionRequest` and comparing the whole `RunFunctionResponse` against an -expected one, rather than asserting on individual fields. `compose-usages` is a -small example. The skeleton: +expected one, rather than asserting on individual fields. `compose-model-cache` +is a good example. The skeleton: ```python @dataclasses.dataclass @@ -321,19 +321,47 @@ plain functions, with no classes, fixtures, or `conftest.py`. They call the async `RunFunction` with `asyncio.run` rather than needing a plugin, and check errors with `pytest.raises(..., match=...)`. -Build the XR with `resource.dict_to_struct(xr.model_dump(exclude_none=True, -mode="json"))` from a generated Pydantic model; build other observed, desired, -and required resources as plain dicts. Because `want` is the whole response, it -must include the parts the function always emits: `meta.ttl` (60s), an empty -`context`, and any conditions, results, and requirements. Give observed -conditions a fixed `lastTransitionTime` so the input is deterministic. Protobuf -maps (`desired.resources`, `requirements.resources`) compare -order-independently, but repeated fields (`conditions`, `results`, status -arrays) must match the order the function emits. - -Some existing tests (`compose-serving-stack`, the second method in -`compose-eks-cluster`) predate this form and assert on individual fields. Don't -model new tests on them. +Cases are data, so a reader should be able to see everything a case asserts by +reading it: + +- **Write each case out in full.** Repetition between cases is fine. Don't + derive one case from another, or from a shared base, by copying and mutating + it, and don't change a request or response once it's built. Pass + requirements, conditions, and results to the constructor. +- **A resource that appears in three or more cases gets a helper,** the XR + included. Count resources by the role they play, such as "the GPU node pool" + or "an endpoint's Backend". An observed resource plays a different role from + the desired resource it reflects, so it gets its own helper. A helper builds + that one resource and returns the `fnv1.Resource` that carries it, or a dict + where another resource embeds it. Everything that varies between the cases + that use it is a keyword argument with no default, including readiness as an + `fnv1.Ready` value, so every call shows every value that varies. Write a + resource that appears in one or two cases inline. Never write a helper that + builds a whole request, response, map of resources, or case. +- **Name a case for what its input sets up** and what it expects. Put a comment + on the case as a whole directly above its `Case(`. A short comment beside a + single value can explain that value. +- **Compare the whole output, once.** A test that calls the same entry point + with different data belongs in that entry point's table as another case. +- **Write values as literals,** in requests and expectations alike, including + names the function hashes. An expectation computed by code, whether the code + under test or the SDK's `child_name`, passes whatever that code does. +- **Build the XR from its generated model,** with + `resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json", + by_alias=True))`. Write composed and observed resources as dicts in their wire + form. The generated models include schema defaults, so a model doesn't fix its + own wire form: the SDK sends only the fields a function sets, while the API + server fills in the defaults. +- **Say why where a test departs from a rule,** in a comment beside the + departure. + +Because `want` is the whole response, it must include the parts the function +always emits: `meta.ttl` (60s), an empty `context`, and any conditions, results, +and requirements. Give observed conditions a fixed `lastTransitionTime` so the +input is deterministic. Protobuf maps (`desired.resources`, +`requirements.resources`) compare order-independently, but repeated fields +(`conditions`, `results`, status arrays) must match the order the function +emits. `nix flake check` runs every function's tests. To run one function's while you work on it: diff --git a/functions/compose-aks-cluster/tests/test_fn.py b/functions/compose-aks-cluster/tests/test_fn.py index e5a28a466..b92412c18 100644 --- a/functions/compose-aks-cluster/tests/test_fn.py +++ b/functions/compose-aks-cluster/tests/test_fn.py @@ -17,6 +17,7 @@ import asyncio import dataclasses import json +from typing import Literal import pytest from crossplane.function import resource @@ -38,129 +39,167 @@ class Case: want: fnv1.RunFunctionResponse -# Names derived like the function derives them - the hash suffix depends only -# on the input names. -_CLUSTER_NAME = resource.child_name("modelplane-system", "test-cluster", "aks") -_KUBECONFIG_SECRET_NAME = resource.child_name("test-cluster", "kubeconfig") - - def _xr( - pools: list[v1alpha1.NodePool], - credentials: v1alpha1.Credentials | None = None, -) -> dict: - """An AKSCluster XR with the given node pools, as a request dict.""" - spec = v1alpha1.Spec(location="westeurope", nodePools=pools) - if credentials is not None: - spec.credentials = credentials - return v1alpha1.AKSCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", + *, + credentials: v1alpha1.Credentials | None, + fabric: Literal["None", "InfiniBand"], + zones: list[v1alpha1.Zone] | None, +) -> fnv1.Resource: + """The observed AKSCluster XR, with one gpuh100 pool.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.AKSCluster( + metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), + spec=v1alpha1.Spec( + location="westeurope", + credentials=credentials, + nodePools=[ + v1alpha1.NodePool( + name="gpuh100", + role="GPU", + vmSize="Standard_ND96isr_H100_v5", + diskSizeGb=200, + nodeCount=1, + minNodeCount=1, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + fabric=fabric, + zones=zones, + ), + ], + ), + ).model_dump(exclude_none=True, mode="json", by_alias=True) ), - spec=spec, - ).model_dump(exclude_none=True, mode="json") + ) -def _req( - pools: list[v1alpha1.NodePool], - observed_resources: dict[str, fnv1.Resource] | None = None, - credentials: v1alpha1.Credentials | None = None, -) -> fnv1.RunFunctionRequest: - return fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(pools, credentials))), - resources=observed_resources or {}, +def _desired_xr() -> fnv1.Resource: + """The desired XR, publishing its kubeconfig Secret and cache StorageClass.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + ], + "cache": {"storageClassName": "modelplane-rwx-fs"}, + }, + } ), ) -def _resource_group(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "azure.m.upbound.io/v1beta1", - "kind": "ResourceGroup", - "metadata": {"name": _CLUSTER_NAME}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": {"location": "westeurope"}, - }, - } +def _resource_group(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The desired ResourceGroup.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "azure.m.upbound.io/v1beta1", + "kind": "ResourceGroup", + "metadata": {"name": "modelplane-system-test-cluster-aks-1173e"}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": {"location": "westeurope"}, + }, + } + ), + ready=ready, + ) -def _virtual_network(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "network.azure.m.upbound.io/v1beta1", - "kind": "VirtualNetwork", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "location": "westeurope", - "addressSpace": ["10.0.0.0/16"], - "resourceGroupNameSelector": {"matchControllerRef": True}, - }, - }, - } +def _virtual_network(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The desired VirtualNetwork.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "network.azure.m.upbound.io/v1beta1", + "kind": "VirtualNetwork", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "location": "westeurope", + "addressSpace": ["10.0.0.0/16"], + "resourceGroupNameSelector": {"matchControllerRef": True}, + }, + }, + } + ), + ready=ready, + ) -def _subnet(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "network.azure.m.upbound.io/v1beta1", - "kind": "Subnet", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "addressPrefixes": ["10.0.0.0/20"], - "resourceGroupNameSelector": {"matchControllerRef": True}, - "virtualNetworkNameSelector": {"matchControllerRef": True}, - }, - }, - } +def _subnet(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The desired Subnet.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "network.azure.m.upbound.io/v1beta1", + "kind": "Subnet", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "addressPrefixes": ["10.0.0.0/20"], + "resourceGroupNameSelector": {"matchControllerRef": True}, + "virtualNetworkNameSelector": {"matchControllerRef": True}, + }, + }, + } + ), + ready=ready, + ) -def _cluster(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "containerservice.azure.m.upbound.io/v1beta1", - "kind": "KubernetesCluster", - "metadata": {"name": _CLUSTER_NAME}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "location": "westeurope", - "kubernetesVersion": "1.34", - "dnsPrefix": _CLUSTER_NAME, - "nodeResourceGroup": f"{_CLUSTER_NAME}-nodes", - "resourceGroupNameSelector": {"matchControllerRef": True}, - "identity": {"type": "SystemAssigned"}, - "defaultNodePool": { - "name": "system", - "vmSize": "Standard_D4s_v5", - "autoScalingEnabled": True, - "minCount": 1, - "maxCount": 2, - "osDiskSizeGb": 100, - "temporaryNameForRotation": "systemtmp", - "nodeLabels": {"modelplane.ai/pool": "system"}, - "vnetSubnetIdSelector": {"matchControllerRef": True}, - }, - "networkProfile": { - "networkPlugin": "azure", - "networkPluginMode": "overlay", - "podCidr": "10.244.0.0/16", - "serviceCidr": "10.96.0.0/16", - "dnsServiceIp": "10.96.0.10", +def _cluster(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The desired KubernetesCluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "containerservice.azure.m.upbound.io/v1beta1", + "kind": "KubernetesCluster", + "metadata": {"name": "modelplane-system-test-cluster-aks-1173e"}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "location": "westeurope", + "kubernetesVersion": "1.34", + "dnsPrefix": "modelplane-system-test-cluster-aks-1173e", + "nodeResourceGroup": "modelplane-system-test-cluster-aks-1173e-nodes", + "resourceGroupNameSelector": {"matchControllerRef": True}, + "identity": {"type": "SystemAssigned"}, + "defaultNodePool": { + "name": "system", + "vmSize": "Standard_D4s_v5", + "autoScalingEnabled": True, + "minCount": 1, + "maxCount": 2, + "osDiskSizeGb": 100, + "temporaryNameForRotation": "systemtmp", + "nodeLabels": {"modelplane.ai/pool": "system"}, + "vnetSubnetIdSelector": {"matchControllerRef": True}, + }, + "networkProfile": { + "networkPlugin": "azure", + "networkPluginMode": "overlay", + "podCidr": "10.244.0.0/16", + "serviceCidr": "10.96.0.0/16", + "dnsServiceIp": "10.96.0.10", + }, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, }, - }, - "writeConnectionSecretToRef": {"name": _KUBECONFIG_SECRET_NAME}, - }, - } + } + ), + ready=ready, + ) -def _nodepool_gpu( - cred_kind: str = "ClusterProviderConfig", - cred_name: str = "default", - **for_provider_extra: object, -) -> dict: - """A GPU node pool golden, merged with extra forProvider fields.""" - return { +def _nodepool_gpu(*, cred_kind: str, cred_name: str, zones: list[str] | None, ready: fnv1.Ready) -> fnv1.Resource: + """The desired gpuh100 node pool, in zones if given.""" + nodepool = { "apiVersion": "containerservice.azure.m.upbound.io/v1beta1", "kind": "KubernetesClusterNodePool", "metadata": {"annotations": {"crossplane.io/external-name": "gpuh100"}}, @@ -184,193 +223,75 @@ def _nodepool_gpu( "modelplane.ai/pool": "gpuh100", }, "nodeTaints": ["nvidia.com/gpu=true:NoSchedule"], - **for_provider_extra, }, }, } + if zones is not None: + nodepool["spec"]["forProvider"]["zones"] = zones + return fnv1.Resource(resource=resource.dict_to_struct(nodepool), ready=ready) -def _network_operator_release() -> dict: - return { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "Release", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { +def _provider_config(*, api_version: str) -> fnv1.Resource: + """A Ready ProviderConfig that reaches the cluster through its kubeconfig Secret.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": api_version, "kind": "ProviderConfig", - "name": _KUBECONFIG_SECRET_NAME, - }, - "forProvider": { - "chart": { - "name": "network-operator", - "repository": "https://helm.ngc.nvidia.com/nvidia", - "version": "26.4.0", - }, - "namespace": "network-operator", - "values": { - "deployCR": True, - "ofedDriver": {"deploy": True}, - "rdmaSharedDevicePlugin": {"deploy": True}, - # The driver and device plugin must tolerate the GPU taint - # to run on the InfiniBand nodes. - "daemonsets": { - "tolerations": [ - { - "key": "nvidia.com/gpu", - "operator": "Exists", - "effect": "NoSchedule", - }, - ], + "metadata": {"name": "test-cluster-kubeconfig-55b57"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", + "key": "kubeconfig", + }, }, }, - }, - }, - } - - -def _storage_class() -> dict: - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { - "kind": "ProviderConfig", - "name": _KUBECONFIG_SECRET_NAME, - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "storage.k8s.io/v1", - "kind": "StorageClass", - "metadata": {"name": "modelplane-rwx-fs"}, - "provisioner": "file.csi.azure.com", - "parameters": {"skuName": "Premium_LRS"}, - "mountOptions": [ - "dir_mode=0777", - "file_mode=0777", - "uid=0", - "gid=0", - "mfsymlinks", - "cache=strict", - "actimeo=30", - "nosharesock", - ], - "reclaimPolicy": "Delete", - "allowVolumeExpansion": True, - "volumeBindingMode": "WaitForFirstConsumer", - }, - }, - }, - } - - -def _provider_config(api_version: str, kind: str) -> dict: - return { - "apiVersion": api_version, - "kind": kind, - "metadata": {"name": _KUBECONFIG_SECRET_NAME}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "name": _KUBECONFIG_SECRET_NAME, - "namespace": "modelplane-system", - "key": "kubeconfig", - }, - }, - }, - } - - -def _status() -> dict: - return { - "status": { - "secrets": [ - { - "type": "Kubeconfig", - "name": _KUBECONFIG_SECRET_NAME, - "key": "kubeconfig", - }, - ], - "cache": {"storageClassName": "modelplane-rwx-fs"}, - }, - } - - -def _observed_ready(desired: dict) -> fnv1.Resource: - """An observed variant of a desired resource with a Ready=True condition.""" - observed = { - **desired, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - }, - } - return fnv1.Resource(resource=resource.dict_to_struct(observed)) - + } + ), + ready=fnv1.READY_TRUE, + ) -_GPU_POOL = v1alpha1.NodePool( - name="gpuh100", - role="GPU", - vmSize="Standard_ND96isr_H100_v5", - diskSizeGb=200, - nodeCount=1, - minNodeCount=1, - maxNodeCount=4, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), -) -_GPU_POOL_INFINIBAND = v1alpha1.NodePool( - name="gpuh100", - role="GPU", - vmSize="Standard_ND96isr_H100_v5", - diskSizeGb=200, - nodeCount=1, - minNodeCount=1, - maxNodeCount=4, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), - fabric="InfiniBand", -) +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) COMPOSE_CASES = [ + # The StorageClass isn't composed yet: the cluster isn't observed, so the + # ProviderConfigs can't reach it. Case( name="first pass composes infra; gated resources wait for the cluster", - req=_req([_GPU_POOL]), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(credentials=None, fabric="None", zones=None), + ), + ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - # The StorageClass isn't composed yet: the cluster - # isn't observed, so the ProviderConfigs can't - # reach it. - "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "resource-group": _resource_group( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "virtual-network": _virtual_network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodepool-gpuh100": _nodepool_gpu( + cred_kind="ClusterProviderConfig", cred_name="default", zones=None, ready=fnv1.READY_UNSPECIFIED + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), }, ), context=structpb.Struct(), @@ -378,45 +299,36 @@ def _observed_ready(desired: dict) -> fnv1.Resource: ), Case( name="zones pass through to the node pool", - req=_req( - [ - v1alpha1.NodePool( - name="gpuh100", - role="GPU", - vmSize="Standard_ND96isr_H100_v5", - diskSizeGb=200, - nodeCount=1, - minNodeCount=1, - maxNodeCount=4, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), - zones=[v1alpha1.Zone("1")], - ), - ] + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(credentials=None, fabric="None", zones=[v1alpha1.Zone("1")]), + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "nodepool-gpuh100": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu(zones=["1"])), + "resource-group": _resource_group( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "virtual-network": _virtual_network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodepool-gpuh100": _nodepool_gpu( + cred_kind="ClusterProviderConfig", + cred_name="default", + zones=["1"], + ready=fnv1.READY_UNSPECIFIED, ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), }, ), context=structpb.Struct(), @@ -424,44 +336,163 @@ def _observed_ready(desired: dict) -> fnv1.Resource: ), Case( name="InfiniBand pool composes the network operator once the cluster is observed", - req=_req( - [_GPU_POOL_INFINIBAND], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(credentials=None, fabric="InfiniBand", zones=None), + resources={ + "cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "containerservice.azure.m.upbound.io/v1beta1", + "kind": "KubernetesCluster", + "metadata": {"name": "modelplane-system-test-cluster-aks-1173e"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "location": "westeurope", + "kubernetesVersion": "1.34", + "dnsPrefix": "modelplane-system-test-cluster-aks-1173e", + "nodeResourceGroup": "modelplane-system-test-cluster-aks-1173e-nodes", + "resourceGroupNameSelector": {"matchControllerRef": True}, + "identity": {"type": "SystemAssigned"}, + "defaultNodePool": { + "name": "system", + "vmSize": "Standard_D4s_v5", + "autoScalingEnabled": True, + "minCount": 1, + "maxCount": 2, + "osDiskSizeGb": 100, + "temporaryNameForRotation": "systemtmp", + "nodeLabels": {"modelplane.ai/pool": "system"}, + "vnetSubnetIdSelector": {"matchControllerRef": True}, + }, + "networkProfile": { + "networkPlugin": "azure", + "networkPluginMode": "overlay", + "podCidr": "10.244.0.0/16", + "serviceCidr": "10.96.0.0/16", + "dnsServiceIp": "10.96.0.10", + }, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, + "resource-group": _resource_group( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), - "release-network-operator": fnv1.Resource( - resource=resource.dict_to_struct(_network_operator_release()), + "virtual-network": _virtual_network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - "storage-class-rwx-fs": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class()), - ready=fnv1.READY_TRUE, + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "nodepool-gpuh100": _nodepool_gpu( + cred_kind="ClusterProviderConfig", cred_name="default", zones=None, ready=fnv1.READY_UNSPECIFIED ), - "provider-config-kubernetes": fnv1.Resource( + "release-network-operator": fnv1.Resource( resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "forProvider": { + "chart": { + "name": "network-operator", + "repository": "https://helm.ngc.nvidia.com/nvidia", + "version": "26.4.0", + }, + "namespace": "network-operator", + "values": { + "deployCR": True, + "ofedDriver": {"deploy": True}, + "rdmaSharedDevicePlugin": {"deploy": True}, + # The driver and device plugin must + # tolerate the GPU taint to run on + # the InfiniBand nodes. + "daemonsets": { + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + }, + ], + }, + }, + }, + }, + } ), - ready=fnv1.READY_TRUE, ), - "provider-config-helm": fnv1.Resource( + "storage-class-rwx-fs": fnv1.Resource( resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "storage.k8s.io/v1", + "kind": "StorageClass", + "metadata": {"name": "modelplane-rwx-fs"}, + "provisioner": "file.csi.azure.com", + "parameters": {"skuName": "Premium_LRS"}, + "mountOptions": [ + "dir_mode=0777", + "file_mode=0777", + "uid=0", + "gid=0", + "mfsymlinks", + "cache=strict", + "actimeo=30", + "nosharesock", + ], + "reclaimPolicy": "Delete", + "allowVolumeExpansion": True, + "volumeBindingMode": "WaitForFirstConsumer", + }, + }, + }, + } ), ready=fnv1.READY_TRUE, ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), }, ), context=structpb.Struct(), @@ -469,135 +500,314 @@ def _observed_ready(desired: dict) -> fnv1.Resource: ), Case( name="InfiniBand pool before the cluster is observed gates the network operator", - req=_req([_GPU_POOL_INFINIBAND]), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(credentials=None, fabric="InfiniBand", zones=None), + ), + ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "resource-group": fnv1.Resource(resource=resource.dict_to_struct(_resource_group())), - "virtual-network": fnv1.Resource(resource=resource.dict_to_struct(_virtual_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "nodepool-gpuh100": fnv1.Resource(resource=resource.dict_to_struct(_nodepool_gpu())), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "resource-group": _resource_group( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "virtual-network": _virtual_network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED ), + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodepool-gpuh100": _nodepool_gpu( + cred_kind="ClusterProviderConfig", cred_name="default", zones=None, ready=fnv1.READY_UNSPECIFIED + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), }, ), context=structpb.Struct(), ), ), + # The ProviderConfigs reach the cluster through its kubeconfig, not the + # cloud credentials. Case( name="custom credentials flow through to all cloud MRs", - req=_req( - [_GPU_POOL], - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="my-azure-account", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-azure-account"), + fabric="None", + zones=None, + ), ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "resource-group": fnv1.Resource( - resource=resource.dict_to_struct(_resource_group("ProviderConfig", "my-azure-account")), - ), - "virtual-network": fnv1.Resource( - resource=resource.dict_to_struct(_virtual_network("ProviderConfig", "my-azure-account")), - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet("ProviderConfig", "my-azure-account")), + "resource-group": _resource_group( + cred_kind="ProviderConfig", cred_name="my-azure-account", ready=fnv1.READY_UNSPECIFIED ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster("ProviderConfig", "my-azure-account")), + "virtual-network": _virtual_network( + cred_kind="ProviderConfig", cred_name="my-azure-account", ready=fnv1.READY_UNSPECIFIED ), - "nodepool-gpuh100": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu("ProviderConfig", "my-azure-account")), + "subnet": _subnet( + cred_kind="ProviderConfig", cred_name="my-azure-account", ready=fnv1.READY_UNSPECIFIED ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "cluster": _cluster( + cred_kind="ProviderConfig", cred_name="my-azure-account", ready=fnv1.READY_UNSPECIFIED ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + "nodepool-gpuh100": _nodepool_gpu( + cred_kind="ProviderConfig", + cred_name="my-azure-account", + zones=None, + ready=fnv1.READY_UNSPECIFIED, ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), }, ), context=structpb.Struct(), ), ), + # The cluster is observed, so the StorageClass is composed too. Case( name="marks managed resources ready from observed conditions", - req=_req( - [_GPU_POOL], - observed_resources={ - "resource-group": _observed_ready(_resource_group()), - "virtual-network": _observed_ready(_virtual_network()), - "subnet": _observed_ready(_subnet()), - "cluster": _observed_ready(_cluster()), - "nodepool-gpuh100": _observed_ready(_nodepool_gpu()), - }, - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(credentials=None, fabric="None", zones=None), resources={ "resource-group": fnv1.Resource( - resource=resource.dict_to_struct(_resource_group()), - ready=fnv1.READY_TRUE, + resource=resource.dict_to_struct( + { + "apiVersion": "azure.m.upbound.io/v1beta1", + "kind": "ResourceGroup", + "metadata": {"name": "modelplane-system-test-cluster-aks-1173e"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": {"location": "westeurope"}, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), ), "virtual-network": fnv1.Resource( - resource=resource.dict_to_struct(_virtual_network()), - ready=fnv1.READY_TRUE, + resource=resource.dict_to_struct( + { + "apiVersion": "network.azure.m.upbound.io/v1beta1", + "kind": "VirtualNetwork", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "location": "westeurope", + "addressSpace": ["10.0.0.0/16"], + "resourceGroupNameSelector": {"matchControllerRef": True}, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), ), "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ready=fnv1.READY_TRUE, + resource=resource.dict_to_struct( + { + "apiVersion": "network.azure.m.upbound.io/v1beta1", + "kind": "Subnet", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "addressPrefixes": ["10.0.0.0/20"], + "resourceGroupNameSelector": {"matchControllerRef": True}, + "virtualNetworkNameSelector": {"matchControllerRef": True}, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), ), "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - # The cluster is observed, so the StorageClass is - # composed too. - "storage-class-rwx-fs": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class()), - ready=fnv1.READY_TRUE, + resource=resource.dict_to_struct( + { + "apiVersion": "containerservice.azure.m.upbound.io/v1beta1", + "kind": "KubernetesCluster", + "metadata": {"name": "modelplane-system-test-cluster-aks-1173e"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "location": "westeurope", + "kubernetesVersion": "1.34", + "dnsPrefix": "modelplane-system-test-cluster-aks-1173e", + "nodeResourceGroup": "modelplane-system-test-cluster-aks-1173e-nodes", + "resourceGroupNameSelector": {"matchControllerRef": True}, + "identity": {"type": "SystemAssigned"}, + "defaultNodePool": { + "name": "system", + "vmSize": "Standard_D4s_v5", + "autoScalingEnabled": True, + "minCount": 1, + "maxCount": 2, + "osDiskSizeGb": 100, + "temporaryNameForRotation": "systemtmp", + "nodeLabels": {"modelplane.ai/pool": "system"}, + "vnetSubnetIdSelector": {"matchControllerRef": True}, + }, + "networkProfile": { + "networkPlugin": "azure", + "networkPluginMode": "overlay", + "podCidr": "10.244.0.0/16", + "serviceCidr": "10.96.0.0/16", + "dnsServiceIp": "10.96.0.10", + }, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), ), "nodepool-gpuh100": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu()), - ready=fnv1.READY_TRUE, - ), - "provider-config-kubernetes": fnv1.Resource( resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + { + "apiVersion": "containerservice.azure.m.upbound.io/v1beta1", + "kind": "KubernetesClusterNodePool", + "metadata": {"annotations": {"crossplane.io/external-name": "gpuh100"}}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "managementPolicies": ["Observe", "Create", "Update", "Delete"], + "initProvider": {"nodeCount": 1}, + "forProvider": { + "kubernetesClusterIdSelector": {"matchControllerRef": True}, + "vnetSubnetIdSelector": {"matchControllerRef": True}, + "mode": "User", + "vmSize": "Standard_ND96isr_H100_v5", + "osDiskSizeGb": 200, + "orchestratorVersion": "1.34", + "autoScalingEnabled": True, + "minCount": 1, + "maxCount": 4, + "gpuDriver": "Install", + "nodeLabels": { + "modelplane.ai/gpu": "nvidia-h100", + "modelplane.ai/pool": "gpuh100", + }, + "nodeTaints": ["nvidia.com/gpu=true:NoSchedule"], + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } ), - ready=fnv1.READY_TRUE, ), - "provider-config-helm": fnv1.Resource( + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "resource-group": _resource_group( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "virtual-network": _virtual_network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "subnet": _subnet(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "cluster": _cluster(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "storage-class-rwx-fs": fnv1.Resource( resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "storage.k8s.io/v1", + "kind": "StorageClass", + "metadata": {"name": "modelplane-rwx-fs"}, + "provisioner": "file.csi.azure.com", + "parameters": {"skuName": "Premium_LRS"}, + "mountOptions": [ + "dir_mode=0777", + "file_mode=0777", + "uid=0", + "gid=0", + "mfsymlinks", + "cache=strict", + "actimeo=30", + "nosharesock", + ], + "reclaimPolicy": "Delete", + "allowVolumeExpansion": True, + "volumeBindingMode": "WaitForFirstConsumer", + }, + }, + }, + } ), ready=fnv1.READY_TRUE, ), + "nodepool-gpuh100": _nodepool_gpu( + cred_kind="ClusterProviderConfig", cred_name="default", zones=None, ready=fnv1.READY_TRUE + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), }, ), context=structpb.Struct(), @@ -606,13 +816,8 @@ def _observed_ready(desired: dict) -> fnv1.Resource: ] -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - @pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: - """The function composes AKS cluster infrastructure.""" + """RunFunction composes AKS cluster infrastructure.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-eks-cluster/tests/test_fn.py b/functions/compose-eks-cluster/tests/test_fn.py index 9658abec0..79b8fbb6d 100644 --- a/functions/compose-eks-cluster/tests/test_fn.py +++ b/functions/compose-eks-cluster/tests/test_fn.py @@ -17,7 +17,6 @@ import asyncio import dataclasses import json -from typing import Any import pytest from crossplane.function import resource @@ -39,1388 +38,1066 @@ class Case: want: fnv1.RunFunctionResponse -_KUBECONFIG_SECRET = "test-cluster-kubeconfig-55b57" -_SUBNET_A = "test-cluster-subnet-us-west-2a-952dc" -_SUBNET_B = "test-cluster-subnet-us-west-2b-2b80f" -_SUBNET_C = "test-cluster-subnet-us-west-2c-03273" -_PRIVATE_SUBNET_A = "test-cluster-private-subnet-us-west-2a-6a89f" -_PRIVATE_SUBNET_B = "test-cluster-private-subnet-us-west-2b-b7832" -_PRIVATE_SUBNET_C = "test-cluster-private-subnet-us-west-2c-ef57d" - - -def _xr(credentials: v1alpha1.Credentials | None = None) -> v1alpha1.EKSCluster: - return v1alpha1.EKSCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - region="us-west-2", - credentials=credentials, - nodePools=[ - v1alpha1.NodePool( - name="gpu-l4", - role="GPU", - instanceType="g6.xlarge", - nodeCount=1, - minNodeCount=0, - maxNodeCount=4, - gpu=v1alpha1.Gpu( - acceleratorType="nvidia-l4", - ), - zones=[v1alpha1.Zone("us-west-2a"), v1alpha1.Zone("us-west-2b")], +def _xr(*, credentials: v1alpha1.Credentials | None, node_pool: v1alpha1.NodePool) -> fnv1.Resource: + """The observed EKSCluster XR in us-west-2, with one node pool.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.EKSCluster( + metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), + spec=v1alpha1.Spec( + region="us-west-2", + credentials=credentials, + nodePools=[node_pool], ), - ], + ).model_dump(exclude_none=True, mode="json", by_alias=True) ), ) -# Launch template name is derived the same way the function derives it, so -# the test can't drift from the function's child_name hashing. -_LAUNCH_TEMPLATE_NAME = resource.child_name("test-cluster", "lt-gpu-h200") -_CAPACITY_RESERVATION_ID = "cr-0123456789abcdef0" - - -def _xr_capacity_block() -> v1alpha1.EKSCluster: - return v1alpha1.EKSCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - region="us-west-2", - nodePools=[ - v1alpha1.NodePool( - name="gpu-h200", - role="GPU", - instanceType="p5en.48xlarge", - nodeCount=2, - minNodeCount=0, - maxNodeCount=2, - diskSizeGb=1024, - gpu=v1alpha1.Gpu( - acceleratorType="nvidia-h200", - ), - capacityBlock=v1alpha1.CapacityBlock( - capacityReservationId=_CAPACITY_RESERVATION_ID, - ), - zones=[v1alpha1.Zone("us-west-2a")], - ), - ], +def _desired_xr() -> fnv1.Resource: + """The desired XR, publishing its kubeconfig Secret and cache StorageClass.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + # `type` is emitted because the function sets it explicitly + # on the Status model, so update_status (exclude_unset) + # keeps it rather than dropping it as an unset field. + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + ], + # write_status always publishes the effective RWX + # StorageClass name, even before the managed class + # materialises on the workload cluster. + "cache": {"storageClassName": "modelplane-rwx-efs"}, + }, + } ), ) -def _launch_template(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "LaunchTemplate", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "name": _LAUNCH_TEMPLATE_NAME, - "instanceType": "p5en.48xlarge", - "blockDeviceMappings": [ - {"deviceName": "/dev/xvda", "ebs": {"volumeSize": 1024}}, - ], - "instanceMarketOptions": {"marketType": "capacity-block"}, - "capacityReservationSpecification": { - "capacityReservationPreference": "capacity-reservations-only", - "capacityReservationTarget": { - "capacityReservationId": _CAPACITY_RESERVATION_ID, +def _vpc(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The VPC.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "VPC", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "cidrBlock": "10.0.0.0/16", + "enableDnsHostnames": True, + "enableDnsSupport": True, }, }, - }, - }, - } - - -def _gpu_node_group_capacity_block(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "NodeGroup", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "managementPolicies": ["Observe", "Create", "Update", "Delete"], - "initProvider": {"scalingConfig": {"desiredSize": 2}}, - "forProvider": { - "region": "us-west-2", - "amiType": "AL2023_x86_64_NVIDIA", - "clusterNameSelector": {"matchControllerRef": True}, - "nodeRoleArnSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "node"}, + } + ), + ) + + +def _subnet(*, name: str, az: str, cidr: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """A public subnet in az.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "Subnet", + "metadata": { + "name": name, + "labels": {"modelplane.ai/zone": az, "modelplane.ai/subnet-tier": "public"}, }, - "capacityType": "CAPACITY_BLOCK", - "launchTemplate": { - "name": _LAUNCH_TEMPLATE_NAME, - "version": "$Latest", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "availabilityZone": az, + "cidrBlock": cidr, + "mapPublicIpOnLaunch": True, + "tags": {"kubernetes.io/role/elb": "1"}, + "vpcIdSelector": {"matchControllerRef": True}, + }, }, - "subnetIdRefs": [{"name": _PRIVATE_SUBNET_A}], - "scalingConfig": {"minSize": 0, "maxSize": 2}, - "labels": { - "modelplane.ai/gpu": "nvidia-h200", - "modelplane.ai/pool": "gpu-h200", + } + ), + ) + + +def _private_subnet(*, name: str, az: str, cidr: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """A private subnet in az.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "Subnet", + "metadata": { + "name": name, + "labels": {"modelplane.ai/zone": az, "modelplane.ai/subnet-tier": "private"}, }, - "taint": [ - { - "key": "nvidia.com/gpu", - "value": "true", - "effect": "NO_SCHEDULE", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "availabilityZone": az, + "cidrBlock": cidr, + "mapPublicIpOnLaunch": False, + "vpcIdSelector": {"matchControllerRef": True}, }, - ], - }, - }, - } + }, + } + ), + ) -# EFA launch template and security-group object names, derived the same way the -# function derives them, so the test can't drift from the child_name hashing. -_EFA_LAUNCH_TEMPLATE_NAME = resource.child_name("test-cluster", "lt-gpu-h200") -_EFA_SECURITY_GROUP_NAME = resource.child_name("test-cluster", "efa-sg") +def _internet_gateway(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The VPC's internet gateway.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "InternetGateway", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "vpcIdSelector": {"matchControllerRef": True}, + }, + }, + } + ), + ) -def _xr_efa() -> v1alpha1.EKSCluster: - return v1alpha1.EKSCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - region="us-west-2", - nodePools=[ - v1alpha1.NodePool( - name="gpu-h200", - role="GPU", - instanceType="p5en.48xlarge", - nodeCount=2, - minNodeCount=0, - maxNodeCount=2, - diskSizeGb=1024, - gpu=v1alpha1.Gpu( - acceleratorType="nvidia-h200", - ), - fabric="EFA", - zones=[v1alpha1.Zone("us-west-2a")], - ), - ], +def _route_table(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The public route table.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "RouteTable", + "metadata": {"labels": {"modelplane.ai/subnet-tier": "public"}}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "vpcIdSelector": {"matchControllerRef": True}, + }, + }, + } ), ) -def _efa_network_interface(card: int, security_groups: list[str] | None = None) -> dict: - ni: dict[str, Any] = { - "networkCardIndex": card, - "deviceIndex": 0 if card == 0 else 1, - "interfaceType": "efa" if card == 0 else "efa-only", - } - if security_groups: - ni["securityGroups"] = security_groups - return ni +def _route_default(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The public route table's default route, through the internet gateway.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "Route", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "destinationCidrBlock": "0.0.0.0/0", + "gatewayIdSelector": {"matchControllerRef": True}, + "routeTableIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/subnet-tier": "public"}, + }, + }, + }, + } + ), + ) -def _launch_template_efa(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - # p5en.48xlarge has 16 network cards; one EFA interface per card. - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "LaunchTemplate", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "name": _EFA_LAUNCH_TEMPLATE_NAME, - "instanceType": "p5en.48xlarge", - "blockDeviceMappings": [ - {"deviceName": "/dev/xvda", "ebs": {"volumeSize": 1024}}, - ], - "networkInterfaces": [_efa_network_interface(card) for card in range(16)], - }, - }, - } - - -def _efa_security_group(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "SecurityGroup", - "metadata": { - "name": _EFA_SECURITY_GROUP_NAME, - "labels": {"modelplane.ai/fabric": "EFA"}, - }, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "name": "test-cluster-efa", - "description": "EFA OS-bypass traffic between gang nodes", - "vpcIdSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _efa_security_group_ingress(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "SecurityGroupIngressRule", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "ipProtocol": "-1", - "referencedSecurityGroupIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/fabric": "EFA"}, - }, - "securityGroupIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/fabric": "EFA"}, - }, - }, - }, - } - - -def _efa_security_group_egress(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "SecurityGroupEgressRule", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "ipProtocol": "-1", - "referencedSecurityGroupIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/fabric": "EFA"}, - }, - "securityGroupIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/fabric": "EFA"}, - }, - }, - }, - } - - -def _gpu_node_group_efa(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "NodeGroup", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "managementPolicies": ["Observe", "Create", "Update", "Delete"], - "initProvider": {"scalingConfig": {"desiredSize": 2}}, - "forProvider": { - "region": "us-west-2", - "amiType": "AL2023_x86_64_NVIDIA", - "clusterNameSelector": {"matchControllerRef": True}, - "nodeRoleArnSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "node"}, - }, - "launchTemplate": { - "name": _EFA_LAUNCH_TEMPLATE_NAME, - "version": "$Latest", - }, - "subnetIdRefs": [{"name": _PRIVATE_SUBNET_A}], - "scalingConfig": {"minSize": 0, "maxSize": 2}, - "labels": { - "modelplane.ai/gpu": "nvidia-h200", - "modelplane.ai/pool": "gpu-h200", - }, - "taint": [ - { - "key": "nvidia.com/gpu", - "value": "true", - "effect": "NO_SCHEDULE", +def _nat_eip(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The NAT gateway's elastic IP.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "EIP", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "domain": "vpc", }, - ], - }, - }, - } - - -def _ready_condition() -> dict: - return { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - } - - -def _vpc(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "VPC", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "cidrBlock": "10.0.0.0/16", - "enableDnsHostnames": True, - "enableDnsSupport": True, - }, - }, - } - - -def _subnet( - name: str, az: str, cidr: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default" -) -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "Subnet", - "metadata": { - "name": name, - "labels": {"modelplane.ai/zone": az, "modelplane.ai/subnet-tier": "public"}, - }, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "availabilityZone": az, - "cidrBlock": cidr, - "mapPublicIpOnLaunch": True, - "tags": {"kubernetes.io/role/elb": "1"}, - "vpcIdSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _private_subnet( - name: str, az: str, cidr: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default" -) -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "Subnet", - "metadata": { - "name": name, - "labels": {"modelplane.ai/zone": az, "modelplane.ai/subnet-tier": "private"}, - }, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "availabilityZone": az, - "cidrBlock": cidr, - "mapPublicIpOnLaunch": False, - "vpcIdSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _internet_gateway(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "InternetGateway", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "vpcIdSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _route_table(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "RouteTable", - "metadata": {"labels": {"modelplane.ai/subnet-tier": "public"}}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "vpcIdSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _route_default(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "Route", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "destinationCidrBlock": "0.0.0.0/0", - "gatewayIdSelector": {"matchControllerRef": True}, - "routeTableIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/subnet-tier": "public"}, }, - }, - }, - } - - -def _nat_eip(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "EIP", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "domain": "vpc", - }, - }, - } - - -def _nat_gateway(az: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "NATGateway", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "allocationIdSelector": {"matchControllerRef": True}, - "subnetIdSelector": { - "matchControllerRef": True, - "matchLabels": { - "modelplane.ai/zone": az, - "modelplane.ai/subnet-tier": "public", + } + ), + ) + + +def _nat_gateway(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The NAT gateway, in the first AZ's public subnet.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "NATGateway", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "allocationIdSelector": {"matchControllerRef": True}, + "subnetIdSelector": { + "matchControllerRef": True, + "matchLabels": { + "modelplane.ai/zone": "us-west-2a", + "modelplane.ai/subnet-tier": "public", + }, + }, }, }, - }, - }, - } - - -def _private_route_table(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "RouteTable", - "metadata": {"labels": {"modelplane.ai/subnet-tier": "private"}}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "vpcIdSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _private_route_default(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "Route", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "destinationCidrBlock": "0.0.0.0/0", - "natGatewayIdSelector": {"matchControllerRef": True}, - "routeTableIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/subnet-tier": "private"}, - }, - }, - }, - } - - -def _route_table_association(az: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "RouteTableAssociation", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "routeTableIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/subnet-tier": "public"}, - }, - "subnetIdSelector": { - "matchControllerRef": True, - "matchLabels": { - "modelplane.ai/zone": az, - "modelplane.ai/subnet-tier": "public", + } + ), + ) + + +def _private_route_table(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The private route table.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "RouteTable", + "metadata": {"labels": {"modelplane.ai/subnet-tier": "private"}}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "vpcIdSelector": {"matchControllerRef": True}, }, }, - }, - }, - } - - -def _private_route_table_association( - az: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default" -) -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "RouteTableAssociation", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "routeTableIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/subnet-tier": "private"}, - }, - "subnetIdSelector": { - "matchControllerRef": True, - "matchLabels": { - "modelplane.ai/zone": az, - "modelplane.ai/subnet-tier": "private", + } + ), + ) + + +def _private_route_default(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The private route table's default route, through the NAT gateway.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "Route", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "destinationCidrBlock": "0.0.0.0/0", + "natGatewayIdSelector": {"matchControllerRef": True}, + "routeTableIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/subnet-tier": "private"}, + }, }, }, - }, - }, - } - - -def _role(role: str, assume_policy: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "iam.aws.m.upbound.io/v1beta1", - "kind": "Role", - "metadata": {"labels": {"modelplane.ai/iam-role": role}}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": {"assumeRolePolicy": assume_policy}, - }, - } - - -def _role_policy_attachment( - role: str, arn: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default" -) -> dict: - return { - "apiVersion": "iam.aws.m.upbound.io/v1beta1", - "kind": "RolePolicyAttachment", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "policyArn": arn, - "roleSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": role}, - }, - }, - }, - } - - -_ASSUME_CLUSTER = ( - '{"Version":"2012-10-17","Statement":[{"Effect":"Allow",' - '"Principal":{"Service":"eks.amazonaws.com"},' - '"Action":"sts:AssumeRole"}]}' -) -_ASSUME_NODE = ( - '{"Version":"2012-10-17","Statement":[{"Effect":"Allow",' - '"Principal":{"Service":"ec2.amazonaws.com"},' - '"Action":"sts:AssumeRole"}]}' -) -_ASSUME_POD_IDENTITY = ( - '{"Version":"2012-10-17","Statement":[{"Effect":"Allow",' - '"Principal":{"Service":"pods.eks.amazonaws.com"},' - '"Action":["sts:AssumeRole","sts:TagSession"]}]}' -) -_POLICY_EFS_CSI = "arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy" -_POLICY_CLUSTER_AUTOSCALER = ( - '{"Version":"2012-10-17","Statement":[' - '{"Effect":"Allow","Action":[' - '"autoscaling:DescribeAutoScalingGroups",' - '"autoscaling:DescribeAutoScalingInstances",' - '"autoscaling:DescribeLaunchConfigurations",' - '"autoscaling:DescribeScalingActivities",' - '"ec2:DescribeImages",' - '"ec2:DescribeInstanceTypes",' - '"ec2:DescribeLaunchTemplateVersions",' - '"ec2:GetInstanceTypesFromInstanceRequirements",' - '"eks:DescribeNodegroup"' - '],"Resource":["*"]},' - '{"Effect":"Allow","Action":[' - '"autoscaling:SetDesiredCapacity",' - '"autoscaling:TerminateInstanceInAutoScalingGroup"' - '],"Resource":["*"]}]}' -) - - -def _eks_cluster(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "Cluster", - "metadata": {"name": "modelplane-system-test-cluster-eks-0865f"}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "version": "1.36", - "roleArnSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "cluster"}, - }, - "accessConfig": { - "authenticationMode": "API_AND_CONFIG_MAP", - "bootstrapClusterCreatorAdminPermissions": True, - }, - "vpcConfig": { - "endpointPrivateAccess": True, - "endpointPublicAccess": True, - "subnetIdSelector": {"matchControllerRef": True}, - }, - }, - }, - } - - -def _cluster_auth(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "ClusterAuth", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "clusterNameSelector": {"matchControllerRef": True}, - }, - "writeConnectionSecretToRef": {"name": _KUBECONFIG_SECRET}, - }, - } - - -def _system_node_group(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "NodeGroup", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "managementPolicies": ["Observe", "Create", "Update", "Delete"], - "initProvider": {"scalingConfig": {"desiredSize": 1}}, - "forProvider": { - "region": "us-west-2", - "amiType": "AL2023_x86_64_STANDARD", - "instanceTypes": ["m6i.xlarge"], - "clusterNameSelector": {"matchControllerRef": True}, - "nodeRoleArnSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "node"}, + } + ), + ) + + +def _route_table_association(*, az: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The association of az's public subnet with the public route table.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "RouteTableAssociation", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "routeTableIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/subnet-tier": "public"}, + }, + "subnetIdSelector": { + "matchControllerRef": True, + "matchLabels": { + "modelplane.ai/zone": az, + "modelplane.ai/subnet-tier": "public", + }, + }, + }, }, - "subnetIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/subnet-tier": "private"}, + } + ), + ) + + +def _private_route_table_association(*, az: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The association of az's private subnet with the private route table.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "RouteTableAssociation", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "routeTableIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/subnet-tier": "private"}, + }, + "subnetIdSelector": { + "matchControllerRef": True, + "matchLabels": { + "modelplane.ai/zone": az, + "modelplane.ai/subnet-tier": "private", + }, + }, + }, }, - "scalingConfig": {"minSize": 1, "maxSize": 2}, - "labels": {"modelplane.ai/pool": "system"}, - }, - }, - } - - -def _gpu_node_group(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "NodeGroup", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "managementPolicies": ["Observe", "Create", "Update", "Delete"], - "initProvider": {"scalingConfig": {"desiredSize": 1}}, - "forProvider": { - "region": "us-west-2", - "amiType": "AL2023_x86_64_NVIDIA", - "instanceTypes": ["g6.xlarge"], - "diskSize": 100, - "clusterNameSelector": {"matchControllerRef": True}, - "nodeRoleArnSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "node"}, + } + ), + ) + + +def _cluster_role(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The IAM role the EKS control plane assumes.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "iam.aws.m.upbound.io/v1beta1", + "kind": "Role", + "metadata": {"labels": {"modelplane.ai/iam-role": "cluster"}}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "assumeRolePolicy": ( + '{"Version":"2012-10-17","Statement":[{"Effect":"Allow",' + '"Principal":{"Service":"eks.amazonaws.com"},' + '"Action":"sts:AssumeRole"}]}' + ), + }, }, - "subnetIdRefs": [{"name": _PRIVATE_SUBNET_A}, {"name": _PRIVATE_SUBNET_B}], - "scalingConfig": {"minSize": 0, "maxSize": 4}, - "labels": { - "modelplane.ai/gpu": "nvidia-l4", - "modelplane.ai/pool": "gpu-l4", + } + ), + ) + + +def _node_role(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The IAM role the nodes assume.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "iam.aws.m.upbound.io/v1beta1", + "kind": "Role", + "metadata": {"labels": {"modelplane.ai/iam-role": "node"}}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "assumeRolePolicy": ( + '{"Version":"2012-10-17","Statement":[{"Effect":"Allow",' + '"Principal":{"Service":"ec2.amazonaws.com"},' + '"Action":"sts:AssumeRole"}]}' + ), + }, }, - "taint": [ - { - "key": "nvidia.com/gpu", - "value": "true", - "effect": "NO_SCHEDULE", + } + ), + ) + + +def _pod_identity_role(*, role: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """An IAM role a ServiceAccount assumes through EKS Pod Identity.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "iam.aws.m.upbound.io/v1beta1", + "kind": "Role", + "metadata": {"labels": {"modelplane.ai/iam-role": role}}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "assumeRolePolicy": ( + '{"Version":"2012-10-17","Statement":[{"Effect":"Allow",' + '"Principal":{"Service":"pods.eks.amazonaws.com"},' + '"Action":["sts:AssumeRole","sts:TagSession"]}]}' + ), }, - ], - }, - }, - } - - -def _addon(name: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "Addon", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "addonName": name, - "clusterNameSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _efs_filesystem(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "efs.aws.m.upbound.io/v1beta1", - "kind": "FileSystem", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": {"region": "us-west-2", "throughputMode": "elastic", "encrypted": True}, - }, - } - - -def _efs_security_group(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "SecurityGroup", - "metadata": {"labels": {"modelplane.ai/sg-role": "efs"}}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "name": "test-cluster-efs", - "description": "NFS access to the ModelCache EFS mount targets", - "vpcIdSelector": {"matchControllerRef": True}, - }, - }, - } - - -def _efs_security_group_ingress(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "ec2.aws.m.upbound.io/v1beta1", - "kind": "SecurityGroupIngressRule", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "ipProtocol": "tcp", - "fromPort": 2049, - "toPort": 2049, - "cidrIpv4": "10.0.0.0/16", - "securityGroupIdSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/sg-role": "efs"}, }, - }, - }, - } - - -def _efs_mount_target(subnet_name: str, cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "efs.aws.m.upbound.io/v1beta1", - "kind": "MountTarget", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "fileSystemIdSelector": {"matchControllerRef": True}, - "subnetIdRef": {"name": subnet_name}, - "securityGroupsSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/sg-role": "efs"}, + } + ), + ) + + +def _role_policy_attachment(*, role: str, arn: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """An attachment of the managed policy arn to the role labelled role.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "iam.aws.m.upbound.io/v1beta1", + "kind": "RolePolicyAttachment", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "policyArn": arn, + "roleSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": role}, + }, + }, }, - }, - }, - } - - -def _pod_identity_association(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "PodIdentityAssociation", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "namespace": "kube-system", - "serviceAccount": "efs-csi-controller-sa", - "clusterNameSelector": {"matchControllerRef": True}, - "roleArnSelector": {"matchControllerRef": True, "matchLabels": {"modelplane.ai/iam-role": "efs-csi"}}, - }, - }, - } - - -def _storage_class_object(filesystem_id: str) -> dict: - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { - "kind": "ProviderConfig", - "name": _KUBECONFIG_SECRET, - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "storage.k8s.io/v1", - "kind": "StorageClass", - "metadata": {"name": "modelplane-rwx-efs"}, - "provisioner": "efs.csi.aws.com", - "parameters": { - "provisioningMode": "efs-ap", - "fileSystemId": filesystem_id, - "directoryPerms": "700", + } + ), + ) + + +def _eks_cluster(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The desired EKS Cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "Cluster", + "metadata": {"name": "modelplane-system-test-cluster-eks-0865f"}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "version": "1.36", + "roleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "cluster"}, + }, + "accessConfig": { + "authenticationMode": "API_AND_CONFIG_MAP", + "bootstrapClusterCreatorAdminPermissions": True, + }, + "vpcConfig": { + "endpointPrivateAccess": True, + "endpointPublicAccess": True, + "subnetIdSelector": {"matchControllerRef": True}, + }, }, - "volumeBindingMode": "Immediate", }, + } + ), + ready=ready, + ) + + +def _observed_eks_cluster(*, cluster_security_group_id: str | None, ready: bool) -> fnv1.Resource: + """The observed EKS Cluster, with a Ready condition if ready and its security group id if given.""" + status: dict = {} + if cluster_security_group_id is not None: + status["atProvider"] = {"vpcConfig": {"clusterSecurityGroupId": cluster_security_group_id}} + if ready: + status["conditions"] = [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", }, - }, - } - - -def _autoscaler_policy(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "iam.aws.m.upbound.io/v1beta1", - "kind": "Policy", - "metadata": {"labels": {"modelplane.ai/iam-role": "cluster-autoscaler"}}, - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": {"policy": _POLICY_CLUSTER_AUTOSCALER}, - }, - } - - -def _autoscaler_attachment(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "iam.aws.m.upbound.io/v1beta1", - "kind": "RolePolicyAttachment", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "policyArnSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "cluster-autoscaler"}, + ] + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "Cluster", + "metadata": {"name": "modelplane-system-test-cluster-eks-0865f"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "version": "1.36", + "roleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "cluster"}, + }, + "accessConfig": { + "authenticationMode": "API_AND_CONFIG_MAP", + "bootstrapClusterCreatorAdminPermissions": True, + }, + "vpcConfig": { + "endpointPrivateAccess": True, + "endpointPublicAccess": True, + "subnetIdSelector": {"matchControllerRef": True}, + }, + }, }, - "roleSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "cluster-autoscaler"}, + "status": status, + } + ), + ) + + +def _cluster_auth(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The desired ClusterAuth.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "ClusterAuth", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "clusterNameSelector": {"matchControllerRef": True}, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, }, - }, - }, - } - - -def _autoscaler_pod_identity(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "eks.aws.m.upbound.io/v1beta1", - "kind": "PodIdentityAssociation", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-west-2", - "namespace": "kube-system", - "serviceAccount": "cluster-autoscaler", - "clusterNameSelector": {"matchControllerRef": True}, - "roleArnSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/iam-role": "cluster-autoscaler"}, + } + ), + ready=ready, + ) + + +def _system_node_group(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The system node group every cluster gets.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "managementPolicies": ["Observe", "Create", "Update", "Delete"], + "initProvider": {"scalingConfig": {"desiredSize": 1}}, + "forProvider": { + "region": "us-west-2", + "amiType": "AL2023_x86_64_STANDARD", + "instanceTypes": ["m6i.xlarge"], + "clusterNameSelector": {"matchControllerRef": True}, + "nodeRoleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "node"}, + }, + "subnetIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/subnet-tier": "private"}, + }, + "scalingConfig": {"minSize": 1, "maxSize": 2}, + "labels": {"modelplane.ai/pool": "system"}, + }, }, - }, - }, - } - - -def _autoscaler_release() -> dict: - return { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "Release", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": {"kind": "ProviderConfig", "name": _KUBECONFIG_SECRET}, - "forProvider": { - "chart": { - "name": "cluster-autoscaler", - "repository": "https://kubernetes.github.io/autoscaler", - "version": "9.57.0", + } + ), + ) + + +def _gpu_node_group(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The gpu-l4 node group, which needs no launch template.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "managementPolicies": ["Observe", "Create", "Update", "Delete"], + "initProvider": {"scalingConfig": {"desiredSize": 1}}, + "forProvider": { + "region": "us-west-2", + "amiType": "AL2023_x86_64_NVIDIA", + "instanceTypes": ["g6.xlarge"], + "diskSize": 100, + "clusterNameSelector": {"matchControllerRef": True}, + "nodeRoleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "node"}, + }, + "subnetIdRefs": [ + {"name": "test-cluster-private-subnet-us-west-2a-6a89f"}, + {"name": "test-cluster-private-subnet-us-west-2b-b7832"}, + ], + "scalingConfig": {"minSize": 0, "maxSize": 4}, + "labels": { + "modelplane.ai/gpu": "nvidia-l4", + "modelplane.ai/pool": "gpu-l4", + }, + "taint": [ + { + "key": "nvidia.com/gpu", + "value": "true", + "effect": "NO_SCHEDULE", + }, + ], + }, }, - "namespace": "kube-system", - "values": { - "cloudProvider": "aws", - "awsRegion": "us-west-2", - "autoDiscovery": {"clusterName": "modelplane-system-test-cluster-eks-0865f"}, - "rbac": {"serviceAccount": {"name": "cluster-autoscaler"}}, - "extraArgs": {"balance-similar-node-groups": True}, + } + ), + ) + + +def _launch_template_efa(*, security_groups: list[str] | None) -> fnv1.Resource: + """The gpu-h200 EFA launch template, with security_groups on every interface if given.""" + # p5en.48xlarge has 16 network cards; one EFA interface per card. + interfaces: list[dict] = [ + {"networkCardIndex": 0, "deviceIndex": 0, "interfaceType": "efa"}, + {"networkCardIndex": 1, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 2, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 3, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 4, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 5, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 6, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 7, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 8, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 9, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 10, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 11, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 12, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 13, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 14, "deviceIndex": 1, "interfaceType": "efa-only"}, + {"networkCardIndex": 15, "deviceIndex": 1, "interfaceType": "efa-only"}, + ] + if security_groups is not None: + for interface in interfaces: + interface["securityGroups"] = security_groups + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "LaunchTemplate", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "name": "test-cluster-lt-gpu-h200-83c00", + "instanceType": "p5en.48xlarge", + "blockDeviceMappings": [ + {"deviceName": "/dev/xvda", "ebs": {"volumeSize": 1024}}, + ], + "networkInterfaces": interfaces, + }, }, - }, - }, - } - - -def _efa_dra_driver_release() -> dict: - return { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "Release", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": {"kind": "ProviderConfig", "name": _KUBECONFIG_SECRET}, - "forProvider": { - "chart": { - "name": "aws-dranet", - "repository": "https://aws.github.io/eks-charts", - "version": "1.0.0", + } + ), + ) + + +def _efa_security_group() -> fnv1.Resource: + """The desired EFA security group.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "SecurityGroup", + "metadata": { + "name": "test-cluster-efa-sg-602a9", + "labels": {"modelplane.ai/fabric": "EFA"}, }, - "namespace": "kube-system", - "values": { - "tolerations": [ - {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}, - ], + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "name": "test-cluster-efa", + "description": "EFA OS-bypass traffic between gang nodes", + "vpcIdSelector": {"matchControllerRef": True}, + }, }, - }, - }, - } - - -def _provider_config(api_version: str) -> dict: - return { - "apiVersion": api_version, - "kind": "ProviderConfig", - "metadata": {"name": _KUBECONFIG_SECRET}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "name": _KUBECONFIG_SECRET, - "namespace": "modelplane-system", - "key": "kubeconfig", + } + ), + ) + + +def _efa_security_group_ingress() -> fnv1.Resource: + """The EFA security group's rule admitting all traffic from itself.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "SecurityGroupIngressRule", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "ipProtocol": "-1", + "referencedSecurityGroupIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/fabric": "EFA"}, + }, + "securityGroupIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/fabric": "EFA"}, + }, + }, }, - }, - }, - } + } + ), + ) -def _expected_status() -> dict: - # `type` is emitted because the function sets it explicitly on the Status - # model, so update_status (exclude_unset) keeps it rather than dropping it - # as an unset field. - status = { - "secrets": [ +def _efa_security_group_egress() -> fnv1.Resource: + """The EFA security group's rule allowing all traffic to itself.""" + return fnv1.Resource( + resource=resource.dict_to_struct( { - "type": "Kubeconfig", - "name": _KUBECONFIG_SECRET, - "key": "kubeconfig", - }, - ], - # write_status always publishes the effective RWX StorageClass name, - # even before the managed class materialises on the workload cluster. - "cache": {"storageClassName": "modelplane-rwx-efs"}, - } - return {"status": status} - - -def _expected_resources() -> dict: - return { - "vpc": fnv1.Resource(resource=resource.dict_to_struct(_vpc())), - "subnet-0": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_A, "us-west-2a", "10.0.0.0/20")), + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "SecurityGroupEgressRule", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "ipProtocol": "-1", + "referencedSecurityGroupIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/fabric": "EFA"}, + }, + "securityGroupIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/fabric": "EFA"}, + }, + }, + }, + } ), - "subnet-1": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_B, "us-west-2b", "10.0.16.0/20")), + ) + + +def _gpu_node_group_efa() -> fnv1.Resource: + """The gpu-h200 EFA node group, which takes its instance type from the launch template.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "managementPolicies": ["Observe", "Create", "Update", "Delete"], + "initProvider": {"scalingConfig": {"desiredSize": 2}}, + "forProvider": { + "region": "us-west-2", + "amiType": "AL2023_x86_64_NVIDIA", + "clusterNameSelector": {"matchControllerRef": True}, + "nodeRoleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "node"}, + }, + "launchTemplate": { + "name": "test-cluster-lt-gpu-h200-83c00", + "version": "$Latest", + }, + "subnetIdRefs": [{"name": "test-cluster-private-subnet-us-west-2a-6a89f"}], + "scalingConfig": {"minSize": 0, "maxSize": 2}, + "labels": { + "modelplane.ai/gpu": "nvidia-h200", + "modelplane.ai/pool": "gpu-h200", + }, + "taint": [ + { + "key": "nvidia.com/gpu", + "value": "true", + "effect": "NO_SCHEDULE", + }, + ], + }, + }, + } ), - "subnet-2": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_C, "us-west-2c", "10.0.32.0/20")), + ) + + +def _addon(*, name: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The EKS addon called name.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "Addon", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "addonName": name, + "clusterNameSelector": {"matchControllerRef": True}, + }, + }, + } ), - "private-subnet-0": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_A, "us-west-2a", "10.0.48.0/20")), + ) + + +def _efs_filesystem(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The desired EFS filesystem backing ModelCache.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "efs.aws.m.upbound.io/v1beta1", + "kind": "FileSystem", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": {"region": "us-west-2", "throughputMode": "elastic", "encrypted": True}, + }, + } ), - "private-subnet-1": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_B, "us-west-2b", "10.0.64.0/20")), + ready=ready, + ) + + +def _efs_security_group(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The security group on the EFS mount targets.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "SecurityGroup", + "metadata": {"labels": {"modelplane.ai/sg-role": "efs"}}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "name": "test-cluster-efs", + "description": "NFS access to the ModelCache EFS mount targets", + "vpcIdSelector": {"matchControllerRef": True}, + }, + }, + } ), - "private-subnet-2": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_C, "us-west-2c", "10.0.80.0/20")), - ), - "internet-gateway": fnv1.Resource(resource=resource.dict_to_struct(_internet_gateway())), - "nat-eip": fnv1.Resource(resource=resource.dict_to_struct(_nat_eip())), - "nat-gateway": fnv1.Resource(resource=resource.dict_to_struct(_nat_gateway("us-west-2a"))), - "route-table": fnv1.Resource(resource=resource.dict_to_struct(_route_table())), - "route-default": fnv1.Resource(resource=resource.dict_to_struct(_route_default())), - "private-route-table": fnv1.Resource(resource=resource.dict_to_struct(_private_route_table())), - "private-route-default": fnv1.Resource(resource=resource.dict_to_struct(_private_route_default())), - "route-table-association-0": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2a")), - ), - "route-table-association-1": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2b")), - ), - "route-table-association-2": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2c")), - ), - "private-route-table-association-0": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2a")), - ), - "private-route-table-association-1": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2b")), - ), - "private-route-table-association-2": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2c")), - ), - "iam-role-cluster": fnv1.Resource( - resource=resource.dict_to_struct(_role("cluster", _ASSUME_CLUSTER)), - ), - "iam-attach-cluster-policy": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("cluster", "arn:aws:iam::aws:policy/AmazonEKSClusterPolicy"), - ), - ), - "iam-role-node": fnv1.Resource(resource=resource.dict_to_struct(_role("node", _ASSUME_NODE))), - "iam-attach-node-worker": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy"), - ), - ), - "iam-attach-node-cni": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy"), - ), - ), - "iam-attach-node-ecr": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly"), - ), - ), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_eks_cluster())), - "cluster-auth": fnv1.Resource(resource=resource.dict_to_struct(_cluster_auth())), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_system_node_group())), - "nodegroup-gpu-l4": fnv1.Resource(resource=resource.dict_to_struct(_gpu_node_group())), - "addon-vpc-cni": fnv1.Resource(resource=resource.dict_to_struct(_addon("vpc-cni"))), - "addon-kube-proxy": fnv1.Resource(resource=resource.dict_to_struct(_addon("kube-proxy"))), - "addon-coredns": fnv1.Resource(resource=resource.dict_to_struct(_addon("coredns"))), - "efs-filesystem": fnv1.Resource(resource=resource.dict_to_struct(_efs_filesystem())), - "efs-security-group": fnv1.Resource(resource=resource.dict_to_struct(_efs_security_group())), - "efs-security-group-ingress": fnv1.Resource(resource=resource.dict_to_struct(_efs_security_group_ingress())), - "efs-mount-target-0": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_A))), - "efs-mount-target-1": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_B))), - "efs-mount-target-2": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_C))), - "iam-role-efs-csi": fnv1.Resource(resource=resource.dict_to_struct(_role("efs-csi", _ASSUME_POD_IDENTITY))), - "iam-attach-efs-csi": fnv1.Resource( - resource=resource.dict_to_struct(_role_policy_attachment("efs-csi", _POLICY_EFS_CSI)), - ), - "addon-eks-pod-identity-agent": fnv1.Resource( - resource=resource.dict_to_struct(_addon("eks-pod-identity-agent")), - ), - "pod-identity-efs-csi": fnv1.Resource(resource=resource.dict_to_struct(_pod_identity_association())), - "addon-aws-efs-csi-driver": fnv1.Resource(resource=resource.dict_to_struct(_addon("aws-efs-csi-driver"))), - "iam-policy-cluster-autoscaler": fnv1.Resource(resource=resource.dict_to_struct(_autoscaler_policy())), - "iam-role-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_role("cluster-autoscaler", _ASSUME_POD_IDENTITY)), - ), - "iam-attach-cluster-autoscaler": fnv1.Resource(resource=resource.dict_to_struct(_autoscaler_attachment())), - "pod-identity-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler_pod_identity()), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config("kubernetes.m.crossplane.io/v1alpha1")), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config("helm.m.crossplane.io/v1beta1")), - ready=fnv1.READY_TRUE, - ), - } + ) -def _compose_cases() -> list[Case]: - """The cases test_compose runs.""" - # Second pass: cluster and cluster-auth observed Ready, function flips - # those two desired resources ready while still emitting everything. - ready_resources = _expected_resources() - ready_resources["cluster"] = fnv1.Resource( - resource=ready_resources["cluster"].resource, - ready=fnv1.READY_TRUE, - ) - ready_resources["cluster-auth"] = fnv1.Resource( - resource=ready_resources["cluster-auth"].resource, - ready=fnv1.READY_TRUE, - ) - ready_resources["efs-filesystem"] = fnv1.Resource( - resource=ready_resources["efs-filesystem"].resource, - ready=fnv1.READY_TRUE, - ) - # Once the EFS filesystem id is observed, the managed StorageClass Object - # is composed (and marked ready) against the cluster's own ProviderConfig. - ready_resources["storage-class-rwx-efs"] = fnv1.Resource( - resource=resource.dict_to_struct(_storage_class_object("fs-0abc123")), - ready=fnv1.READY_TRUE, +def _efs_security_group_ingress(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The EFS security group's rule admitting NFS from the VPC.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "SecurityGroupIngressRule", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "ipProtocol": "tcp", + "fromPort": 2049, + "toPort": 2049, + "cidrIpv4": "10.0.0.0/16", + "securityGroupIdSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/sg-role": "efs"}, + }, + }, + }, + } + ), ) - # With the cluster observed, the autoscaler Helm release is composed (it's - # gated on the cluster existing so provider-helm can reach it). It carries - # no Ready condition yet, so it stays not-ready this pass. - ready_resources["release-cluster-autoscaler"] = fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler_release()), + + +def _efs_mount_target(*, subnet_name: str, cred_kind: str, cred_name: str) -> fnv1.Resource: + """An EFS mount target in the named subnet.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "efs.aws.m.upbound.io/v1beta1", + "kind": "MountTarget", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "fileSystemIdSelector": {"matchControllerRef": True}, + "subnetIdRef": {"name": subnet_name}, + "securityGroupsSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/sg-role": "efs"}, + }, + }, + }, + } + ), ) - return [ - Case( - name="first pass composes infra resources; none ready", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr().model_dump(exclude_none=True, mode="json"), - ), - ), - ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources=_expected_resources(), - ), - context=structpb.Struct(), - ), + +def _pod_identity_association(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The Pod Identity association for the EFS CSI controller.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "PodIdentityAssociation", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "namespace": "kube-system", + "serviceAccount": "efs-csi-controller-sa", + "clusterNameSelector": {"matchControllerRef": True}, + "roleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "efs-csi"}, + }, + }, + }, + } ), - Case( - name="second pass with observed cluster ready marks cluster resources ready", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr().model_dump(exclude_none=True, mode="json"), - ), - ), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_eks_cluster(), - "status": {"conditions": [_ready_condition()]}, - }, - ), - ), - "cluster-auth": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_cluster_auth(), - "status": {"conditions": [_ready_condition()]}, - }, - ), - ), - "efs-filesystem": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_efs_filesystem(), - "metadata": {"annotations": {"crossplane.io/external-name": "fs-0abc123"}}, - "status": {"conditions": [_ready_condition()]}, - }, - ), + ) + + +def _autoscaler_policy(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The IAM policy the cluster autoscaler needs.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "iam.aws.m.upbound.io/v1beta1", + "kind": "Policy", + "metadata": {"labels": {"modelplane.ai/iam-role": "cluster-autoscaler"}}, + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "policy": ( + '{"Version":"2012-10-17","Statement":[' + '{"Effect":"Allow","Action":[' + '"autoscaling:DescribeAutoScalingGroups",' + '"autoscaling:DescribeAutoScalingInstances",' + '"autoscaling:DescribeLaunchConfigurations",' + '"autoscaling:DescribeScalingActivities",' + '"ec2:DescribeImages",' + '"ec2:DescribeInstanceTypes",' + '"ec2:DescribeLaunchTemplateVersions",' + '"ec2:GetInstanceTypesFromInstanceRequirements",' + '"eks:DescribeNodegroup"' + '],"Resource":["*"]},' + '{"Effect":"Allow","Action":[' + '"autoscaling:SetDesiredCapacity",' + '"autoscaling:TerminateInstanceInAutoScalingGroup"' + '],"Resource":["*"]}]}' ), }, - ), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources=ready_resources, - ), - context=structpb.Struct(), - ), + }, + } ), - ] + ) -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) +def _autoscaler_attachment(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The attachment of the autoscaler's policy to its role.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "iam.aws.m.upbound.io/v1beta1", + "kind": "RolePolicyAttachment", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "policyArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "cluster-autoscaler"}, + }, + "roleSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "cluster-autoscaler"}, + }, + }, + }, + } + ), + ) -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) -def test_compose(case: Case) -> None: - """The function composes EKS cluster infrastructure.""" - got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) +def _autoscaler_pod_identity(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The Pod Identity association for the cluster autoscaler.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "PodIdentityAssociation", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-west-2", + "namespace": "kube-system", + "serviceAccount": "cluster-autoscaler", + "clusterNameSelector": {"matchControllerRef": True}, + "roleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "cluster-autoscaler"}, + }, + }, + }, + } + ), + ) -def test_compose_capacity_block() -> None: - """A Capacity Block pool composes a launch template and a CAPACITY_BLOCK node group.""" - # The GPU node group must not set instanceTypes (EKS takes the type - # from the launch template), must set capacityType=CAPACITY_BLOCK, and - # must reference the launch template. The launch template targets the - # reservation via the capacity-block market type. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr_capacity_block().model_dump(exclude_none=True, mode="json"), - ), - ), +def _autoscaler_release() -> fnv1.Resource: + """The cluster autoscaler's Helm release.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster-kubeconfig-55b57"}, + "forProvider": { + "chart": { + "name": "cluster-autoscaler", + "repository": "https://kubernetes.github.io/autoscaler", + "version": "9.57.0", + }, + "namespace": "kube-system", + "values": { + "cloudProvider": "aws", + "awsRegion": "us-west-2", + "autoDiscovery": {"clusterName": "modelplane-system-test-cluster-eks-0865f"}, + "rbac": {"serviceAccount": {"name": "cluster-autoscaler"}}, + "extraArgs": {"balance-similar-node-groups": True}, + }, + }, + }, + } ), ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - resources = got.desired.resources - - # The launch template is composed and targets the reservation. - assert "launch-template-gpu-h200" in resources - assert resource.struct_to_dict(resources["launch-template-gpu-h200"].resource) == _launch_template() - - # The GPU node group uses CAPACITY_BLOCK + the launch template and - # carries no instanceTypes. - assert resource.struct_to_dict(resources["nodegroup-gpu-h200"].resource) == _gpu_node_group_capacity_block() - - -def test_compose_efa() -> None: - """An EFA GPU pool composes EFA infrastructure end to end.""" - # The node group's launch template carries one EFA interface per network - # card (card 0 keeps device index 0 for the node's IP traffic, the rest - # device index 1 for RDMA), the cluster gets an EFA security group with - # self-referencing all-traffic ingress and egress rules, and the node - # group references the launch template instead of setting instanceTypes. - want_resources = { - "vpc": fnv1.Resource(resource=resource.dict_to_struct(_vpc())), - "subnet-0": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_A, "us-west-2a", "10.0.0.0/20")), - ), - "subnet-1": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_B, "us-west-2b", "10.0.16.0/20")), - ), - "subnet-2": fnv1.Resource( - resource=resource.dict_to_struct(_subnet(_SUBNET_C, "us-west-2c", "10.0.32.0/20")), - ), - "private-subnet-0": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_A, "us-west-2a", "10.0.48.0/20")), - ), - "private-subnet-1": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_B, "us-west-2b", "10.0.64.0/20")), - ), - "private-subnet-2": fnv1.Resource( - resource=resource.dict_to_struct(_private_subnet(_PRIVATE_SUBNET_C, "us-west-2c", "10.0.80.0/20")), - ), - "internet-gateway": fnv1.Resource(resource=resource.dict_to_struct(_internet_gateway())), - "nat-eip": fnv1.Resource(resource=resource.dict_to_struct(_nat_eip())), - "nat-gateway": fnv1.Resource(resource=resource.dict_to_struct(_nat_gateway("us-west-2a"))), - "route-table": fnv1.Resource(resource=resource.dict_to_struct(_route_table())), - "route-default": fnv1.Resource(resource=resource.dict_to_struct(_route_default())), - "private-route-table": fnv1.Resource(resource=resource.dict_to_struct(_private_route_table())), - "private-route-default": fnv1.Resource(resource=resource.dict_to_struct(_private_route_default())), - "route-table-association-0": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2a")), - ), - "route-table-association-1": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2b")), - ), - "route-table-association-2": fnv1.Resource( - resource=resource.dict_to_struct(_route_table_association("us-west-2c")), - ), - "private-route-table-association-0": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2a")), - ), - "private-route-table-association-1": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2b")), - ), - "private-route-table-association-2": fnv1.Resource( - resource=resource.dict_to_struct(_private_route_table_association("us-west-2c")), - ), - "iam-role-cluster": fnv1.Resource( - resource=resource.dict_to_struct(_role("cluster", _ASSUME_CLUSTER)), - ), - "iam-attach-cluster-policy": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("cluster", "arn:aws:iam::aws:policy/AmazonEKSClusterPolicy"), - ), - ), - "iam-role-node": fnv1.Resource(resource=resource.dict_to_struct(_role("node", _ASSUME_NODE))), - "iam-attach-node-worker": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy"), - ), - ), - "iam-attach-node-cni": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy"), - ), - ), - "iam-attach-node-ecr": fnv1.Resource( - resource=resource.dict_to_struct( - _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly"), - ), - ), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_eks_cluster())), - "cluster-auth": fnv1.Resource(resource=resource.dict_to_struct(_cluster_auth())), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_system_node_group())), - "launch-template-gpu-h200": fnv1.Resource(resource=resource.dict_to_struct(_launch_template_efa())), - "efa-security-group": fnv1.Resource(resource=resource.dict_to_struct(_efa_security_group())), - "efa-security-group-ingress": fnv1.Resource( - resource=resource.dict_to_struct(_efa_security_group_ingress()), - ), - "efa-security-group-egress": fnv1.Resource( - resource=resource.dict_to_struct(_efa_security_group_egress()), - ), - "nodegroup-gpu-h200": fnv1.Resource(resource=resource.dict_to_struct(_gpu_node_group_efa())), - "addon-vpc-cni": fnv1.Resource(resource=resource.dict_to_struct(_addon("vpc-cni"))), - "addon-kube-proxy": fnv1.Resource(resource=resource.dict_to_struct(_addon("kube-proxy"))), - "addon-coredns": fnv1.Resource(resource=resource.dict_to_struct(_addon("coredns"))), - "efs-filesystem": fnv1.Resource(resource=resource.dict_to_struct(_efs_filesystem())), - "efs-security-group": fnv1.Resource(resource=resource.dict_to_struct(_efs_security_group())), - "efs-security-group-ingress": fnv1.Resource( - resource=resource.dict_to_struct(_efs_security_group_ingress()), - ), - "efs-mount-target-0": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_A))), - "efs-mount-target-1": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_B))), - "efs-mount-target-2": fnv1.Resource(resource=resource.dict_to_struct(_efs_mount_target(_PRIVATE_SUBNET_C))), - "iam-role-efs-csi": fnv1.Resource( - resource=resource.dict_to_struct(_role("efs-csi", _ASSUME_POD_IDENTITY)), - ), - "iam-attach-efs-csi": fnv1.Resource( - resource=resource.dict_to_struct(_role_policy_attachment("efs-csi", _POLICY_EFS_CSI)), - ), - "addon-eks-pod-identity-agent": fnv1.Resource( - resource=resource.dict_to_struct(_addon("eks-pod-identity-agent")), - ), - "pod-identity-efs-csi": fnv1.Resource(resource=resource.dict_to_struct(_pod_identity_association())), - "addon-aws-efs-csi-driver": fnv1.Resource( - resource=resource.dict_to_struct(_addon("aws-efs-csi-driver")), - ), - "iam-policy-cluster-autoscaler": fnv1.Resource(resource=resource.dict_to_struct(_autoscaler_policy())), - "iam-role-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_role("cluster-autoscaler", _ASSUME_POD_IDENTITY)), - ), - "iam-attach-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler_attachment()), - ), - "pod-identity-cluster-autoscaler": fnv1.Resource( - resource=resource.dict_to_struct(_autoscaler_pod_identity()), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config("kubernetes.m.crossplane.io/v1alpha1")), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config("helm.m.crossplane.io/v1beta1")), - ready=fnv1.READY_TRUE, + +def _provider_config(*, api_version: str) -> fnv1.Resource: + """A Ready ProviderConfig that reaches the cluster through its kubeconfig Secret.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": api_version, + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster-kubeconfig-55b57"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", + "key": "kubeconfig", + }, + }, + }, + } ), - } + ready=fnv1.READY_TRUE, + ) - case = Case( - name="an EFA pool composes EFA launch template, security group, and rules", + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +COMPOSE_CASES = [ + Case( + name="first pass composes infra resources; only the ProviderConfigs are ready", req=fnv1.RunFunctionRequest( observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr_efa().model_dump(exclude_none=True, mode="json"), + composite=_xr( + credentials=None, + node_pool=v1alpha1.NodePool( + name="gpu-l4", + role="GPU", + instanceType="g6.xlarge", + nodeCount=1, + minNodeCount=0, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l4"), + zones=[v1alpha1.Zone("us-west-2a"), v1alpha1.Zone("us-west-2b")], ), ), ), @@ -1428,215 +1105,1822 @@ def test_compose_efa() -> None: want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources=want_resources, + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ClusterProviderConfig", cred_name="default"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "internet-gateway": _internet_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-eip": _nat_eip(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-gateway": _nat_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-table": _route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-default": _route_default(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-table": _private_route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-default": _private_route_default( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster": _cluster_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-node": _node_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "cluster": _eks_cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster-auth": _cluster_auth( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-system": _system_node_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "nodegroup-gpu-l4": _gpu_node_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-filesystem": _efs_filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "efs-security-group": _efs_security_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, ), context=structpb.Struct(), ), - ) - - got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) - - -def test_compose_efa_cluster_security_group() -> None: - """Once both security groups are observed, every interface carries them.""" - # A launch template with networkInterfaces makes its security groups - # authoritative, so the interfaces must carry both the EFA security group - # and the EKS cluster security group or the node never joins. Both are set - # as raw IDs in securityGroups (not securityGroupRefs): the provider's - # reference resolver no-ops once that field is populated, so a ref mixed - # with a literal would be dropped. The EFA group's ID comes from its - # observed external name, the cluster group's from the observed cluster's - # status, so both appear only once their resources report them. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr_efa().model_dump(exclude_none=True, mode="json"), + ), + # The cluster, its ClusterAuth and the EFS filesystem are observed Ready, so + # the function marks them ready, alongside the ProviderConfigs it always marks + # ready. The observed filesystem id lets it compose the StorageClass Object, + # also ready, against the cluster's own ProviderConfig. The observed cluster + # lets it compose the autoscaler Helm release, since provider-helm can now + # reach the cluster. The release isn't observed yet, so it isn't ready. + Case( + name="second pass with observed cluster ready marks cluster resources ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pool=v1alpha1.NodePool( + name="gpu-l4", + role="GPU", + instanceType="g6.xlarge", + nodeCount=1, + minNodeCount=0, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l4"), + zones=[v1alpha1.Zone("us-west-2a"), v1alpha1.Zone("us-west-2b")], + ), ), - ), - resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_eks_cluster(), - "status": { - "atProvider": { - "vpcConfig": {"clusterSecurityGroupId": "sg-0cluster"}, + resources={ + "cluster": _observed_eks_cluster(cluster_security_group_id=None, ready=True), + "cluster-auth": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "ClusterAuth", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "clusterNameSelector": {"matchControllerRef": True}, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, }, - }, - }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), ), - ), - "efa-security-group": fnv1.Resource( - resource=resource.dict_to_struct( - { - **_efa_security_group(), - "metadata": { - **_efa_security_group()["metadata"], - "annotations": {"crossplane.io/external-name": "sg-0efa"}, - }, - }, + "efs-filesystem": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "efs.aws.m.upbound.io/v1beta1", + "kind": "FileSystem", + "metadata": {"annotations": {"crossplane.io/external-name": "fs-0abc123"}}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "throughputMode": "elastic", + "encrypted": True, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), ), - ), - }, - ), - ) - - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - lt = resource.struct_to_dict(got.desired.resources["launch-template-gpu-h200"].resource) - interfaces = lt["spec"]["forProvider"]["networkInterfaces"] - - # Every interface carries both SGs as raw IDs (EFA first, then cluster) - # and no securityGroupRefs; no interface requests a public IP (nodes are - # in private subnets). - assert interfaces[0]["interfaceType"] == "efa" - for ni in interfaces: - assert "securityGroupRefs" not in ni - assert ni["securityGroups"] == ["sg-0efa", "sg-0cluster"] - assert "associatePublicIpAddress" not in ni - for ni in interfaces[1:]: - assert ni["interfaceType"] == "efa-only" - - -def test_compose_efa_dra_driver() -> None: - """An EFA pool installs the EFA DRA driver Helm release, and a pool without EFA doesn't.""" - # Like the autoscaler, the release is gated on the cluster being observed - # so provider-helm can reach it. A pool without the EFA fabric installs no - # driver even once the cluster is observed. - observed_cluster = { - "cluster": fnv1.Resource( - resource=resource.dict_to_struct( - {**_eks_cluster(), "status": {"conditions": [_ready_condition()]}}, + }, ), ), - } - - got_efa = asyncio.run( - fn.FunctionRunner().RunFunction( - fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_xr_efa().model_dump(exclude_none=True, mode="json")), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ClusterProviderConfig", cred_name="default"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", ), - resources=observed_cluster, - ), - ), - None, - ) - ) - assert "release-efa-dra-driver" in got_efa.desired.resources - assert ( - resource.struct_to_dict(got_efa.desired.resources["release-efa-dra-driver"].resource) - == _efa_dra_driver_release() - ) - - got_none = asyncio.run( - fn.FunctionRunner().RunFunction( - fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_xr().model_dump(exclude_none=True, mode="json")), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", ), - resources=observed_cluster, - ), - ), - None, - ) - ) - assert "release-efa-dra-driver" not in got_none.desired.resources - - -def test_custom_credentials() -> None: - """Custom credentials flow through to all cloud MRs.""" - # When spec.credentials is set with a custom type and name, every cloud - # provider MR (VPC, subnets, IAM roles, EKS cluster, node groups, addons, - # EFS resources, autoscaler IAM resources) carries the corresponding - # providerConfigRef. The kubeconfig-based resources (provider-config-kubernetes, - # provider-config-helm, release-*, storage-class-*) are unaffected. - ck = "ProviderConfig" - cn = "my-aws-account" - creds = v1alpha1.Credentials(type=ck, name=cn) - - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr(credentials=creds).model_dump(exclude_none=True, mode="json"), - ), - ), - ), - ) - - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - rs = got.desired.resources - - cloud_checks = { - "vpc": _vpc(ck, cn), - "subnet-0": _subnet(_SUBNET_A, "us-west-2a", "10.0.0.0/20", ck, cn), - "subnet-1": _subnet(_SUBNET_B, "us-west-2b", "10.0.16.0/20", ck, cn), - "subnet-2": _subnet(_SUBNET_C, "us-west-2c", "10.0.32.0/20", ck, cn), - "private-subnet-0": _private_subnet(_PRIVATE_SUBNET_A, "us-west-2a", "10.0.48.0/20", ck, cn), - "private-subnet-1": _private_subnet(_PRIVATE_SUBNET_B, "us-west-2b", "10.0.64.0/20", ck, cn), - "private-subnet-2": _private_subnet(_PRIVATE_SUBNET_C, "us-west-2c", "10.0.80.0/20", ck, cn), - "internet-gateway": _internet_gateway(ck, cn), - "nat-eip": _nat_eip(ck, cn), - "nat-gateway": _nat_gateway("us-west-2a", ck, cn), - "route-table": _route_table(ck, cn), - "route-default": _route_default(ck, cn), - "private-route-table": _private_route_table(ck, cn), - "private-route-default": _private_route_default(ck, cn), - "route-table-association-0": _route_table_association("us-west-2a", ck, cn), - "route-table-association-1": _route_table_association("us-west-2b", ck, cn), - "route-table-association-2": _route_table_association("us-west-2c", ck, cn), - "private-route-table-association-0": _private_route_table_association("us-west-2a", ck, cn), - "private-route-table-association-1": _private_route_table_association("us-west-2b", ck, cn), - "private-route-table-association-2": _private_route_table_association("us-west-2c", ck, cn), - "iam-role-cluster": _role("cluster", _ASSUME_CLUSTER, ck, cn), - "iam-attach-cluster-policy": _role_policy_attachment( - "cluster", "arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", ck, cn - ), - "iam-role-node": _role("node", _ASSUME_NODE, ck, cn), - "iam-attach-node-worker": _role_policy_attachment( - "node", "arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", ck, cn - ), - "iam-attach-node-cni": _role_policy_attachment("node", "arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", ck, cn), - "iam-attach-node-ecr": _role_policy_attachment( - "node", "arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", ck, cn - ), - "cluster": _eks_cluster(ck, cn), - "cluster-auth": _cluster_auth(ck, cn), - "nodegroup-system": _system_node_group(ck, cn), - "nodegroup-gpu-l4": _gpu_node_group(ck, cn), - "addon-vpc-cni": _addon("vpc-cni", ck, cn), - "addon-kube-proxy": _addon("kube-proxy", ck, cn), - "addon-coredns": _addon("coredns", ck, cn), - "efs-filesystem": _efs_filesystem(ck, cn), - "efs-security-group": _efs_security_group(ck, cn), - "efs-security-group-ingress": _efs_security_group_ingress(ck, cn), - "efs-mount-target-0": _efs_mount_target(_PRIVATE_SUBNET_A, ck, cn), - "efs-mount-target-1": _efs_mount_target(_PRIVATE_SUBNET_B, ck, cn), - "efs-mount-target-2": _efs_mount_target(_PRIVATE_SUBNET_C, ck, cn), - "iam-role-efs-csi": _role("efs-csi", _ASSUME_POD_IDENTITY, ck, cn), - "iam-attach-efs-csi": _role_policy_attachment("efs-csi", _POLICY_EFS_CSI, ck, cn), - "addon-eks-pod-identity-agent": _addon("eks-pod-identity-agent", ck, cn), - "pod-identity-efs-csi": _pod_identity_association(ck, cn), - "addon-aws-efs-csi-driver": _addon("aws-efs-csi-driver", ck, cn), - "iam-policy-cluster-autoscaler": _autoscaler_policy(ck, cn), - "iam-role-cluster-autoscaler": _role("cluster-autoscaler", _ASSUME_POD_IDENTITY, ck, cn), - "iam-attach-cluster-autoscaler": _autoscaler_attachment(ck, cn), - "pod-identity-cluster-autoscaler": _autoscaler_pod_identity(ck, cn), - } - - for key, want in cloud_checks.items(): - assert key in rs, f"resource {key!r} not found in desired" - got_dict = resource.struct_to_dict(rs[key].resource) - assert got_dict == want, f"resource {key!r} mismatch" - - # kubeconfig-based resources must NOT carry the cloud providerConfigRef - for key in ("provider-config-kubernetes", "provider-config-helm"): - got_dict = resource.struct_to_dict(rs[key].resource) - assert "providerConfigRef" not in got_dict.get("spec", {}), f"{key} should not have providerConfigRef" + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "internet-gateway": _internet_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-eip": _nat_eip(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-gateway": _nat_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-table": _route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-default": _route_default(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-table": _private_route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-default": _private_route_default( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster": _cluster_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-node": _node_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "cluster": _eks_cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "cluster-auth": _cluster_auth( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "nodegroup-system": _system_node_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "nodegroup-gpu-l4": _gpu_node_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-filesystem": _efs_filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "efs-security-group": _efs_security_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "storage-class-rwx-efs": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "storage.k8s.io/v1", + "kind": "StorageClass", + "metadata": {"name": "modelplane-rwx-efs"}, + "provisioner": "efs.csi.aws.com", + "parameters": { + "provisioningMode": "efs-ap", + "fileSystemId": "fs-0abc123", + "directoryPerms": "700", + }, + "volumeBindingMode": "Immediate", + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "release-cluster-autoscaler": _autoscaler_release(), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + ), + ), + # The launch template targets the reservation through the capacity-block + # market type. The node group sets capacityType CAPACITY_BLOCK and references + # the launch template. It sets no instanceTypes, because EKS takes the type + # from the launch template. + Case( + name="a Capacity Block pool composes a launch template and a CAPACITY_BLOCK node group", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pool=v1alpha1.NodePool( + name="gpu-h200", + role="GPU", + instanceType="p5en.48xlarge", + nodeCount=2, + minNodeCount=0, + maxNodeCount=2, + diskSizeGb=1024, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h200"), + capacityBlock=v1alpha1.CapacityBlock(capacityReservationId="cr-0123456789abcdef0"), + zones=[v1alpha1.Zone("us-west-2a")], + ), + ), + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ClusterProviderConfig", cred_name="default"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "internet-gateway": _internet_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-eip": _nat_eip(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-gateway": _nat_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-table": _route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-default": _route_default(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-table": _private_route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-default": _private_route_default( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster": _cluster_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-node": _node_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "cluster": _eks_cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster-auth": _cluster_auth( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-system": _system_node_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "launch-template-gpu-h200": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "LaunchTemplate", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "name": "test-cluster-lt-gpu-h200-83c00", + "instanceType": "p5en.48xlarge", + "blockDeviceMappings": [ + {"deviceName": "/dev/xvda", "ebs": {"volumeSize": 1024}}, + ], + "instanceMarketOptions": {"marketType": "capacity-block"}, + "capacityReservationSpecification": { + "capacityReservationPreference": "capacity-reservations-only", + "capacityReservationTarget": { + "capacityReservationId": "cr-0123456789abcdef0", + }, + }, + }, + }, + } + ), + ), + "nodegroup-gpu-h200": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "eks.aws.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "managementPolicies": ["Observe", "Create", "Update", "Delete"], + "initProvider": {"scalingConfig": {"desiredSize": 2}}, + "forProvider": { + "region": "us-west-2", + "amiType": "AL2023_x86_64_NVIDIA", + "clusterNameSelector": {"matchControllerRef": True}, + "nodeRoleArnSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/iam-role": "node"}, + }, + "capacityType": "CAPACITY_BLOCK", + "launchTemplate": { + "name": "test-cluster-lt-gpu-h200-83c00", + "version": "$Latest", + }, + "subnetIdRefs": [{"name": "test-cluster-private-subnet-us-west-2a-6a89f"}], + "scalingConfig": {"minSize": 0, "maxSize": 2}, + "labels": { + "modelplane.ai/gpu": "nvidia-h200", + "modelplane.ai/pool": "gpu-h200", + }, + "taint": [ + { + "key": "nvidia.com/gpu", + "value": "true", + "effect": "NO_SCHEDULE", + }, + ], + }, + }, + } + ), + ), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-filesystem": _efs_filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "efs-security-group": _efs_security_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + ), + ), + # The node group's launch template carries one EFA interface per network card + # (card 0 keeps device index 0 for the node's IP traffic, the rest device + # index 1 for RDMA), the cluster gets an EFA security group with + # self-referencing all-traffic ingress and egress rules, and the node group + # references the launch template instead of setting instanceTypes. + Case( + name="an EFA pool composes EFA launch template, security group, and rules", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pool=v1alpha1.NodePool( + name="gpu-h200", + role="GPU", + instanceType="p5en.48xlarge", + nodeCount=2, + minNodeCount=0, + maxNodeCount=2, + diskSizeGb=1024, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h200"), + fabric="EFA", + zones=[v1alpha1.Zone("us-west-2a")], + ), + ), + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ClusterProviderConfig", cred_name="default"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "internet-gateway": _internet_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-eip": _nat_eip(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-gateway": _nat_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-table": _route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-default": _route_default(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-table": _private_route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-default": _private_route_default( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster": _cluster_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-node": _node_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "cluster": _eks_cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster-auth": _cluster_auth( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-system": _system_node_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "launch-template-gpu-h200": _launch_template_efa(security_groups=None), + "efa-security-group": _efa_security_group(), + "efa-security-group-ingress": _efa_security_group_ingress(), + "efa-security-group-egress": _efa_security_group_egress(), + "nodegroup-gpu-h200": _gpu_node_group_efa(), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-filesystem": _efs_filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "efs-security-group": _efs_security_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + ), + ), + # A launch template with networkInterfaces makes its security groups + # authoritative, so the interfaces must carry both the EFA security group and + # the EKS cluster security group or the node never joins. Both are set as raw + # IDs in securityGroups (not securityGroupRefs): the provider's reference + # resolver no-ops once that field is populated, so a ref mixed with a literal + # would be dropped. The EFA group's ID comes from its observed external name, + # the cluster group's from the observed cluster's status, so both appear only + # once their resources report them. Every interface carries both, EFA first, + # and none requests a public IP, because the nodes are in private subnets. + # The cluster is observed, though not yet Ready, so both Helm releases are + # composed. + Case( + name="once both security groups are observed, every EFA interface carries them", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pool=v1alpha1.NodePool( + name="gpu-h200", + role="GPU", + instanceType="p5en.48xlarge", + nodeCount=2, + minNodeCount=0, + maxNodeCount=2, + diskSizeGb=1024, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h200"), + fabric="EFA", + zones=[v1alpha1.Zone("us-west-2a")], + ), + ), + resources={ + "cluster": _observed_eks_cluster(cluster_security_group_id="sg-0cluster", ready=False), + "efa-security-group": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "ec2.aws.m.upbound.io/v1beta1", + "kind": "SecurityGroup", + "metadata": { + "name": "test-cluster-efa-sg-602a9", + "labels": {"modelplane.ai/fabric": "EFA"}, + "annotations": {"crossplane.io/external-name": "sg-0efa"}, + }, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "region": "us-west-2", + "name": "test-cluster-efa", + "description": "EFA OS-bypass traffic between gang nodes", + "vpcIdSelector": {"matchControllerRef": True}, + }, + }, + } + ), + ), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ClusterProviderConfig", cred_name="default"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "internet-gateway": _internet_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-eip": _nat_eip(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-gateway": _nat_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-table": _route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-default": _route_default(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-table": _private_route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-default": _private_route_default( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster": _cluster_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-node": _node_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "cluster": _eks_cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster-auth": _cluster_auth( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-system": _system_node_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "launch-template-gpu-h200": _launch_template_efa(security_groups=["sg-0efa", "sg-0cluster"]), + "efa-security-group": _efa_security_group(), + "efa-security-group-ingress": _efa_security_group_ingress(), + "efa-security-group-egress": _efa_security_group_egress(), + "nodegroup-gpu-h200": _gpu_node_group_efa(), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-filesystem": _efs_filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "efs-security-group": _efs_security_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "release-cluster-autoscaler": _autoscaler_release(), + "release-efa-dra-driver": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "forProvider": { + "chart": { + "name": "aws-dranet", + "repository": "https://aws.github.io/eks-charts", + "version": "1.0.0", + }, + "namespace": "kube-system", + "values": { + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}, + ], + }, + }, + }, + } + ) + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + ), + ), + # Like the autoscaler, the EFA DRA driver release is gated on the cluster being + # observed so provider-helm can reach it. + Case( + name="an EFA pool installs the EFA DRA driver once the cluster is observed", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pool=v1alpha1.NodePool( + name="gpu-h200", + role="GPU", + instanceType="p5en.48xlarge", + nodeCount=2, + minNodeCount=0, + maxNodeCount=2, + diskSizeGb=1024, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h200"), + fabric="EFA", + zones=[v1alpha1.Zone("us-west-2a")], + ), + ), + resources={ + "cluster": _observed_eks_cluster(cluster_security_group_id=None, ready=True), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ClusterProviderConfig", cred_name="default"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "internet-gateway": _internet_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-eip": _nat_eip(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-gateway": _nat_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-table": _route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-default": _route_default(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-table": _private_route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-default": _private_route_default( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster": _cluster_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-node": _node_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "cluster": _eks_cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "cluster-auth": _cluster_auth( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-system": _system_node_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "launch-template-gpu-h200": _launch_template_efa(security_groups=None), + "efa-security-group": _efa_security_group(), + "efa-security-group-ingress": _efa_security_group_ingress(), + "efa-security-group-egress": _efa_security_group_egress(), + "nodegroup-gpu-h200": _gpu_node_group_efa(), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-filesystem": _efs_filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "efs-security-group": _efs_security_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "release-cluster-autoscaler": _autoscaler_release(), + "release-efa-dra-driver": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "forProvider": { + "chart": { + "name": "aws-dranet", + "repository": "https://aws.github.io/eks-charts", + "version": "1.0.0", + }, + "namespace": "kube-system", + "values": { + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}, + ], + }, + }, + }, + } + ) + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + ), + ), + Case( + name="a pool without the EFA fabric installs no EFA DRA driver once the cluster is observed", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=None, + node_pool=v1alpha1.NodePool( + name="gpu-l4", + role="GPU", + instanceType="g6.xlarge", + nodeCount=1, + minNodeCount=0, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l4"), + zones=[v1alpha1.Zone("us-west-2a"), v1alpha1.Zone("us-west-2b")], + ), + ), + resources={ + "cluster": _observed_eks_cluster(cluster_security_group_id=None, ready=True), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ClusterProviderConfig", cred_name="default"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "internet-gateway": _internet_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-eip": _nat_eip(cred_kind="ClusterProviderConfig", cred_name="default"), + "nat-gateway": _nat_gateway(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-table": _route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "route-default": _route_default(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-table": _private_route_table(cred_kind="ClusterProviderConfig", cred_name="default"), + "private-route-default": _private_route_default( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster": _cluster_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-node": _node_role(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "cluster": _eks_cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "cluster-auth": _cluster_auth( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-system": _system_node_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "nodegroup-gpu-l4": _gpu_node_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ClusterProviderConfig", cred_name="default"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-filesystem": _efs_filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "efs-security-group": _efs_security_group(cred_kind="ClusterProviderConfig", cred_name="default"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ClusterProviderConfig", cred_name="default" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ClusterProviderConfig", cred_name="default" + ), + "release-cluster-autoscaler": _autoscaler_release(), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + ), + ), + # Every cloud provider MR carries the providerConfigRef that spec.credentials + # names. The ProviderConfigs reach the cluster through its kubeconfig, so they + # don't carry the cloud credentials. + Case( + name="custom credentials flow through to all cloud MRs", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-aws-account"), + node_pool=v1alpha1.NodePool( + name="gpu-l4", + role="GPU", + instanceType="g6.xlarge", + nodeCount=1, + minNodeCount=0, + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l4"), + zones=[v1alpha1.Zone("us-west-2a"), v1alpha1.Zone("us-west-2b")], + ), + ), + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "vpc": _vpc(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "subnet-0": _subnet( + name="test-cluster-subnet-us-west-2a-952dc", + az="us-west-2a", + cidr="10.0.0.0/20", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "subnet-1": _subnet( + name="test-cluster-subnet-us-west-2b-2b80f", + az="us-west-2b", + cidr="10.0.16.0/20", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "subnet-2": _subnet( + name="test-cluster-subnet-us-west-2c-03273", + az="us-west-2c", + cidr="10.0.32.0/20", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "private-subnet-0": _private_subnet( + name="test-cluster-private-subnet-us-west-2a-6a89f", + az="us-west-2a", + cidr="10.0.48.0/20", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "private-subnet-1": _private_subnet( + name="test-cluster-private-subnet-us-west-2b-b7832", + az="us-west-2b", + cidr="10.0.64.0/20", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "private-subnet-2": _private_subnet( + name="test-cluster-private-subnet-us-west-2c-ef57d", + az="us-west-2c", + cidr="10.0.80.0/20", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "internet-gateway": _internet_gateway(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "nat-eip": _nat_eip(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "nat-gateway": _nat_gateway(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "route-table": _route_table(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "route-default": _route_default(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "private-route-table": _private_route_table(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "private-route-default": _private_route_default( + cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "route-table-association-0": _route_table_association( + az="us-west-2a", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "route-table-association-1": _route_table_association( + az="us-west-2b", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "route-table-association-2": _route_table_association( + az="us-west-2c", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "private-route-table-association-0": _private_route_table_association( + az="us-west-2a", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "private-route-table-association-1": _private_route_table_association( + az="us-west-2b", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "private-route-table-association-2": _private_route_table_association( + az="us-west-2c", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "iam-role-cluster": _cluster_role(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "iam-attach-cluster-policy": _role_policy_attachment( + role="cluster", + arn="arn:aws:iam::aws:policy/AmazonEKSClusterPolicy", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "iam-role-node": _node_role(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "iam-attach-node-worker": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKSWorkerNodePolicy", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "iam-attach-node-cni": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEKS_CNI_Policy", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "iam-attach-node-ecr": _role_policy_attachment( + role="node", + arn="arn:aws:iam::aws:policy/AmazonEC2ContainerRegistryReadOnly", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "cluster": _eks_cluster( + cred_kind="ProviderConfig", cred_name="my-aws-account", ready=fnv1.READY_UNSPECIFIED + ), + "cluster-auth": _cluster_auth( + cred_kind="ProviderConfig", cred_name="my-aws-account", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-system": _system_node_group(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "nodegroup-gpu-l4": _gpu_node_group(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "addon-vpc-cni": _addon(name="vpc-cni", cred_kind="ProviderConfig", cred_name="my-aws-account"), + "addon-kube-proxy": _addon( + name="kube-proxy", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "addon-coredns": _addon(name="coredns", cred_kind="ProviderConfig", cred_name="my-aws-account"), + "efs-filesystem": _efs_filesystem( + cred_kind="ProviderConfig", cred_name="my-aws-account", ready=fnv1.READY_UNSPECIFIED + ), + "efs-security-group": _efs_security_group(cred_kind="ProviderConfig", cred_name="my-aws-account"), + "efs-security-group-ingress": _efs_security_group_ingress( + cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "efs-mount-target-0": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2a-6a89f", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "efs-mount-target-1": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2b-b7832", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "efs-mount-target-2": _efs_mount_target( + subnet_name="test-cluster-private-subnet-us-west-2c-ef57d", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "iam-role-efs-csi": _pod_identity_role( + role="efs-csi", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "iam-attach-efs-csi": _role_policy_attachment( + role="efs-csi", + arn="arn:aws:iam::aws:policy/service-role/AmazonEFSCSIDriverPolicy", + cred_kind="ProviderConfig", + cred_name="my-aws-account", + ), + "addon-eks-pod-identity-agent": _addon( + name="eks-pod-identity-agent", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "pod-identity-efs-csi": _pod_identity_association( + cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "addon-aws-efs-csi-driver": _addon( + name="aws-efs-csi-driver", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "iam-policy-cluster-autoscaler": _autoscaler_policy( + cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "iam-role-cluster-autoscaler": _pod_identity_role( + role="cluster-autoscaler", cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "iam-attach-cluster-autoscaler": _autoscaler_attachment( + cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "pod-identity-cluster-autoscaler": _autoscaler_pod_identity( + cred_kind="ProviderConfig", cred_name="my-aws-account" + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + ), + ), +] + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes EKS cluster infrastructure.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-gke-cluster/tests/test_fn.py b/functions/compose-gke-cluster/tests/test_fn.py index 07898ac6b..a4ca1e9b8 100644 --- a/functions/compose-gke-cluster/tests/test_fn.py +++ b/functions/compose-gke-cluster/tests/test_fn.py @@ -38,35 +38,9 @@ class Case: want: fnv1.RunFunctionResponse -_DEFAULT_CRED_KIND = "ClusterProviderConfig" -_DEFAULT_CRED_NAME = "default" - -_GCP_PROVIDER_CONFIG = { - "apiVersion": "gcp.m.upbound.io/v1beta1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "default"}, - "spec": { - "projectID": "my-gcp-project", - "credentials": { - "source": "Secret", - "secretRef": { - "name": "gcp-credentials", - "namespace": "crossplane-system", - "key": "credentials", - }, - }, - }, -} - -_GCP_PROVIDER_CONFIG_SELECTOR = fnv1.ResourceSelector( - api_version="gcp.m.upbound.io/v1beta1", - kind="ClusterProviderConfig", - match_name="default", -) - - -def _gke_xr(credentials: v1alpha1.Credentials | None = None) -> v1alpha1.GKECluster: - return v1alpha1.GKECluster( +def _xr(*, credentials: v1alpha1.Credentials | None) -> fnv1.Resource: + """The observed GKECluster XR, with the given credentials.""" + xr = v1alpha1.GKECluster( metadata=metav1.ObjectMeta( name="test-cluster", namespace="modelplane-system", @@ -87,526 +61,304 @@ def _gke_xr(credentials: v1alpha1.Credentials | None = None) -> v1alpha1.GKEClus ], ), ) + return fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json", by_alias=True))) -def _network(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "compute.gcp.m.upbound.io/v1beta1", - "kind": "Network", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "autoCreateSubnetworks": False, - }, - }, - } - - -def _projectservice_filestore(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ProjectService", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "service": "file.googleapis.com", - "disableOnDestroy": False, - }, - }, - } - - -def _subnet(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "compute.gcp.m.upbound.io/v1beta1", - "kind": "Subnetwork", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "region": "us-central1", - "networkSelector": {"matchControllerRef": True}, - "ipCidrRange": "10.0.0.0/24", - "secondaryIpRange": [ - {"rangeName": "pods", "ipCidrRange": "10.1.0.0/16"}, - {"rangeName": "services", "ipCidrRange": "10.2.0.0/16"}, - ], - }, - }, - } - - -def _cluster(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "container.gcp.m.upbound.io/v1beta1", - "kind": "Cluster", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "location": "us-central1", - "deletionProtection": False, - "removeDefaultNodePool": True, - "initialNodeCount": 1, - "minMasterVersion": "1.35", - "networkSelector": {"matchControllerRef": True}, - "subnetworkSelector": {"matchControllerRef": True}, - "ipAllocationPolicy": { - "clusterSecondaryRangeName": "pods", - "servicesSecondaryRangeName": "services", - }, - "releaseChannel": {"channel": "REGULAR"}, - "workloadIdentityConfig": { - "workloadPool": "my-gcp-project.svc.id.goog", - }, - "addonsConfig": { - "gcpFilestoreCsiDriverConfig": {"enabled": True}, - }, - }, - "writeConnectionSecretToRef": { - "name": "test-cluster-kubeconfig-55b57", - }, - }, - } - - -def _nodepool_system(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "container.gcp.m.upbound.io/v1beta1", - "kind": "NodePool", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "location": "us-central1", - "clusterSelector": {"matchControllerRef": True}, - "initialNodeCount": 1, - "autoscaling": {"minNodeCount": 1, "maxNodeCount": 2}, - "nodeConfig": { - "machineType": "e2-standard-4", - "imageType": "COS_CONTAINERD", - "oauthScopes": [ - "https://www.googleapis.com/auth/cloud-platform", - ], - "labels": {"modelplane.ai/pool": "system"}, - }, - }, - }, - } - - -def _nodepool_gpu(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "container.gcp.m.upbound.io/v1beta1", - "kind": "NodePool", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "location": "us-central1", - "clusterSelector": {"matchControllerRef": True}, - "initialNodeCount": 1, - "autoscaling": {"minNodeCount": 0, "maxNodeCount": 8}, - "nodeConfig": { - "machineType": "a2-highgpu-8g", - "diskSizeGb": 100, - "imageType": "COS_CONTAINERD", - "oauthScopes": [ - "https://www.googleapis.com/auth/cloud-platform", - ], - "guestAccelerator": [ +def _desired_xr() -> fnv1.Resource: + """The desired XR, publishing its connection Secrets and cache StorageClass.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "secrets": [ { - "type": "nvidia-tesla-a100", - "count": 8, - "gpuDriverInstallationConfig": { - "gpuDriverVersion": "DEFAULT", - }, + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + { + "type": "GoogleApplicationCredentials", + "name": "test-cluster-sa-key-3295c", + "key": "private_key", }, ], - "labels": { - "modelplane.ai/gpu": "nvidia-tesla-a100", - "modelplane.ai/pool": "gpu-pool", - "cloud.google.com/gke-nvidia-gpu-dra-driver": "true", - }, - }, - }, - }, - } - - -def _service_account(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ServiceAccount", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "displayName": "Crossplane GKECluster test-cluster", - }, - }, - } - - -def _service_account_key(cred_kind: str = _DEFAULT_CRED_KIND, cred_name: str = _DEFAULT_CRED_NAME) -> dict: - return { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ServiceAccountKey", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "serviceAccountIdSelector": {"matchControllerRef": True}, - }, - "writeConnectionSecretToRef": { - "name": "test-cluster-sa-key-3295c", - }, - }, - } - - -def _iam_binding( - sa_email: str = "test-sa@my-gcp-project.iam.gserviceaccount.com", - cred_kind: str = _DEFAULT_CRED_KIND, - cred_name: str = _DEFAULT_CRED_NAME, -) -> dict: - return { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ProjectIAMMember", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "role": "roles/container.admin", - "member": f"serviceAccount:{sa_email}", - "project": "my-gcp-project", - }, - }, - } - - -def _provider_config_kubernetes() -> dict: - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ProviderConfig", - "metadata": {"name": "test-cluster-kubeconfig-55b57"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "name": "test-cluster-kubeconfig-55b57", - "namespace": "modelplane-system", - "key": "kubeconfig", - }, - }, - "identity": { - "type": "GoogleApplicationCredentials", - "source": "Secret", - "secretRef": { - "name": "test-cluster-sa-key-3295c", - "namespace": "modelplane-system", - "key": "private_key", + "cache": {"storageClassName": "modelplane-rwx"}, }, - }, - }, - } - - -def _provider_config_helm() -> dict: - return { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "ProviderConfig", - "metadata": {"name": "test-cluster-kubeconfig-55b57"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "name": "test-cluster-kubeconfig-55b57", - "namespace": "modelplane-system", - "key": "kubeconfig", - }, - }, - "identity": { - "type": "GoogleApplicationCredentials", - "source": "Secret", - "secretRef": { - "name": "test-cluster-sa-key-3295c", - "namespace": "modelplane-system", - "key": "private_key", + } + ), + ) + + +def _gcp_provider_config(*, kind: str, name: str, namespace: str | None) -> fnv1.Resource: + """The GCP provider config the function requires, in a namespace if given.""" + metadata = {"name": name} + if namespace is not None: + metadata["namespace"] = namespace + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "gcp.m.upbound.io/v1beta1", + "kind": kind, + "metadata": metadata, + "spec": { + "projectID": "my-gcp-project", + "credentials": { + "source": "Secret", + "secretRef": { + "name": "gcp-credentials", + "namespace": "crossplane-system", + "key": "credentials", + }, + }, }, - }, - }, - } - - -def _storage_class_rwx(network_name: str) -> dict: - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { - "kind": "ProviderConfig", - "name": "test-cluster-kubeconfig-55b57", - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "storage.k8s.io/v1", - "kind": "StorageClass", - "metadata": {"name": "modelplane-rwx"}, - "provisioner": "filestore.csi.storage.gke.io", - "parameters": { - "tier": "enterprise", - "network": network_name, + } + ), + ) + + +def _network(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed VPC Network.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.gcp.m.upbound.io/v1beta1", + "kind": "Network", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "autoCreateSubnetworks": False, }, - "volumeBindingMode": "Immediate", - "allowVolumeExpansion": True, }, - }, - }, - } - - -def _expected_status() -> dict: - return { - "status": { - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-55b57", - "key": "kubeconfig", + } + ), + ready=ready, + ) + + +def _projectservice_filestore(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The composed ProjectService that enables the Filestore API.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ProjectService", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "service": "file.googleapis.com", + "disableOnDestroy": False, + }, }, - { - "type": "GoogleApplicationCredentials", - "name": "test-cluster-sa-key-3295c", - "key": "private_key", + } + ), + ) + + +def _subnet(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The composed Subnetwork, with secondary ranges for pods and services.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.gcp.m.upbound.io/v1beta1", + "kind": "Subnetwork", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "region": "us-central1", + "networkSelector": {"matchControllerRef": True}, + "ipCidrRange": "10.0.0.0/24", + "secondaryIpRange": [ + {"rangeName": "pods", "ipCidrRange": "10.1.0.0/16"}, + {"rangeName": "services", "ipCidrRange": "10.2.0.0/16"}, + ], + }, }, - ], - "cache": {"storageClassName": "modelplane-rwx"}, - }, - } - - -def _compose_cases() -> list[Case]: - """The cases for test_compose.""" - req1 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr().model_dump(exclude_none=True, mode="json"), - ), - ), + } ), ) - req1.required_resources["gcp-provider-config"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(_GCP_PROVIDER_CONFIG)) + + +def _cluster(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The composed GKE Cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "container.gcp.m.upbound.io/v1beta1", + "kind": "Cluster", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "location": "us-central1", + "deletionProtection": False, + "removeDefaultNodePool": True, + "initialNodeCount": 1, + "minMasterVersion": "1.35", + "networkSelector": {"matchControllerRef": True}, + "subnetworkSelector": {"matchControllerRef": True}, + "ipAllocationPolicy": { + "clusterSecondaryRangeName": "pods", + "servicesSecondaryRangeName": "services", + }, + "releaseChannel": {"channel": "REGULAR"}, + "workloadIdentityConfig": { + "workloadPool": "my-gcp-project.svc.id.goog", + }, + "addonsConfig": { + "gcpFilestoreCsiDriverConfig": {"enabled": True}, + }, + }, + "writeConnectionSecretToRef": { + "name": "test-cluster-kubeconfig-55b57", + }, + }, + } + ), ) - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ), - "projectservice-filestore": fnv1.Resource( - resource=resource.dict_to_struct(_projectservice_filestore()), - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "nodepool-system": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_system()), - ), - "nodepool-gpu-pool": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu()), - ), - "service-account": fnv1.Resource( - resource=resource.dict_to_struct(_service_account()), - ), - "service-account-key": fnv1.Resource( - resource=resource.dict_to_struct(_service_account_key()), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_kubernetes()), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - }, + +def _nodepool_system(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The system NodePool the function adds to every cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "container.gcp.m.upbound.io/v1beta1", + "kind": "NodePool", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "location": "us-central1", + "clusterSelector": {"matchControllerRef": True}, + "initialNodeCount": 1, + "autoscaling": {"minNodeCount": 1, "maxNodeCount": 2}, + "nodeConfig": { + "machineType": "e2-standard-4", + "imageType": "COS_CONTAINERD", + "oauthScopes": [ + "https://www.googleapis.com/auth/cloud-platform", + ], + "labels": {"modelplane.ai/pool": "system"}, + }, + }, + }, + } ), - context=structpb.Struct(), ) - want1.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) - req2 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr().model_dump(exclude_none=True, mode="json"), - ), - ), - resources={ - "service-account": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ServiceAccount", - "spec": { - "forProvider": {}, - }, - "status": { - "atProvider": { - "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", - }, - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", + +def _nodepool_gpu(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The NodePool for the XR's gpu-pool.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "container.gcp.m.upbound.io/v1beta1", + "kind": "NodePool", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "location": "us-central1", + "clusterSelector": {"matchControllerRef": True}, + "initialNodeCount": 1, + "autoscaling": {"minNodeCount": 0, "maxNodeCount": 8}, + "nodeConfig": { + "machineType": "a2-highgpu-8g", + "diskSizeGb": 100, + "imageType": "COS_CONTAINERD", + "oauthScopes": [ + "https://www.googleapis.com/auth/cloud-platform", + ], + "guestAccelerator": [ + { + "type": "nvidia-tesla-a100", + "count": 8, + "gpuDriverInstallationConfig": { + "gpuDriverVersion": "DEFAULT", }, - ], - }, - } - ), - ), - "network": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "compute.gcp.m.upbound.io/v1beta1", - "kind": "Network", - # The external-name annotation carries the - # provider-generated VPC name, which the - # function pins the Filestore StorageClass to. - "metadata": { - "annotations": {"crossplane.io/external-name": "test-cluster-abc12"}, - }, - "spec": { - "forProvider": { - "autoCreateSubnetworks": False, }, + ], + "labels": { + "modelplane.ai/gpu": "nvidia-tesla-a100", + "modelplane.ai/pool": "gpu-pool", + "cloud.google.com/gke-nvidia-gpu-dra-driver": "true", }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - }, - } - ), - ), - }, + }, + }, + }, + } ), ) - req2.required_resources["gcp-provider-config"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(_GCP_PROVIDER_CONFIG)) - ) - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_expected_status()), - ), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ready=fnv1.READY_TRUE, - ), - "projectservice-filestore": fnv1.Resource( - resource=resource.dict_to_struct(_projectservice_filestore()), - ), - # With the network name known, the managed Filestore - # StorageClass is composed against the cluster's own - # provider-kubernetes ProviderConfig, pinned to the - # observed VPC. StorageClass has no Ready condition, - # so readiness is SuccessfulCreate. It's orphaned (no - # Delete policy) so it dies with the cluster instead of - # wedging on a deleted kubeconfig Secret during teardown. - "storage-class-rwx": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class_rwx("test-cluster-abc12")), - ready=fnv1.READY_TRUE, - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "nodepool-system": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_system()), - ), - "nodepool-gpu-pool": fnv1.Resource( - resource=resource.dict_to_struct(_nodepool_gpu()), - ), - "service-account": fnv1.Resource( - resource=resource.dict_to_struct(_service_account()), - ready=fnv1.READY_TRUE, - ), - "service-account-key": fnv1.Resource( - resource=resource.dict_to_struct(_service_account_key()), - ), - "iam-binding": fnv1.Resource( - resource=resource.dict_to_struct(_iam_binding()), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_kubernetes()), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config_helm()), - ready=fnv1.READY_TRUE, - ), - }, + +def _service_account(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed GCP ServiceAccount.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ServiceAccount", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "displayName": "Crossplane GKECluster test-cluster", + }, + }, + } ), - context=structpb.Struct(), + ready=ready, ) - want2.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) - # The ProviderConfig resolved to nothing and no cluster is observed to - # take the project from, so nothing can be composed. The XR is marked - # not ready rather than left to aggregate to trivially ready. - req3 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr().model_dump(exclude_none=True, mode="json"), - ), - ), + +def _service_account_key(*, cred_kind: str, cred_name: str) -> fnv1.Resource: + """The composed ServiceAccountKey.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ServiceAccountKey", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "serviceAccountIdSelector": {"matchControllerRef": True}, + }, + "writeConnectionSecretToRef": { + "name": "test-cluster-sa-key-3295c", + }, + }, + } ), ) - req3.required_resources["gcp-provider-config"].SetInParent() - - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for GCP ClusterProviderConfig default", - ), - ], - context=structpb.Struct(), - ) - want3.requirements.resources["gcp-provider-config"].CopyFrom(_GCP_PROVIDER_CONFIG_SELECTOR) - - return [ - Case(name="first pass composes infra resources; IAM binding gated", req=req1, want=want1), - Case(name="a missing ProviderConfig composes nothing and isn't ready", req=req3, want=want3), - Case( - name="second pass with observed SA email composes IAM binding and marks ready resources", - req=req2, - want=want2, + + +def _provider_config(*, api_version: str) -> fnv1.Resource: + """The provider-kubernetes or provider-helm ProviderConfig for the composed cluster, which is always ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": api_version, + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster-kubeconfig-55b57"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", + "key": "kubeconfig", + }, + }, + "identity": { + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": { + "name": "test-cluster-sa-key-3295c", + "namespace": "modelplane-system", + "key": "private_key", + }, + }, + }, + } ), - ] + ready=fnv1.READY_TRUE, + ) def _to_dict(msg: message.Message) -> dict: @@ -614,98 +366,347 @@ def _to_dict(msg: message.Message) -> dict: return json.loads(json_format.MessageToJson(msg, sort_keys=True)) -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) -def test_compose(case: Case) -> None: - """RunFunction composes GKE cluster infrastructure.""" - got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) - - -def test_custom_credentials() -> None: - """Custom credentials flow through to all cloud MRs.""" - # When spec.credentials is set with a custom type and name, every cloud - # provider MR (Network, ProjectService, Subnetwork, Cluster, NodePools, - # ServiceAccount, ServiceAccountKey, IAM binding) carries the corresponding - # providerConfigRef. The kubeconfig-based resources (provider-config-kubernetes, - # provider-config-helm, storage-class-rwx) are unaffected. - ck = "ProviderConfig" - cn = "my-gcp-account" - creds = v1alpha1.Credentials(type=ck, name=cn) - custom_pc = { - "apiVersion": "gcp.m.upbound.io/v1beta1", - "kind": "ProviderConfig", - "metadata": {"name": cn, "namespace": "crossplane-system"}, - "spec": { - "projectID": "my-gcp-project", - "credentials": { - "source": "Secret", - "secretRef": { - "name": "gcp-credentials", - "namespace": "crossplane-system", - "key": "credentials", - }, +COMPOSE_CASES = [ + Case( + name="first pass composes infra resources; IAM binding gated", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(credentials=None)), + required_resources={ + "gcp-provider-config": fnv1.Resources( + items=[_gcp_provider_config(kind="ClusterProviderConfig", name="default", namespace=None)], + ), }, - }, - } - - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _gke_xr(credentials=creds).model_dump(exclude_none=True, mode="json"), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "network": _network( + cred_kind="ClusterProviderConfig", + cred_name="default", + ready=fnv1.READY_UNSPECIFIED, + ), + "projectservice-filestore": _projectservice_filestore( + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + "subnet": _subnet(cred_kind="ClusterProviderConfig", cred_name="default"), + "cluster": _cluster(cred_kind="ClusterProviderConfig", cred_name="default"), + "nodepool-system": _nodepool_system(cred_kind="ClusterProviderConfig", cred_name="default"), + "nodepool-gpu-pool": _nodepool_gpu(cred_kind="ClusterProviderConfig", cred_name="default"), + "service-account": _service_account( + cred_kind="ClusterProviderConfig", + cred_name="default", + ready=fnv1.READY_UNSPECIFIED, + ), + "service-account-key": _service_account_key(cred_kind="ClusterProviderConfig", cred_name="default"), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gcp-provider-config": fnv1.ResourceSelector( + api_version="gcp.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", + ), + }, + ), + ), + ), + # The ProviderConfig resolved to nothing and no cluster is observed to + # take the project from, so nothing can be composed. The XR is marked + # not ready rather than left to aggregate to trivially ready. + Case( + name="a missing ProviderConfig composes nothing and isn't ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(credentials=None)), + required_resources={"gcp-provider-config": fnv1.Resources()}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for GCP ClusterProviderConfig default", ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gcp-provider-config": fnv1.ResourceSelector( + api_version="gcp.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", + ), + }, ), - resources={ - "service-account": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", - "kind": "ServiceAccount", - "spec": { - "forProvider": {}, - }, - "status": { - "atProvider": { - "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", + ), + ), + Case( + name="second pass with observed SA email composes IAM binding and marks ready resources", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(credentials=None), + resources={ + "service-account": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ServiceAccount", + "spec": {"forProvider": {}}, + "status": { + "atProvider": { + "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", + }, + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], }, - }, - } + } + ), + ), + "network": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.gcp.m.upbound.io/v1beta1", + "kind": "Network", + "metadata": { + # The external-name annotation carries the + # provider-generated VPC name, which the + # function pins the Filestore StorageClass to. + "annotations": {"crossplane.io/external-name": "test-cluster-abc12"}, + }, + "spec": { + "forProvider": { + "autoCreateSubnetworks": False, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), ), + }, + ), + required_resources={ + "gcp-provider-config": fnv1.Resources( + items=[_gcp_provider_config(kind="ClusterProviderConfig", name="default", namespace=None)], ), }, ), - ) - req.required_resources["gcp-provider-config"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(custom_pc)) - ) + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "network": _network( + cred_kind="ClusterProviderConfig", + cred_name="default", + ready=fnv1.READY_TRUE, + ), + "projectservice-filestore": _projectservice_filestore( + cred_kind="ClusterProviderConfig", + cred_name="default", + ), + # Composed once the observed Network's external name gives + # the VPC to pin Filestore to. A StorageClass has no Ready + # condition, hence SuccessfulCreate, and the Object omits + # Delete so it dies with the cluster rather than wedging + # teardown on the deleted kubeconfig Secret. + "storage-class-rwx": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "storage.k8s.io/v1", + "kind": "StorageClass", + "metadata": {"name": "modelplane-rwx"}, + "provisioner": "filestore.csi.storage.gke.io", + "parameters": { + "tier": "enterprise", + "network": "test-cluster-abc12", + }, + "volumeBindingMode": "Immediate", + "allowVolumeExpansion": True, + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "subnet": _subnet(cred_kind="ClusterProviderConfig", cred_name="default"), + "cluster": _cluster(cred_kind="ClusterProviderConfig", cred_name="default"), + "nodepool-system": _nodepool_system(cred_kind="ClusterProviderConfig", cred_name="default"), + "nodepool-gpu-pool": _nodepool_gpu(cred_kind="ClusterProviderConfig", cred_name="default"), + "service-account": _service_account( + cred_kind="ClusterProviderConfig", + cred_name="default", + ready=fnv1.READY_TRUE, + ), + "service-account-key": _service_account_key(cred_kind="ClusterProviderConfig", cred_name="default"), + "iam-binding": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ProjectIAMMember", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "role": "roles/container.admin", + "member": "serviceAccount:test-sa@my-gcp-project.iam.gserviceaccount.com", + "project": "my-gcp-project", + }, + }, + } + ), + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gcp-provider-config": fnv1.ResourceSelector( + api_version="gcp.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", + ), + }, + ), + ), + ), + # Custom credentials set every cloud MR's providerConfigRef. The + # kubeconfig-based ProviderConfigs carry none, so they're unaffected. + Case( + name="custom credentials flow through to all cloud MRs", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-gcp-account")), + resources={ + "service-account": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ServiceAccount", + "spec": {"forProvider": {}}, + "status": { + "atProvider": { + "email": "test-sa@my-gcp-project.iam.gserviceaccount.com", + }, + }, + } + ), + ), + }, + ), + required_resources={ + "gcp-provider-config": fnv1.Resources( + items=[ + _gcp_provider_config( + kind="ProviderConfig", + name="my-gcp-account", + namespace="crossplane-system", + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(), + resources={ + "network": _network( + cred_kind="ProviderConfig", + cred_name="my-gcp-account", + ready=fnv1.READY_UNSPECIFIED, + ), + "projectservice-filestore": _projectservice_filestore( + cred_kind="ProviderConfig", + cred_name="my-gcp-account", + ), + "subnet": _subnet(cred_kind="ProviderConfig", cred_name="my-gcp-account"), + "cluster": _cluster(cred_kind="ProviderConfig", cred_name="my-gcp-account"), + "nodepool-system": _nodepool_system(cred_kind="ProviderConfig", cred_name="my-gcp-account"), + "nodepool-gpu-pool": _nodepool_gpu(cred_kind="ProviderConfig", cred_name="my-gcp-account"), + "service-account": _service_account( + cred_kind="ProviderConfig", + cred_name="my-gcp-account", + ready=fnv1.READY_UNSPECIFIED, + ), + "service-account-key": _service_account_key(cred_kind="ProviderConfig", cred_name="my-gcp-account"), + "iam-binding": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "cloudplatform.gcp.m.upbound.io/v1beta1", + "kind": "ProjectIAMMember", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "my-gcp-account"}, + "forProvider": { + "role": "roles/container.admin", + "member": "serviceAccount:test-sa@my-gcp-project.iam.gserviceaccount.com", + "project": "my-gcp-project", + }, + }, + } + ), + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + # A ProviderConfig, unlike a ClusterProviderConfig, is namespaced, + # so the function requires it from the XR's namespace. Crossplane + # wouldn't return the request's crossplane-system one for this + # selector, but the function doesn't check its namespace. + requirements=fnv1.Requirements( + resources={ + "gcp-provider-config": fnv1.ResourceSelector( + api_version="gcp.m.upbound.io/v1beta1", + kind="ProviderConfig", + match_name="my-gcp-account", + namespace="modelplane-system", + ), + }, + ), + ), + ), +] - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - rs = got.desired.resources - - cloud_checks = { - "network": _network(ck, cn), - "projectservice-filestore": _projectservice_filestore(ck, cn), - "subnet": _subnet(ck, cn), - "cluster": _cluster(ck, cn), - "nodepool-system": _nodepool_system(ck, cn), - "nodepool-gpu-pool": _nodepool_gpu(ck, cn), - "service-account": _service_account(ck, cn), - "service-account-key": _service_account_key(ck, cn), - "iam-binding": _iam_binding("test-sa@my-gcp-project.iam.gserviceaccount.com", ck, cn), - } - - got_cloud = {key: resource.struct_to_dict(r.resource) for key, r in rs.items() if key in cloud_checks} - assert got_cloud == cloud_checks - - # kubeconfig-based resources must NOT carry the cloud providerConfigRef - for key in ("provider-config-kubernetes", "provider-config-helm"): - got_dict = resource.struct_to_dict(rs[key].resource) - assert "providerConfigRef" not in got_dict.get("spec", {}), f"{key} should not have providerConfigRef" - - custom_selector = fnv1.ResourceSelector( - api_version="gcp.m.upbound.io/v1beta1", - kind="ProviderConfig", - match_name=cn, - namespace="modelplane-system", - ) - assert _to_dict(got.requirements.resources["gcp-provider-config"]) == _to_dict(custom_selector) + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes GKE cluster infrastructure.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-inference-class/tests/test_fn.py b/functions/compose-inference-class/tests/test_fn.py index 327807286..3fbb6944d 100644 --- a/functions/compose-inference-class/tests/test_fn.py +++ b/functions/compose-inference-class/tests/test_fn.py @@ -38,6 +38,11 @@ class Case: want: fnv1.RunFunctionResponse +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + COMPOSE_CASES = [ Case( name="marks XR ready with Accepted condition and empty status", @@ -59,7 +64,7 @@ class Case: ), ], ), - ).model_dump(exclude_none=True, mode="json") + ).model_dump(exclude_none=True, mode="json", by_alias=True) ), ), ), @@ -72,6 +77,7 @@ class Case: ready=fnv1.READY_TRUE, ), ), + context=structpb.Struct(), conditions=[ fnv1.Condition( type="Accepted", @@ -79,17 +85,11 @@ class Case: reason="Available", ), ], - context=structpb.Struct(), ), ), ] -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - @pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction marks the InferenceClass ready.""" diff --git a/functions/compose-inference-cluster/tests/test_fn.py b/functions/compose-inference-cluster/tests/test_fn.py index 9b7625902..3fae42ed1 100644 --- a/functions/compose-inference-cluster/tests/test_fn.py +++ b/functions/compose-inference-cluster/tests/test_fn.py @@ -15,7 +15,6 @@ """Tests for the compose-inference-cluster function.""" import asyncio -import copy import dataclasses import json @@ -29,198 +28,315 @@ from models.ai.modelplane.inferencecluster import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -# The internal name Modelplane derives for this cluster's gateway, which -# compose-inference-gateway resolves. Built from the SDK's own child_name, so the -# namespace and suffix are asserted independently of the function under test. -_GATEWAY_HOSTNAME = f"{resource.child_name('gateway', 'test-cluster')}.modelplane-system.svc.cluster.local" - @dataclasses.dataclass -class Case: - """A test case for compose-inference-cluster.""" +class ComposeCase: + """A test case for RunFunction.""" name: str req: fnv1.RunFunctionRequest want: fnv1.RunFunctionResponse -_ACTIVATION_API_VERSION = "apiextensions.crossplane.io/v1alpha1" - +@dataclasses.dataclass +class GatewayHostnameCase: + """A test case for _gateway_hostname.""" -def _observe_activated(req: fnv1.RunFunctionRequest, kinds: tuple[str, ...]) -> None: - """Observe the composed activation policy with the kinds in status.activated, - so the function composes the cluster XR rather than waiting for activation.""" - req.observed.resources["activation"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": _ACTIVATION_API_VERSION, - "kind": "ManagedResourceActivationPolicy", - "status": {"activated": list(kinds)}, - }, - ), - ), - ) + name: str + cluster_name: str + want: str -def _want_activation(want: fnv1.RunFunctionResponse, kinds: tuple[str, ...]) -> None: - """Add the activation policy the function composes for a cloud cluster.""" - want.desired.resources["activation"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": _ACTIVATION_API_VERSION, - "kind": "ManagedResourceActivationPolicy", - "spec": {"activate": list(kinds)}, - }, - ), - ready=fnv1.READY_TRUE, +def _inference_cluster(*, cluster: v1alpha1.Cluster, node_pools: list[v1alpha1.NodePool] | None) -> fnv1.Resource: + """The observed InferenceCluster XR, test-cluster, with node_pools unless they're None.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceCluster( + metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), + spec=v1alpha1.Spec(cluster=cluster, nodePools=node_pools), + ).model_dump(exclude_none=True, mode="json", by_alias=True) ), ) -def _eks_ready_extras(want: fnv1.RunFunctionResponse, storage_class: str) -> None: - """Apply the EKS-ready deltas on top of the EKS first-pass response: mark the - EKSCluster ready and relay the backing cluster's status.cache up to the - InferenceCluster's status.cache.storageClassName.""" - want.desired.resources["eks-cluster"].ready = fnv1.READY_TRUE - status = want.desired.composite.resource.fields["status"].struct_value - status.fields["cache"].struct_value.fields["storageClassName"].string_value = storage_class +def _desired_inference_cluster(*, gpu_pools: list[dict], cache: dict | None, gateway: dict | None) -> fnv1.Resource: + """The desired InferenceCluster XR's status, with its cache and gateway unless they're None.""" + status: dict = { + "providerConfigRef": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "namespace": "modelplane-system", + "gpuPools": gpu_pools, + } + if cache is not None: + status["cache"] = cache + if gateway is not None: + status["gateway"] = gateway + return fnv1.Resource(resource=resource.dict_to_struct({"status": status})) -def _gateways_selector() -> fnv1.ResourceSelector: - """Every InferenceGateway. A cluster gateway accepts client certificates - from each of their CAs, which is how an InferenceGateway proves itself.""" - return fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway") +def _inference_class(*, name: str, count: int, memory: str, provisioning: dict) -> fnv1.Resource: + """An InferenceClass of count DRA-claimed NVIDIA GPUs with memory each, as its class requirement returns it.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceClass", + "metadata": {"name": name}, + "spec": { + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": count, + "capacity": {"memory": {"value": memory}}, + }, + ], + "provisioning": provisioning, + }, + } + ) + ) -def _replicas_selector(cluster_name: str) -> fnv1.ResourceSelector: - """The ModelReplica guard requirement: replicas scheduled to a cluster, - across all namespaces.""" - sel = fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica") - sel.match_labels.labels.update({"modelplane.ai/cluster": cluster_name}) - return sel +def _inference_gateway(*, name: str, cluster: str, status: dict | None) -> fnv1.Resource: + """An InferenceGateway on cluster, as the gateways requirement returns it, with status unless it's None.""" + gateway: dict = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceGateway", + "metadata": {"name": name}, + "spec": {"clusterName": cluster}, + } + if status is not None: + gateway["status"] = status + return fnv1.Resource(resource=resource.dict_to_struct(gateway)) -def _routes_selector(cluster_name: str) -> fnv1.ResourceSelector: - """The ModelRoute requirement: routes scheduled to a cluster, across all - namespaces. Their teams' namespaces are mirrored alongside the replicas'.""" - sel = fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelRoute") - sel.match_labels.labels.update({"modelplane.ai/cluster": cluster_name}) - return sel +def _model_cache(*, name: str, namespace: str, cluster: str) -> fnv1.Resource: + """A ModelCache staged onto cluster, as the model-caches requirement returns it.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelCache", + "metadata": {"name": name, "namespace": namespace}, + "spec": {"source": "HuggingFace"}, + "status": {"clusters": [{"name": cluster, "phase": "Ready"}]}, + } + ) + ) -def _caches_selector() -> fnv1.ResourceSelector: - """Every ModelCache. A cache fans out to many clusters, so it can't be - label-selected to one; the function filters by status.clusters[].""" - return fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache") +def _observed_activation_policy(*, activated: list[str]) -> fnv1.Resource: + """The observed ManagedResourceActivationPolicy, reporting the kinds it has activated.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "apiextensions.crossplane.io/v1alpha1", + "kind": "ManagedResourceActivationPolicy", + "status": {"activated": activated}, + } + ) + ) -def _replica_item(name: str, namespace: str) -> fnv1.Resource: - """An observed ModelReplica labelled for test-cluster.""" +def _observed_serving_stack(*, gateway: dict) -> fnv1.Resource: + """The observed ServingStack, Ready, with the gateway status it has published.""" return fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": name, - "namespace": namespace, - "labels": {"modelplane.ai/cluster": "test-cluster"}, - }, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": {"name": "test-cluster-serving-stack-fd00b"}, + "status": {"conditions": [{"type": "Ready", "status": "True"}], "gateway": gateway}, } ) ) -def _route_item(name: str, namespace: str) -> fnv1.Resource: - """An observed ModelRoute labelled for test-cluster.""" +def _activation_policy(*, activate: list[str], ready: fnv1.Ready) -> fnv1.Resource: + """The composed ManagedResourceActivationPolicy, activating a cloud's managed resource kinds.""" return fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelRoute", - "metadata": { - "name": name, - "namespace": namespace, - "labels": {"modelplane.ai/cluster": "test-cluster"}, - }, + "apiVersion": "apiextensions.crossplane.io/v1alpha1", + "kind": "ManagedResourceActivationPolicy", + "spec": {"activate": activate}, } - ) + ), + ready=ready, ) -def _cache_item(name: str, namespace: str, clusters: list[str]) -> fnv1.Resource: - """An observed ModelCache staging onto the named clusters (status.clusters). +def _gke_cluster(*, credentials: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The composed GKECluster with l4-pool, using credentials unless they're None.""" + spec: dict = { + "region": "us-central1", + "kubernetesVersion": "1.35", + "nodePools": [ + { + "name": "l4-pool", + "role": "GPU", + "machineType": "g2-standard-48", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": {"acceleratorType": "nvidia-l4", "acceleratorCount": 1}, + "zones": ["us-central1-a"], + }, + ], + } + if credentials is not None: + spec["credentials"] = credentials + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": spec, + } + ), + ready=ready, + ) + - Caches carry no cluster label; the function reads status.clusters[] to tell - which cluster a cache lands on, so only those naming this one are mirrored.""" +def _eks_cluster( + *, zones: list[str], capacity_block: dict | None, fabric: str | None, ready: fnv1.Ready +) -> fnv1.Resource: + """The composed EKSCluster with l4-pool in zones, setting its capacityBlock and fabric unless they're None.""" + pool: dict = { + "name": "l4-pool", + "role": "GPU", + "instanceType": "g6.xlarge", + "nodeCount": 2, + "minNodeCount": None, + "maxNodeCount": 4, + "diskSizeGb": 100, + "gpu": {"acceleratorType": "nvidia-l4"}, + "zones": zones, + } + if capacity_block is not None: + pool["capacityBlock"] = capacity_block + if fabric is not None: + pool["fabric"] = fabric return fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelCache", - "metadata": {"name": name, "namespace": namespace}, - "spec": {"source": "HuggingFace"}, - "status": {"clusters": [{"name": c, "phase": "Ready"} for c in clusters]}, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "EKSCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": {"region": "us-west-2", "kubernetesVersion": "1.36", "nodePools": [pool]}, } - ) + ), + ready=ready, ) -def _namespace_object(ns: str, name: str) -> fnv1.Resource: - """The mirrored namespace the function composes for a team with a replica or - route on the cluster, labelled for the gateways' route selector and kept (no - Delete). Marked ready so a new team's namespace can't flap the cluster's - readiness. +def _vultr_cluster(*, credentials: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The composed VultrCluster with l40s-pool, using credentials unless they're None.""" + spec: dict = { + "region": "ewr", + "kubernetesVersion": "v1.36.2+1", + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "plan": "vcg-l40s-16c-180g-48vram", + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": {"acceleratorType": "nvidia-l40s"}, + }, + ], + } + if credentials is not None: + spec["credentials"] = credentials + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "VultrCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": spec, + } + ), + ready=ready, + ) + - name is spelled out rather than computed with child_name, since the other - functions that land objects in it hardcode the same derivation, and a test - computing it the same way would pass whichever way any of them drifted.""" +def _cluster_provider_config(*, kubeconfig: str, identity: dict | None) -> fnv1.Resource: + """The ClusterProviderConfig reaching test-cluster with the kubeconfig Secret, as identity unless it's None.""" + spec: dict = { + "credentials": { + "source": "Secret", + "secretRef": {"namespace": "modelplane-system", "name": kubeconfig, "key": "kubeconfig"}, + }, + } + if identity is not None: + spec["identity"] = identity return fnv1.Resource( resource=resource.dict_to_struct( { "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { - "kind": "ClusterProviderConfig", - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "v1", - "kind": "Namespace", - "metadata": {"name": name, "labels": {"modelplane.ai/namespace": ns}}, - }, - }, - }, + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": spec, } ), ready=fnv1.READY_TRUE, ) -def _gateway_item(name: str, cluster: str) -> fnv1.Resource: - """An observed InferenceGateway running on the named cluster, with no client - CA published yet so it doesn't change the ServingStack's gateway spec.""" +def _serving_stack( + *, cloud: str, secrets: list[dict], client_cas: list[dict] | None, ready: fnv1.Ready +) -> fnv1.Resource: + """The composed ServingStack, its gateway accepting client_cas unless they're None.""" + gateway: dict = {"hostname": "gateway-test-cluster-09532.modelplane-system.svc.cluster.local"} + if client_cas is not None: + gateway["clientCAs"] = client_cas return fnv1.Resource( resource=resource.dict_to_struct( { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceGateway", - "metadata": {"name": name}, - "spec": {"clusterName": cluster}, + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": {"name": "test-cluster-serving-stack-fd00b", "namespace": "modelplane-system"}, + "spec": {"cloud": cloud, "gateway": gateway, "stack": "Standard", "secrets": secrets}, } - ) + ), + ready=ready, + ) + + +def _backend_usage(*, cluster_kind: str) -> fnv1.Resource: + """The composed Usage holding the cluster_kind cluster XR until the ServingStack is gone.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "of": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": cluster_kind, + "resourceSelector": {"matchControllerRef": True}, + }, + "by": { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "resourceSelector": {"matchControllerRef": True}, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, ) -def _guard_clusterusage(reason: str) -> fnv1.Resource: - """The reason-only ClusterUsage the guard composes for test-cluster.""" +def _guard_clusterusage(*, reason: str) -> fnv1.Resource: + """The reason-only ClusterUsage the deletion guard composes for test-cluster.""" return fnv1.Resource( resource=resource.dict_to_struct( { @@ -241,2952 +357,4067 @@ def _guard_clusterusage(reason: str) -> fnv1.Resource: ) -def _guard_case( - base_req: fnv1.RunFunctionRequest, base_want: fnv1.RunFunctionResponse -) -> tuple[fnv1.RunFunctionRequest, fnv1.RunFunctionResponse]: - """Build the guard-and-namespaces case from a base request and response. - - Observes ModelReplicas, ModelRoutes and ModelCaches across several - namespaces, so the function composes a single reason-only ClusterUsage - blocking the InferenceCluster's deletion, whatever their count or namespace, - and mirrors the deduplicated union of their namespaces: team-a (replica), - team-b (replica and route), team-c (route), team-d (cache staging onto this - cluster). A cache staging only onto another cluster (team-e) is filtered out - by its status.clusters[], proving the namespaces track what actually lands - here. The guard's reason names every kind in use. - """ - req = fnv1.RunFunctionRequest() - req.CopyFrom(base_req) - req.required_resources["model-replicas"].items.append(_replica_item("deploy-test-cluster-0", "team-a")) - req.required_resources["model-replicas"].items.append(_replica_item("deploy-test-cluster-0", "team-b")) - req.required_resources["model-routes"].items.append(_route_item("svc-eu", "team-b")) - req.required_resources["model-routes"].items.append(_route_item("svc-eu", "team-c")) - req.required_resources["model-caches"].items.append(_cache_item("qwen", "team-d", ["test-cluster"])) - req.required_resources["model-caches"].items.append(_cache_item("kimi", "team-e", ["other-cluster"])) - - want = fnv1.RunFunctionResponse() - want.CopyFrom(base_want) - want.desired.resources["usage-replicas"].CopyFrom( - _guard_clusterusage("ModelReplicas, ModelRoutes and ModelCaches use this InferenceCluster") - ) - want.desired.resources["namespace-team-a"].CopyFrom(_namespace_object("team-a", "mp-team-a-bd964")) - want.desired.resources["namespace-team-b"].CopyFrom(_namespace_object("team-b", "mp-team-b-6bd62")) - want.desired.resources["namespace-team-c"].CopyFrom(_namespace_object("team-c", "mp-team-c-d79d9")) - want.desired.resources["namespace-team-d"].CopyFrom(_namespace_object("team-d", "mp-team-d-c2383")) - return req, want - - -def _route_guard_case( - base_req: fnv1.RunFunctionRequest, base_want: fnv1.RunFunctionResponse -) -> tuple[fnv1.RunFunctionRequest, fnv1.RunFunctionResponse]: - """A ModelRoute on the cluster composes the guard on its own. - - A route composes its routing Objects through the cluster's - ClusterProviderConfig, so it blocks deletion without any replica there, and - its team's namespace is mirrored. - """ - req = fnv1.RunFunctionRequest() - req.CopyFrom(base_req) - req.required_resources["model-routes"].items.append(_route_item("svc-eu", "team-c")) - - want = fnv1.RunFunctionResponse() - want.CopyFrom(base_want) - want.desired.resources["usage-replicas"].CopyFrom(_guard_clusterusage("ModelRoutes use this InferenceCluster")) - want.desired.resources["namespace-team-c"].CopyFrom(_namespace_object("team-c", "mp-team-c-d79d9")) - return req, want - - -def _cache_guard_case( - base_req: fnv1.RunFunctionRequest, base_want: fnv1.RunFunctionResponse -) -> tuple[fnv1.RunFunctionRequest, fnv1.RunFunctionResponse]: - """A ModelCache staging onto the cluster composes the guard on its own. - - A cache composes its PVC through the cluster's ClusterProviderConfig, so it - blocks deletion without any replica there, and its team's namespace is - mirrored. - """ - req = fnv1.RunFunctionRequest() - req.CopyFrom(base_req) - req.required_resources["model-caches"].items.append(_cache_item("qwen", "team-d", ["test-cluster"])) - - want = fnv1.RunFunctionResponse() - want.CopyFrom(base_want) - want.desired.resources["usage-replicas"].CopyFrom(_guard_clusterusage("ModelCaches use this InferenceCluster")) - want.desired.resources["namespace-team-d"].CopyFrom(_namespace_object("team-d", "mp-team-d-c2383")) - return req, want - - -def _gateway_guard_case( - base_req: fnv1.RunFunctionRequest, base_want: fnv1.RunFunctionResponse -) -> tuple[fnv1.RunFunctionRequest, fnv1.RunFunctionResponse]: - """An InferenceGateway running on the cluster composes the guard on its own. - - A gateway composes its Gateway and routing Objects through the cluster's - ClusterProviderConfig just as a replica does, so it blocks deletion too. It - is cluster scoped, so it mirrors no namespace. - """ - req = fnv1.RunFunctionRequest() - req.CopyFrom(base_req) - req.required_resources["gateways"].items.append(_gateway_item("public", "test-cluster")) - - want = fnv1.RunFunctionResponse() - want.CopyFrom(base_want) - want.desired.resources["usage-replicas"].CopyFrom( - _guard_clusterusage("InferenceGateways use this InferenceCluster") - ) - return req, want - - -def _unused_case( - base_req: fnv1.RunFunctionRequest, base_want: fnv1.RunFunctionResponse -) -> tuple[fnv1.RunFunctionRequest, fnv1.RunFunctionResponse]: - """Every guard requirement resolved to nothing on this cluster: no ClusterUsage. - - This is the teardown transition - the last user is gone, so the function - stops composing the guard and the cluster becomes deletable. A cache staging - onto another cluster and a gateway running on another cluster don't hold it. - base_want must not contain usage-replicas. - """ - req = fnv1.RunFunctionRequest() - req.CopyFrom(base_req) - # Empty-but-present requirements, as Crossplane returns when a selector - # matched nothing. - req.required_resources["model-replicas"].ClearField("items") - req.required_resources["model-routes"].ClearField("items") - req.required_resources["model-caches"].items.append(_cache_item("kimi", "team-e", ["other-cluster"])) - req.required_resources["gateways"].items.append(_gateway_item("elsewhere", "other-cluster")) - return req, base_want - - -def _early_return_guard_case() -> tuple[fnv1.RunFunctionRequest, fnv1.RunFunctionResponse]: - """The guard is composed even when compose() returns early. - - resolve_classes() returns False whenever a referenced InferenceClass isn't - observed yet - a routine transient. The function returns before composing - the cluster, but the guard runs first, so a referencing replica still blocks - deletion. This is the case that regresses if the guard is gated behind class - resolution or cluster source. - """ - xr = v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), - ), - nodePools=[v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4)], - ), - ) - # The class requirement is declared but not fulfilled, so resolve_classes - # gates and compose() returns early. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json"))) - ), - ) - req.required_resources["model-replicas"].items.append(_replica_item("deploy-test-cluster-0", "team-a")) - - want = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - # The guard and namespace are marked ready, so the XR is marked not - # ready while it waits for its classes. - composite=fnv1.Resource(ready=fnv1.READY_FALSE), - resources={ - "usage-replicas": _guard_clusterusage("ModelReplicas use this InferenceCluster"), - "namespace-team-a": _namespace_object("team-a", "mp-team-a-bd964"), - }, +def _namespace_object(*, team: str, name: str) -> fnv1.Resource: + """The composed Object that mirrors team's namespace onto the cluster as the Namespace name.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "test-cluster-cluster-kubeconfig-d0f89", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Namespace", + "metadata": {"name": name, "labels": {"modelplane.ai/namespace": team}}, + }, + }, + }, + } ), - context=structpb.Struct(), - ) - want.requirements.resources["gateways"].CopyFrom(_gateways_selector()) - want.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) - want.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) - want.requirements.resources["model-caches"].CopyFrom(_caches_selector()) - want.requirements.resources["class-gpu-l4"].CopyFrom( - fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4") - ) - want.conditions.append( - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForClasses", - message="Waiting for InferenceClasses: gpu-l4", - ) + ready=fnv1.READY_TRUE, ) - want.results.append(fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for InferenceClasses: gpu-l4")) - return req, want -def _compose_cases() -> list[Case]: # noqa: PLR0915 - """The RunFunction cases, built from shared bases. - - Many table entries, each exercising a distinct compose path across the - GKE, EKS, and Existing sources, push this over the statement limit. - """ - # Shared InferenceClass resource for required_resources. - inference_class_l4 = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-l4"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], - "provisioning": { - "provider": "GKE", - "gke": { - "machineType": "g2-standard-48", - "diskSizeGb": 100, - "accelerator": { - "type": "nvidia-l4", - "count": 1, - }, - }, - }, - }, - } +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - # Shared resource selector for class requirement. - class_selector = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-l4", - ) - # --- Case 1: Existing cluster with secrets composes backend and CPC. --- - req1 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing( - secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), - ), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - ), - ], - ), - ).model_dump(exclude_none=True, mode="json") +# Every want requires what the function reads, before it composes anything: the +# ModelReplicas and ModelRoutes labelled for test-cluster, across all +# namespaces; every ModelCache, since a cache fans out to many clusters and so +# can't be label-selected to one, leaving the function to filter by +# status.clusters[]; every InferenceGateway, since the cluster gateway accepts +# client certificates from each of their CAs, which is how an InferenceGateway +# proves itself; and the InferenceClass behind each node pool. +# +# A cloud cluster's case observes its ManagedResourceActivationPolicy with every +# kind in status.activated, so the function composes the cluster XR rather than +# waiting for activation, unless the case says otherwise. +# +# gateway-test-cluster-09532.modelplane-system.svc.cluster.local is the internal +# name Modelplane derives for test-cluster's gateway, which +# compose-inference-gateway resolves. +COMPOSE_CASES = [ + ComposeCase( + name="existing cluster with secrets composes backend and CPC", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], ), ), - ), - ) - req1.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - "namespace": "modelplane-system", - "gpuPools": [ + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - } + ], + cache=None, + gateway=None, ), - ), - resources={ - "cluster-provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "my-kubeconfig", - "key": "kubeconfig", - }, - }, - }, - } + resources={ + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None ), - ready=fnv1.READY_TRUE, - ), - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Existing", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "my-kubeconfig", - "key": "kubeconfig", - }, - ], - }, - } + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + ready=fnv1.READY_UNSPECIFIED, ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", + }, ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, ), - ], - context=structpb.Struct(), - ) - want1.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - - # --- Case 1b: Existing cluster with a non-GCP identity threads the - # declared identity type into the CPC and the ServingStack. --- - req1b = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing( - secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), - identitySecretRef=v1alpha1.IdentitySecretRef( - name="nebius-creds", - key="credentials.json", - type="NebiusServiceAccountCredentials", - ), - ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], + ), + ), + # A non-GCP identity threads the declared identity type into the CPC, and + # into an extra ServingStack identity secret of the same type. + ComposeCase( + name="existing cluster with a non-GCP identity threads the identity type", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing( + secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), + identitySecretRef=v1alpha1.IdentitySecretRef( + name="nebius-creds", + key="credentials.json", + type="NebiusServiceAccountCredentials", ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - ), - ], ), - ).model_dump(exclude_none=True, mode="json") + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], ), ), - ), - ) - req1b.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - # want1b mirrors want1 but with the Nebius identity on the CPC and an - # extra ServingStack identity secret of the same type. - want1b = fnv1.RunFunctionResponse() - want1b.CopyFrom(want1) - cpc1b = want1b.desired.resources["cluster-provider-config-kubernetes"] - cpc1b_dict = resource.struct_to_dict(cpc1b.resource) - cpc1b_dict["spec"]["identity"] = { - "type": "NebiusServiceAccountCredentials", - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "nebius-creds", - "key": "credentials.json", - }, - } - cpc1b.resource.CopyFrom(resource.dict_to_struct(cpc1b_dict)) - backend1b = want1b.desired.resources["serving-stack"] - backend1b_dict = resource.struct_to_dict(backend1b.resource) - backend1b_dict["spec"]["secrets"].append( - { - "type": "NebiusServiceAccountCredentials", - "name": "nebius-creds", - "key": "credentials.json", - } - ) - backend1b.resource.CopyFrom(resource.dict_to_struct(backend1b_dict)) - # want1 gains the replica, route, cache and gateway requirements in place - # from the guard cases below, after this snapshot; add them here so want1b - # matches on its own. - want1b.requirements.resources["gateways"].CopyFrom(_gateways_selector()) - want1b.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) - want1b.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) - want1b.requirements.resources["model-caches"].CopyFrom(_caches_selector()) - - # --- Case 2: GKE cluster first pass - no observed GKE, classes resolved. --- - req2 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="GKE", - gke=v1alpha1.Gke( - region="us-central1", - ), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - zones=["us-central1-a"], - ), - ], + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, ), - ).model_dump(exclude_none=True, mode="json") + ], ), - ), + }, ), - ) - req2.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - } + ], + cache=None, + gateway=None, ), - ), - resources={ - "gke-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", + resources={ + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", + identity={ + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "nebius-creds", + "key": "credentials.json", }, - "spec": { - "region": "us-central1", - "kubernetesVersion": "1.35", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "machineType": "g2-standard-48", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - "acceleratorCount": 1, - }, - "zones": ["us-central1-a"], - }, - ], + }, + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[ + {"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}, + { + "type": "NebiusServiceAccountCredentials", + "name": "nebius-creds", + "key": "credentials.json", }, - } + ], + client_cas=None, + ready=fnv1.READY_UNSPECIFIED, ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", + }, ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, ), - ], - context=structpb.Struct(), - ) - want2.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - - # --- Case 3: Existing cluster second pass - backend observed ready. --- - req3 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing( - secretRef=v1alpha1.SecretRef(name="my-kubeconfig"), - ), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - ), - ], + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], + ), + ), + # The first pass: no GKECluster observed yet, and the classes resolved. + ComposeCase( + name="GKE cluster first pass composes only the policy and the GKECluster XR", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="GKE", gke=v1alpha1.Gke(region="us-central1")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], ), - ).model_dump(exclude_none=True, mode="json") + ], ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", + ] + ), + }, ), - resources={ - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": {"name": "test-cluster-serving-stack-fd00b"}, - "status": { - "conditions": [{"type": "Ready", "status": "True"}], - "gateway": {"address": "34.55.100.10"}, + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - } - ), + ), + ], ), }, ), - ) - req3.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], - "gateway": {"address": "34.55.100.10"}, }, - } + ], + cache=None, + gateway=None, ), + resources={ + "activation": _activation_policy( + activate=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "gke-cluster": _gke_cluster(credentials=None, ready=fnv1.READY_UNSPECIFIED), + }, ), - resources={ - "cluster-provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "my-kubeconfig", - "key": "kubeconfig", - }, - }, - }, - } + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), ), - ready=fnv1.READY_TRUE, - ), - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Existing", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "my-kubeconfig", - "key": "kubeconfig", - }, - ], - }, - } + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" ), - ready=fnv1.READY_TRUE, - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="BackendHealthy", - ), - ], - context=structpb.Struct(), - ) - want3.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - - # --- Case 4: EKS cluster first pass - no observed EKS, classes resolved. --- - inference_class_l4_eks = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-l4-eks"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), ], - "provisioning": { - "provider": "EKS", - "eks": { - "instanceType": "g6.xlarge", - "diskSizeGb": 100, - "accelerator": {"type": "nvidia-l4", "count": 1}, - }, - }, - }, - } - class_selector_eks = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-l4-eks", - ) - - req4 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", + ), + ), + ComposeCase( + name="GKE credentials pass through to GKECluster spec", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="GKE", + gke=v1alpha1.Gke( + region="us-central1", + credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-gcp-account"), ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="EKS", - eks=v1alpha1.Eks(region="us-west-2"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4-eks", - nodeCount=2, - maxNodeCount=4, - zones=["us-west-2a", "us-west-2b"], - ), - ], + ), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], ), - ).model_dump(exclude_none=True, mode="json"), + ], ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", + ] + ), + }, ), - ), - ) - req4.required_resources["class-gpu-l4-eks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), - ) - - want4 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - "namespace": "modelplane-system", - "gpuPools": [ + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - }, + ], + cache=None, + gateway=None, ), + resources={ + "activation": _activation_policy( + activate=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "gke-cluster": _gke_cluster( + credentials={"type": "ProviderConfig", "name": "my-gcp-account"}, ready=fnv1.READY_UNSPECIFIED + ), + }, ), - resources={ - "eks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-west-2", - "kubernetesVersion": "1.36", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - }, - "zones": ["us-west-2a", "us-west-2b"], - }, - ], - }, - }, + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], + ), + ), + # The second pass: the ServingStack is observed ready, with a gateway + # address. + ComposeCase( + name="existing cluster second pass with backend ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], + ), + resources={ + "serving-stack": _observed_serving_stack(gateway={"address": "34.55.100.10"}), + }, ), - ], - context=structpb.Struct(), - ) - want4.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) - - # --- Case 8: EKS first pass with a node pool backed by a Capacity - # Block. The reservation ID flows through to the EKSCluster node pool's - # capacityBlock, which compose-eks-cluster turns into a CAPACITY_BLOCK - # node group. --- - req8 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="EKS", - eks=v1alpha1.Eks(region="us-west-2"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4-eks", - nodeCount=2, - maxNodeCount=4, - zones=["us-west-2a"], - capacityBlock=v1alpha1.CapacityBlock( - capacityReservationId="cr-0123456789abcdef0", - ), - ), - ], + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, ), - ).model_dump(exclude_none=True, mode="json"), + ], ), - ), + }, ), - ) - req8.required_resources["class-gpu-l4-eks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), - ) - - want8 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - }, + ], + cache=None, + gateway={"address": "34.55.100.10"}, ), - ), - resources={ - "eks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-west-2", - "kubernetesVersion": "1.36", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - }, - "zones": ["us-west-2a"], - "capacityBlock": { - "capacityReservationId": "cr-0123456789abcdef0", - }, - }, - ], - }, - }, + resources={ + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + ready=fnv1.READY_TRUE, + ), + }, ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, ), - ], - context=structpb.Struct(), - ) - want8.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) - - # --- Case 9: EKS first pass with a node pool that opts into the EFA - # fabric. fabric.type flows through to the EKSCluster node pool, which - # compose-eks-cluster turns into EFA launch-template interfaces. --- - req9 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="EKS", - eks=v1alpha1.Eks(region="us-west-2"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4-eks", - nodeCount=2, - maxNodeCount=4, - zones=["us-west-2a"], - fabric=v1alpha1.Fabric(type="EFA"), - ), - ], + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_TRUE, reason="BackendHealthy"), + ], + ), + ), + # The first pass: no EKSCluster observed yet, and the classes resolved. + ComposeCase( + name="EKS cluster first pass composes only the policy and the EKSCluster XR", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="EKS", eks=v1alpha1.Eks(region="us-west-2")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a", "us-west-2b"], ), - ).model_dump(exclude_none=True, mode="json"), + ], ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ] + ), + }, ), - ), - ) - req9.required_resources["class-gpu-l4-eks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4_eks)), - ) - - want9 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + required_resources={ + "class-gpu-l4-eks": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4-eks", + count=1, + memory="24Gi", + provisioning={ + "provider": "EKS", + "eks": { + "instanceType": "g6.xlarge", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - "namespace": "modelplane-system", - "gpuPools": [ + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - }, + ], + cache=None, + gateway=None, ), - ), - resources={ - "eks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-west-2", - "kubernetesVersion": "1.36", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - }, - "zones": ["us-west-2a"], - "fabric": "EFA", - }, - ], - }, - }, + resources={ + "activation": _activation_policy( + activate=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "eks-cluster": _eks_cluster( + zones=["us-west-2a", "us-west-2b"], + capacity_block=None, + fabric=None, + ready=fnv1.READY_UNSPECIFIED, ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want9.requirements.resources["class-gpu-l4-eks"].CopyFrom(class_selector_eks) - - # --- Case 5: EKS cluster not yet ready (no kubeconfig observed) but a - # ClusterProviderConfig already exists from a prior reconcile. The CPC - # is built only from the kubeconfig, so without one it's simply omitted - # from desired state this reconcile (and recreated once the kubeconfig - # is observed again) - it is never emitted with an empty secretRef. - observed_cpc = { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, - }, - } - req5 = fnv1.RunFunctionRequest() - req5.CopyFrom(req4) - req5.observed.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource(resource=resource.dict_to_struct(observed_cpc)), - ) - - # Desired state is identical to case 4: no ClusterProviderConfig. - want5 = fnv1.RunFunctionResponse() - want5.CopyFrom(want4) - - # --- Case 6: GKE cluster ready - composes CPC, backend, usage, and the - # VPC-pinned modelplane-rwx Filestore StorageClass on the workload - # cluster (default cache storage class). --- - observed_gke_ready = { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "region": "us-central1", - "nodePools": [{"name": "system", "role": "System", "machineType": "e2-standard-4"}], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2026-06-08T00:00:00Z", }, - ], - # The backing GKECluster reports its effective RWX StorageClass; - # the InferenceCluster relays it up to its own status.cache. - "cache": {"storageClassName": "modelplane-rwx"}, - "secrets": [ - {"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}, - { - "type": "GoogleApplicationCredentials", - "name": "test-cluster-sa-key-fghij", - "key": "credentials.json", + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4-eks": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4-eks" + ), }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), ], - }, - } - req6 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="GKE", - gke=v1alpha1.Gke( - region="us-central1", - ), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - zones=["us-central1-a"], - ), - ], + ), + ), + # The EKSCluster isn't observed yet, so neither is its kubeconfig, but a + # ClusterProviderConfig is observed from a prior reconcile. The CPC is built + # only from the kubeconfig, so without one it's left out of desired state + # this reconcile, and recreated once the kubeconfig is observed again. It is + # never emitted with an empty secretRef. + ComposeCase( + name="EKS cluster not ready omits the observed CPC", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="EKS", eks=v1alpha1.Eks(region="us-west-2")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a", "us-west-2b"], ), - ).model_dump(exclude_none=True, mode="json") + ], ), + resources={ + "cluster-provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ClusterProviderConfig", + "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + }, + }, + } + ), + ), + "activation": _observed_activation_policy( + activated=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ] + ), + }, ), - resources={ - "gke-cluster": fnv1.Resource(resource=resource.dict_to_struct(observed_gke_ready)), + required_resources={ + "class-gpu-l4-eks": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4-eks", + count=1, + memory="24Gi", + provisioning={ + "provider": "EKS", + "eks": { + "instanceType": "g6.xlarge", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + ), + ], + ), }, ), - ) - req6.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - want6 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - } - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], - # Relayed from the backing GKECluster's status.cache. - "cache": {"storageClassName": "modelplane-rwx"}, }, - } + ], + cache=None, + gateway=None, ), + resources={ + "activation": _activation_policy( + activate=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "eks-cluster": _eks_cluster( + zones=["us-west-2a", "us-west-2b"], + capacity_block=None, + fabric=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, ), - resources={ - "gke-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-central1", - "kubernetesVersion": "1.35", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "machineType": "g2-standard-48", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - "acceleratorCount": 1, - }, - "zones": ["us-central1-a"], - }, - ], - }, - } + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), ), - ready=fnv1.READY_TRUE, - ), - "cluster-provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, - "identity": { - "type": "GoogleApplicationCredentials", - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-sa-key-fghij", - "key": "credentials.json", - }, - }, - }, - } + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), ), - ready=fnv1.READY_TRUE, + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4-eks": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4-eks" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], + ), + ), + # The GKECluster is observed ready with its secrets, so the function + # composes the CPC with the GKE service account identity, the ServingStack + # with both secrets, and the Usage that blocks GKECluster deletion until the + # ServingStack is gone. It relays the GKECluster's RWX StorageClass up to + # status.cache. + ComposeCase( + name="GKE cluster ready composes CPC, backend and usage, and relays its RWX StorageClass", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="GKE", gke=v1alpha1.Gke(region="us-central1")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], + ), + ], ), - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "GKE", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - { - "type": "GoogleApplicationCredentials", - "name": "test-cluster-sa-key-fghij", - "key": "credentials.json", - }, - ], - }, - } + resources={ + "gke-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "us-central1", + "nodePools": [{"name": "system", "role": "System", "machineType": "e2-standard-4"}], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + }, + ], + "cache": {"storageClassName": "modelplane-rwx"}, + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + { + "type": "GoogleApplicationCredentials", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", + }, + ], + }, + } + ), ), - ), - "usage-gke-by-backend": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } + "activation": _observed_activation_policy( + activated=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", + ] ), - ready=fnv1.READY_TRUE, + }, + ), + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + ), + ], ), }, ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="GKE cluster ready, composing backend", - ), - ], - context=structpb.Struct(), - ) - want6.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - - # --- Case 7: EKS cluster ready - kubeconfig observed on the EKSCluster - # status. The function wires the ClusterProviderConfig, composes the - # ServingStack backend, and emits the Usage that blocks EKSCluster - # deletion until the ServingStack is gone. --- - req7 = fnv1.RunFunctionRequest() - req7.CopyFrom(req4) - req7.observed.resources["eks-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "region": "us-west-2", - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "instanceType": "g6.xlarge", - "nodeCount": 2, - }, - ], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + cache={"storageClassName": "modelplane-rwx"}, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", ], - # The backing EKSCluster reports its effective RWX - # StorageClass; the InferenceCluster relays it up. - "cache": {"storageClassName": "modelplane-rwx-efs"}, - }, - } - ), - ), - ) - - want7 = fnv1.RunFunctionResponse() - want7.CopyFrom(want4) - # Mark the EKSCluster ready and relay its status.cache up to status.cache. - _eks_ready_extras(want7, "modelplane-rwx-efs") - want7.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { + ready=fnv1.READY_TRUE, + ), + "gke-cluster": _gke_cluster(credentials=None, ready=fnv1.READY_TRUE), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="test-cluster-kubeconfig-abcde", + identity={ + "type": "GoogleApplicationCredentials", "source": "Secret", "secretRef": { "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", }, }, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - ) - want7.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "EKS", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ + ), + "serving-stack": _serving_stack( + cloud="GKE", + secrets=[ + {"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}, { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + "type": "GoogleApplicationCredentials", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", }, ], - }, - } - ), - ), - ) - want7.desired.resources["usage-eks-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "EKSCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } + client_cas=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-gke-by-backend": _backend_usage(cluster_kind="GKECluster"), + }, ), - ready=fnv1.READY_TRUE, - ), - ) - del want7.conditions[:] - want7.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want7.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="EKS cluster ready, composing backend", - ) - ) - - # --- Case 10: Nebius first pass composes the NebiusCluster XR only. - # The pool's InfiniBand fabric flows through to the NebiusCluster - # pool's fabric, and minNodeCount stays unset so the pool's - # autoscaling floor defaults to its node count downstream. --- - inference_class_h100_nebius = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-h100-nebius"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="GKE cluster ready, composing backend")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), ], - "provisioning": { - "provider": "Nebius", - "nebius": { - "platform": "gpu-h100-sxm", - "preset": "8gpu-128vcpu-1600gb", - "diskSizeGb": 200, - "accelerator": {"type": "nvidia-h100", "count": 8}, - }, - }, - }, - } - class_selector_nebius = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-h100-nebius", - ) - - req10 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Nebius", - nebius=v1alpha1.Nebius(), - ), - nodePools=[ - v1alpha1.NodePool( - name="h100-pool", - className="gpu-h100-nebius", - nodeCount=2, - maxNodeCount=4, - fabric=v1alpha1.Fabric( - type="InfiniBand", - infiniband=v1alpha1.Infiniband(fabric="fabric-2"), - ), - ), - ], + ), + ), + # The kubeconfig is observed on the EKSCluster status. The function wires + # the ClusterProviderConfig, composes the ServingStack backend, and emits + # the Usage that blocks EKSCluster deletion until the ServingStack is gone. + # It marks the EKSCluster ready and relays its status.cache up to the + # InferenceCluster's status.cache. + ComposeCase( + name="EKS cluster ready composes ServingStack and Usage", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="EKS", eks=v1alpha1.Eks(region="us-west-2")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a", "us-west-2b"], ), - ).model_dump(exclude_none=True, mode="json"), + ], ), - ), - ), - ) - req10.required_resources["class-gpu-h100-nebius"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_nebius)), - ) - - want10 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ - { - "name": "h100-pool", - "nodes": 4, - "devices": [ + resources={ + "eks-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "EKSCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "us-west-2", + "nodePools": [ + {"name": "l4-pool", "role": "GPU", "instanceType": "g6.xlarge", "nodeCount": 2}, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", }, ], + "cache": {"storageClassName": "modelplane-rwx-efs"}, }, - ], - }, - }, - ), + } + ), + ), + "activation": _observed_activation_policy( + activated=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ] + ), + }, ), - resources={ - "nebius-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "NebiusCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "kubernetesVersion": "1.34", - "nodePools": [ - { - "name": "h100-pool", - "role": "GPU", - "platform": "gpu-h100-sxm", - "preset": "8gpu-128vcpu-1600gb", - "diskSizeGb": 200, - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-h100", - "driversPreset": "cuda13.0", - }, - "fabric": { - "type": "InfiniBand", - "infiniband": {"fabric": "fabric-2"}, - }, - }, - ], + required_resources={ + "class-gpu-l4-eks": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4-eks", + count=1, + memory="24Gi", + provisioning={ + "provider": "EKS", + "eks": { + "instanceType": "g6.xlarge", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - }, - ), + ), + ], ), }, ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - ], - context=structpb.Struct(), - ) - want10.requirements.resources["class-gpu-h100-nebius"].CopyFrom(class_selector_nebius) - - # --- Case 11: Nebius cluster ready - kubeconfig and service account - # credentials observed on the NebiusCluster status. The function wires - # the ClusterProviderConfig with the Nebius identity (the mk8s - # kubeconfig has no embedded credentials), composes the ServingStack - # backend with both secrets, and emits the Usage that blocks - # NebiusCluster deletion until the ServingStack is gone. --- - req11 = fnv1.RunFunctionRequest() - req11.CopyFrom(req10) - req11.observed.resources["nebius-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "NebiusCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "nodePools": [ - { - "name": "h100-pool", - "role": "GPU", - "platform": "gpu-h100-sxm", - "preset": "8gpu-128vcpu-1600gb", - "nodeCount": 2, - }, - ], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - # The credential entry carries a namespace: it - # is the Nebius ClusterProviderConfig's Secret, - # which lives outside modelplane-system. - { - "type": "NebiusServiceAccountCredentials", - "name": "nebius-credentials", - "key": "credentials.json", - "namespace": "crossplane-system", - }, - ], - }, - } - ), - ), - ) - - want11 = fnv1.RunFunctionResponse() - want11.CopyFrom(want10) - want11.desired.resources["nebius-cluster"].ready = fnv1.READY_TRUE - want11.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, - "identity": { - "type": "NebiusServiceAccountCredentials", - "source": "Secret", - "secretRef": { - "namespace": "crossplane-system", - "name": "nebius-credentials", - "key": "credentials.json", - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], }, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - ) - want11.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Nebius", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - { - "type": "NebiusServiceAccountCredentials", - "name": "nebius-credentials", - "key": "credentials.json", - "namespace": "crossplane-system", - }, + ], + cache={"storageClassName": "modelplane-rwx-efs"}, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", ], - }, - } - ), - ), - ) - want11.desired.resources["usage-nebius-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "NebiusCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } + ready=fnv1.READY_TRUE, + ), + "eks-cluster": _eks_cluster( + zones=["us-west-2a", "us-west-2b"], capacity_block=None, fabric=None, ready=fnv1.READY_TRUE + ), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="test-cluster-kubeconfig-abcde", identity=None + ), + "serving-stack": _serving_stack( + cloud="EKS", + secrets=[{"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}], + client_cas=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-eks-by-backend": _backend_usage(cluster_kind="EKSCluster"), + }, ), - ready=fnv1.READY_TRUE, - ), - ) - del want11.conditions[:] - want11.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want11.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Nebius cluster ready, composing backend", - ) - ) - - # --- Case 12: AKS first pass composes the AKSCluster XR only. The - # pool's InfiniBand fabric flows through to the AKSCluster pool as the - # plain fabric string - Azure has no user-selectable fabric ID. --- - inference_class_h100_aks = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-h100-aks"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="EKS cluster ready, composing backend")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4-eks": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4-eks" + ), }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), ], - "provisioning": { - "provider": "AKS", - "aks": { - "vmSize": "Standard_ND96isr_H100_v5", - "diskSizeGb": 200, - "accelerator": {"type": "nvidia-h100", "count": 8}, - }, - }, - }, - } - class_selector_aks = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-h100-aks", - ) - - req12 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="AKS", - aks=v1alpha1.Aks(location="westeurope"), + ), + ), + # The first pass with a node pool backed by a Capacity Block. The + # reservation ID flows through to the EKSCluster node pool's capacityBlock, + # which compose-eks-cluster turns into a CAPACITY_BLOCK node group. + ComposeCase( + name="EKS node pool with a Capacity Block sets capacityBlock on the EKSCluster pool", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="EKS", eks=v1alpha1.Eks(region="us-west-2")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a"], + capacityBlock=v1alpha1.CapacityBlock( + capacityReservationId="cr-0123456789abcdef0", ), - nodePools=[ - v1alpha1.NodePool( - name="h100pool", - className="gpu-h100-aks", - nodeCount=2, - # AKS GPU pools keep minNodeCount at - # 1: the AKS autoscaler can't scale a - # DRA pool up from zero nodes. - minNodeCount=1, - maxNodeCount=4, - fabric=v1alpha1.Fabric(type="InfiniBand"), - ), - ], ), - ).model_dump(exclude_none=True, mode="json"), + ], ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ] + ), + }, ), - ), - ) - req12.required_resources["class-gpu-h100-aks"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_h100_aks)), - ) - - want12 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + required_resources={ + "class-gpu-l4-eks": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4-eks", + count=1, + memory="24Gi", + provisioning={ + "provider": "EKS", + "eks": { + "instanceType": "g6.xlarge", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - "namespace": "modelplane-system", - "gpuPools": [ + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "h100pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 8, - "capacity": {"memory": {"value": "81559Mi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - }, + ], + cache=None, + gateway=None, ), + resources={ + "activation": _activation_policy( + activate=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "eks-cluster": _eks_cluster( + zones=["us-west-2a"], + capacity_block={"capacityReservationId": "cr-0123456789abcdef0"}, + fabric=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, ), - resources={ - "aks-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "AKSCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "location": "westeurope", - "kubernetesVersion": "1.34", - "nodePools": [ - { - "name": "h100pool", - "role": "GPU", - "vmSize": "Standard_ND96isr_H100_v5", - "diskSizeGb": 200, - "nodeCount": 2, - "minNodeCount": 1, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-h100", - }, - "fabric": "InfiniBand", - }, - ], - }, - }, + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4-eks": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4-eks" + ), + }, ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], + ), + ), + # The first pass with a node pool that opts into the EFA fabric. fabric.type + # flows through to the EKSCluster node pool, which compose-eks-cluster turns + # into EFA launch-template interfaces. + ComposeCase( + name="EKS node pool with fabric EFA sets fabric on the EKSCluster pool", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="EKS", eks=v1alpha1.Eks(region="us-west-2")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4-eks", + nodeCount=2, + maxNodeCount=4, + zones=["us-west-2a"], + fabric=v1alpha1.Fabric(type="EFA"), + ), + ], + ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ] + ), + }, ), - ], - context=structpb.Struct(), - ) - want12.requirements.resources["class-gpu-h100-aks"].CopyFrom(class_selector_aks) - - # --- Case 13: AKS cluster ready - kubeconfig observed on the - # AKSCluster status. The kubeconfig embeds a client certificate, so - # the ClusterProviderConfig carries no identity (unlike GKE/Nebius). - # The function composes the ServingStack backend and the Usage that - # blocks AKSCluster deletion until the ServingStack is gone, and - # relays the AKSCluster's status.cache up to status.cache. --- - req13 = fnv1.RunFunctionRequest() - req13.CopyFrom(req12) - req13.observed.resources["aks-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "AKSCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "location": "westeurope", - "nodePools": [ - { - "name": "h100pool", - "role": "GPU", - "vmSize": "Standard_ND96isr_H100_v5", - "nodeCount": 2, + required_resources={ + "class-gpu-l4-eks": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4-eks", + count=1, + memory="24Gi", + provisioning={ + "provider": "EKS", + "eks": { + "instanceType": "g6.xlarge", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - ], - }, - "status": { - "conditions": [ + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "eips.ec2.aws.m.upbound.io", + "internetgateways.ec2.aws.m.upbound.io", + "launchtemplates.ec2.aws.m.upbound.io", + "natgateways.ec2.aws.m.upbound.io", + "routes.ec2.aws.m.upbound.io", + "routetables.ec2.aws.m.upbound.io", + "routetableassociations.ec2.aws.m.upbound.io", + "securitygroups.ec2.aws.m.upbound.io", + "securitygroupegressrules.ec2.aws.m.upbound.io", + "securitygroupingressrules.ec2.aws.m.upbound.io", + "subnets.ec2.aws.m.upbound.io", + "vpcs.ec2.aws.m.upbound.io", + "filesystems.efs.aws.m.upbound.io", + "mounttargets.efs.aws.m.upbound.io", + "addons.eks.aws.m.upbound.io", + "clusters.eks.aws.m.upbound.io", + "clusterauths.eks.aws.m.upbound.io", + "nodegroups.eks.aws.m.upbound.io", + "podidentityassociations.eks.aws.m.upbound.io", + "policies.iam.aws.m.upbound.io", + "roles.iam.aws.m.upbound.io", + "rolepolicyattachments.iam.aws.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "eks-cluster": _eks_cluster( + zones=["us-west-2a"], capacity_block=None, fabric="EFA", ready=fnv1.READY_UNSPECIFIED + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4-eks": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4-eks" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], + ), + ), + # The first pass composes only the activation policy and the NebiusCluster + # XR. The pool's InfiniBand fabric flows through to the NebiusCluster pool's + # fabric, and minNodeCount stays unset so the pool's autoscaling floor + # defaults to its node count downstream. + ComposeCase( + name="Nebius cluster first pass composes only the policy and the NebiusCluster XR", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="Nebius", nebius=v1alpha1.Nebius()), + node_pools=[ + v1alpha1.NodePool( + name="h100-pool", + className="gpu-h100-nebius", + nodeCount=2, + maxNodeCount=4, + fabric=v1alpha1.Fabric( + type="InfiniBand", + infiniband=v1alpha1.Infiniband(fabric="fabric-2"), + ), + ), + ], + ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "filesystems.compute.nebius.m.upbound.io", + "gpuclusters.compute.nebius.m.upbound.io", + "clusters.mk8s.nebius.m.upbound.io", + "nodegroups.mk8s.nebius.m.upbound.io", + "networks.vpc.nebius.m.upbound.io", + "subnets.vpc.nebius.m.upbound.io", + ] + ), + }, + ), + required_resources={ + "class-gpu-h100-nebius": fnv1.Resources( + items=[ + _inference_class( + name="gpu-h100-nebius", + count=8, + memory="81559Mi", + provisioning={ + "provider": "Nebius", + "nebius": { + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "diskSizeGb": 200, + "accelerator": {"type": "nvidia-h100", "count": 8}, + }, + }, + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "h100-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "filesystems.compute.nebius.m.upbound.io", + "gpuclusters.compute.nebius.m.upbound.io", + "clusters.mk8s.nebius.m.upbound.io", + "nodegroups.mk8s.nebius.m.upbound.io", + "networks.vpc.nebius.m.upbound.io", + "subnets.vpc.nebius.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "nebius-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "NebiusCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "kubernetesVersion": "1.34", + "nodePools": [ + { + "name": "h100-pool", + "role": "GPU", + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "diskSizeGb": 200, + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": {"acceleratorType": "nvidia-h100", "driversPreset": "cuda13.0"}, + "fabric": {"type": "InfiniBand", "infiniband": {"fabric": "fabric-2"}}, + }, + ], + }, + } + ), + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-h100-nebius": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-h100-nebius" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], + ), + ), + # The kubeconfig and service account credentials are observed on the + # NebiusCluster status. The function wires the ClusterProviderConfig with + # the Nebius identity (the mk8s kubeconfig has no embedded credentials), + # composes the ServingStack backend with both secrets, and emits the Usage + # that blocks NebiusCluster deletion until the ServingStack is gone. The + # credentials Secret carries a namespace: it is the Nebius + # ClusterProviderConfig's Secret, which lives outside modelplane-system. + ComposeCase( + name="Nebius cluster ready composes CPC with Nebius identity, ServingStack, and Usage", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="Nebius", nebius=v1alpha1.Nebius()), + node_pools=[ + v1alpha1.NodePool( + name="h100-pool", + className="gpu-h100-nebius", + nodeCount=2, + maxNodeCount=4, + fabric=v1alpha1.Fabric( + type="InfiniBand", + infiniband=v1alpha1.Infiniband(fabric="fabric-2"), + ), + ), + ], + ), + resources={ + "nebius-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "NebiusCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "nodePools": [ + { + "name": "h100-pool", + "role": "GPU", + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "nodeCount": 2, + }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + { + "type": "NebiusServiceAccountCredentials", + "name": "nebius-credentials", + "key": "credentials.json", + "namespace": "crossplane-system", + }, + ], + }, + } + ), + ), + "activation": _observed_activation_policy( + activated=[ + "filesystems.compute.nebius.m.upbound.io", + "gpuclusters.compute.nebius.m.upbound.io", + "clusters.mk8s.nebius.m.upbound.io", + "nodegroups.mk8s.nebius.m.upbound.io", + "networks.vpc.nebius.m.upbound.io", + "subnets.vpc.nebius.m.upbound.io", + ] + ), + }, + ), + required_resources={ + "class-gpu-h100-nebius": fnv1.Resources( + items=[ + _inference_class( + name="gpu-h100-nebius", + count=8, + memory="81559Mi", + provisioning={ + "provider": "Nebius", + "nebius": { + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "diskSizeGb": 200, + "accelerator": {"type": "nvidia-h100", "count": 8}, + }, + }, + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "h100-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "filesystems.compute.nebius.m.upbound.io", + "gpuclusters.compute.nebius.m.upbound.io", + "clusters.mk8s.nebius.m.upbound.io", + "nodegroups.mk8s.nebius.m.upbound.io", + "networks.vpc.nebius.m.upbound.io", + "subnets.vpc.nebius.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "nebius-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "NebiusCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "kubernetesVersion": "1.34", + "nodePools": [ + { + "name": "h100-pool", + "role": "GPU", + "platform": "gpu-h100-sxm", + "preset": "8gpu-128vcpu-1600gb", + "diskSizeGb": 200, + "nodeCount": 2, + "maxNodeCount": 4, + "gpu": {"acceleratorType": "nvidia-h100", "driversPreset": "cuda13.0"}, + "fabric": {"type": "InfiniBand", "infiniband": {"fabric": "fabric-2"}}, + }, + ], + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="test-cluster-kubeconfig-abcde", + identity={ + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": { + "namespace": "crossplane-system", + "name": "nebius-credentials", + "key": "credentials.json", + }, + }, + ), + "serving-stack": _serving_stack( + cloud="Nebius", + secrets=[ + {"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}, + { + "type": "NebiusServiceAccountCredentials", + "name": "nebius-credentials", + "key": "credentials.json", + "namespace": "crossplane-system", + }, + ], + client_cas=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-nebius-by-backend": _backend_usage(cluster_kind="NebiusCluster"), + }, + ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Nebius cluster ready, composing backend")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-h100-nebius": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-h100-nebius" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], + ), + ), + # The first pass composes only the activation policy and the AKSCluster XR. + # The pool's InfiniBand fabric flows through to the AKSCluster pool as the + # plain fabric string - Azure has no user-selectable fabric ID. The pool + # sets minNodeCount to 1, as an AKS GPU pool must, because the AKS + # autoscaler can't scale a DRA pool up from zero nodes. + ComposeCase( + name="AKS cluster first pass composes only the policy and the AKSCluster XR", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="AKS", aks=v1alpha1.Aks(location="westeurope")), + node_pools=[ + v1alpha1.NodePool( + name="h100pool", + className="gpu-h100-aks", + nodeCount=2, + minNodeCount=1, + maxNodeCount=4, + fabric=v1alpha1.Fabric(type="InfiniBand"), + ), + ], + ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "kubernetesclusters.containerservice.azure.m.upbound.io", + "kubernetesclusternodepools.containerservice.azure.m.upbound.io", + "subnets.network.azure.m.upbound.io", + "virtualnetworks.network.azure.m.upbound.io", + "resourcegroups.azure.m.upbound.io", + ] + ), + }, + ), + required_resources={ + "class-gpu-h100-aks": fnv1.Resources( + items=[ + _inference_class( + name="gpu-h100-aks", + count=8, + memory="81559Mi", + provisioning={ + "provider": "AKS", + "aks": { + "vmSize": "Standard_ND96isr_H100_v5", + "diskSizeGb": 200, + "accelerator": {"type": "nvidia-h100", "count": 8}, + }, + }, + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "h100pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "kubernetesclusters.containerservice.azure.m.upbound.io", + "kubernetesclusternodepools.containerservice.azure.m.upbound.io", + "subnets.network.azure.m.upbound.io", + "virtualnetworks.network.azure.m.upbound.io", + "resourcegroups.azure.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "aks-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "AKSCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "location": "westeurope", + "kubernetesVersion": "1.34", + "nodePools": [ + { + "name": "h100pool", + "role": "GPU", + "vmSize": "Standard_ND96isr_H100_v5", + "diskSizeGb": 200, + "nodeCount": 2, + "minNodeCount": 1, + "maxNodeCount": 4, + "gpu": {"acceleratorType": "nvidia-h100"}, + "fabric": "InfiniBand", + }, + ], + }, + } + ), + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-h100-aks": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-h100-aks" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], + ), + ), + # The kubeconfig is observed on the AKSCluster status. It embeds a client + # certificate, so the ClusterProviderConfig carries no identity (unlike + # GKE and Nebius). The function composes the ServingStack backend and the + # Usage that blocks AKSCluster deletion until the ServingStack is gone, and + # relays the AKSCluster's status.cache up to status.cache. + ComposeCase( + name="AKS cluster ready composes CPC without identity, ServingStack, and Usage", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="AKS", aks=v1alpha1.Aks(location="westeurope")), + node_pools=[ + v1alpha1.NodePool( + name="h100pool", + className="gpu-h100-aks", + nodeCount=2, + minNodeCount=1, + maxNodeCount=4, + fabric=v1alpha1.Fabric(type="InfiniBand"), + ), + ], + ), + resources={ + "aks-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "AKSCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "location": "westeurope", + "nodePools": [ + { + "name": "h100pool", + "role": "GPU", + "vmSize": "Standard_ND96isr_H100_v5", + "nodeCount": 2, + }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + "cache": {"storageClassName": "modelplane-rwx-fs"}, + }, + } + ), + ), + "activation": _observed_activation_policy( + activated=[ + "kubernetesclusters.containerservice.azure.m.upbound.io", + "kubernetesclusternodepools.containerservice.azure.m.upbound.io", + "subnets.network.azure.m.upbound.io", + "virtualnetworks.network.azure.m.upbound.io", + "resourcegroups.azure.m.upbound.io", + ] + ), + }, + ), + required_resources={ + "class-gpu-h100-aks": fnv1.Resources( + items=[ + _inference_class( + name="gpu-h100-aks", + count=8, + memory="81559Mi", + provisioning={ + "provider": "AKS", + "aks": { + "vmSize": "Standard_ND96isr_H100_v5", + "diskSizeGb": 200, + "accelerator": {"type": "nvidia-h100", "count": 8}, + }, + }, + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "h100pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "capacity": {"memory": {"value": "81559Mi"}}, + }, + ], + }, + ], + cache={"storageClassName": "modelplane-rwx-fs"}, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "kubernetesclusters.containerservice.azure.m.upbound.io", + "kubernetesclusternodepools.containerservice.azure.m.upbound.io", + "subnets.network.azure.m.upbound.io", + "virtualnetworks.network.azure.m.upbound.io", + "resourcegroups.azure.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "aks-cluster": fnv1.Resource( + resource=resource.dict_to_struct( { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "AKSCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "location": "westeurope", + "kubernetesVersion": "1.34", + "nodePools": [ + { + "name": "h100pool", + "role": "GPU", + "vmSize": "Standard_ND96isr_H100_v5", + "diskSizeGb": 200, + "nodeCount": 2, + "minNodeCount": 1, + "maxNodeCount": 4, + "gpu": {"acceleratorType": "nvidia-h100"}, + "fabric": "InfiniBand", + }, + ], + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="test-cluster-kubeconfig-abcde", identity=None + ), + "serving-stack": _serving_stack( + cloud="AKS", + secrets=[{"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}], + client_cas=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-aks-by-backend": _backend_usage(cluster_kind="AKSCluster"), + }, + ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="AKS cluster ready, composing backend")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-h100-aks": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-h100-aks" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], + ), + ), + # While the policy is missing even one of the kinds from status.activated + # (e.g. a provider still installing), and with no cluster observed, the + # function composes only the activation policy, not the cluster XR. It + # doesn't mark the policy ready, so the composite doesn't report ready. + # Observing the policy with a kind missing, here + # nodepools.container.gcp.m.upbound.io, exercises the all-kinds check rather + # than the policy-absent branch. + ComposeCase( + name="cloud cluster not activated composes only the policy", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="GKE", gke=v1alpha1.Gke(region="us-central1")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], + ), + ], + ), + resources={ + "activation": _observed_activation_policy( + activated=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + ] + ), + }, + ), + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", + ], + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], + ), + ), + # Once the cluster is observed, the function keeps composing it even when + # the policy momentarily stops reporting the kinds active, so an activation + # blip never drops a provisioned cluster from desired state. Here the + # policy isn't observed at all. + ComposeCase( + name="observed cluster keeps composing through an activation blip", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="GKE", gke=v1alpha1.Gke(region="us-central1")), + node_pools=[ + v1alpha1.NodePool( + name="l4-pool", + className="gpu-l4", + nodeCount=2, + maxNodeCount=4, + zones=["us-central1-a"], + ), + ], + ), + resources={ + "gke-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "GKECluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "us-central1", + "nodePools": [{"name": "system", "role": "System", "machineType": "e2-standard-4"}], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + }, + ], + "cache": {"storageClassName": "modelplane-rwx"}, + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + { + "type": "GoogleApplicationCredentials", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", + }, + ], + }, + } + ), + ), + }, + ), + required_resources={ + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + cache={"storageClassName": "modelplane-rwx"}, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=[ + "projectiammembers.cloudplatform.gcp.m.upbound.io", + "projectservices.cloudplatform.gcp.m.upbound.io", + "serviceaccounts.cloudplatform.gcp.m.upbound.io", + "serviceaccountkeys.cloudplatform.gcp.m.upbound.io", + "networks.compute.gcp.m.upbound.io", + "subnetworks.compute.gcp.m.upbound.io", + "clusters.container.gcp.m.upbound.io", + "nodepools.container.gcp.m.upbound.io", + ], + ready=fnv1.READY_TRUE, + ), + "gke-cluster": _gke_cluster(credentials=None, ready=fnv1.READY_TRUE), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="test-cluster-kubeconfig-abcde", + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", }, - ], - "secrets": [ + }, + ), + "serving-stack": _serving_stack( + cloud="GKE", + secrets=[ + {"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}, { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + "type": "GoogleApplicationCredentials", + "name": "test-cluster-sa-key-fghij", + "key": "credentials.json", }, ], - # The backing AKSCluster reports its effective RWX - # StorageClass; the InferenceCluster relays it up. - "cache": {"storageClassName": "modelplane-rwx-fs"}, - }, - } + client_cas=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-gke-by-backend": _backend_usage(cluster_kind="GKECluster"), + }, + ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="GKE cluster ready, composing backend")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], ), - ) - - want13 = fnv1.RunFunctionResponse() - want13.CopyFrom(want12) - want13.desired.resources["aks-cluster"].ready = fnv1.READY_TRUE - status13 = want13.desired.composite.resource.fields["status"].struct_value - status13.fields["cache"].struct_value.fields["storageClassName"].string_value = "modelplane-rwx-fs" - want13.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", + ), + # The first pass composes only the activation policy and the VultrCluster + # XR. minNodeCount stays unset so the pool's autoscaling floor defaults to + # its node count downstream. + ComposeCase( + name="Vultr cluster first pass composes only the policy and the VultrCluster XR", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="Vultr", vultr=v1alpha1.Vultr(region="ewr")), + node_pools=[ + v1alpha1.NodePool(name="l40s-pool", className="gpu-l40s-vultr", nodeCount=2, maxNodeCount=4), + ], + ), + resources={ + "activation": _observed_activation_policy( + activated=["kubernetes.vke.vultr.m.upbound.io", "kubernetesnodepools.vke.vultr.m.upbound.io"] + ), + }, + ), + required_resources={ + "class-gpu-l40s-vultr": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l40s-vultr", + count=1, + memory="46068Mi", + provisioning={ + "provider": "Vultr", + "vultr": { + "plan": "vcg-l40s-16c-180g-48vram", + "accelerator": {"type": "nvidia-l40s", "count": 1}, + }, }, + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l40s-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, + }, + ], }, - }, - } + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=["kubernetes.vke.vultr.m.upbound.io", "kubernetesnodepools.vke.vultr.m.upbound.io"], + ready=fnv1.READY_TRUE, + ), + "vultr-cluster": _vultr_cluster(credentials=None, ready=fnv1.READY_UNSPECIFIED), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l40s-vultr": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l40s-vultr" + ), + }, ), - ready=fnv1.READY_TRUE, + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], ), - ) - want13.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "AKS", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - ], - }, - } + ), + # Vultr credentials pass through to the VultrCluster spec, mirroring the GKE + # passthrough. + ComposeCase( + name="Vultr credentials pass through to VultrCluster spec", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Vultr", + vultr=v1alpha1.Vultr( + region="ewr", + credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-vultr-account"), + ), + ), + node_pools=[ + v1alpha1.NodePool(name="l40s-pool", className="gpu-l40s-vultr", nodeCount=2, maxNodeCount=4), + ], + ), + resources={ + "activation": _observed_activation_policy( + activated=["kubernetes.vke.vultr.m.upbound.io", "kubernetesnodepools.vke.vultr.m.upbound.io"] + ), + }, ), + required_resources={ + "class-gpu-l40s-vultr": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l40s-vultr", + count=1, + memory="46068Mi", + provisioning={ + "provider": "Vultr", + "vultr": { + "plan": "vcg-l40s-16c-180g-48vram", + "accelerator": {"type": "nvidia-l40s", "count": 1}, + }, + }, + ), + ], + ), + }, ), - ) - want13.desired.resources["usage-aks-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "AKSCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l40s-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, + }, + ], }, - "replayDeletion": True, - }, - } + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=["kubernetes.vke.vultr.m.upbound.io", "kubernetesnodepools.vke.vultr.m.upbound.io"], + ready=fnv1.READY_TRUE, + ), + "vultr-cluster": _vultr_cluster( + credentials={"type": "ProviderConfig", "name": "my-vultr-account"}, ready=fnv1.READY_UNSPECIFIED + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l40s-vultr": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l40s-vultr" + ), + }, ), - ready=fnv1.READY_TRUE, + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Provisioning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="WaitingForCluster"), + ], ), - ) - del want13.conditions[:] - want13.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want13.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="AKS cluster ready, composing backend", - ) - ) - - # --- Case 14: Vultr first pass composes the VultrCluster XR only. - # minNodeCount stays unset so the pool's autoscaling floor defaults - # to its node count downstream. --- - inference_class_l40s_vultr = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceClass", - "metadata": {"name": "gpu-l40s-vultr"}, - "spec": { - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, + ), + # The kubeconfig is observed on the VultrCluster status. The VKE kubeconfig + # embeds static client certificates, so the ClusterProviderConfig carries no + # identity (unlike Nebius). The function composes the ServingStack backend + # with the kubeconfig and emits the Usage that blocks VultrCluster deletion + # until the ServingStack is gone. VultrCluster reports no cache + # StorageClass, so status.cache stays unset. + ComposeCase( + name="Vultr cluster ready composes CPC without identity, ServingStack, and Usage", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster(source="Vultr", vultr=v1alpha1.Vultr(region="ewr")), + node_pools=[ + v1alpha1.NodePool(name="l40s-pool", className="gpu-l40s-vultr", nodeCount=2, maxNodeCount=4), + ], + ), + resources={ + "vultr-cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "VultrCluster", + "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, + "spec": { + "region": "ewr", + "nodePools": [ + { + "name": "l40s-pool", + "role": "GPU", + "plan": "vcg-l40s-16c-180g-48vram", + "nodeCount": 2, + }, + ], + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-abcde", + "key": "kubeconfig", + }, + ], + }, + } + ), + ), + "activation": _observed_activation_policy( + activated=["kubernetes.vke.vultr.m.upbound.io", "kubernetesnodepools.vke.vultr.m.upbound.io"] + ), }, + ), + required_resources={ + "class-gpu-l40s-vultr": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l40s-vultr", + count=1, + memory="46068Mi", + provisioning={ + "provider": "Vultr", + "vultr": { + "plan": "vcg-l40s-16c-180g-48vram", + "accelerator": {"type": "nvidia-l40s", "count": 1}, + }, + }, + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l40s-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "46068Mi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "activation": _activation_policy( + activate=["kubernetes.vke.vultr.m.upbound.io", "kubernetesnodepools.vke.vultr.m.upbound.io"], + ready=fnv1.READY_TRUE, + ), + "vultr-cluster": _vultr_cluster(credentials=None, ready=fnv1.READY_TRUE), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="test-cluster-kubeconfig-abcde", identity=None + ), + "serving-stack": _serving_stack( + cloud="Vultr", + secrets=[{"type": "Kubeconfig", "name": "test-cluster-kubeconfig-abcde", "key": "kubeconfig"}], + client_cas=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-vultr-by-backend": _backend_usage(cluster_kind="VultrCluster"), + }, + ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Vultr cluster ready, composing backend")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l40s-vultr": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l40s-vultr" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), ], - "provisioning": { - "provider": "Vultr", - "vultr": { - "plan": "vcg-l40s-16c-180g-48vram", - "accelerator": {"type": "nvidia-l40s", "count": 1}, + ), + ), + # The deletion guard cases up to the early return observe an existing + # cluster along with whatever uses it. + # + # ModelReplicas, ModelRoutes and ModelCaches across several namespaces + # compose a single reason-only ClusterUsage blocking the InferenceCluster's + # deletion, whatever their count or namespace, and mirror the deduplicated + # union of their namespaces: team-a (replica), team-b (replica and route), + # team-c (route), team-d (cache staging onto this cluster). A cache staging + # only onto another cluster (team-e) is filtered out by its + # status.clusters[], proving the namespaces track what actually lands here. + # The guard's reason names every kind in use. + ComposeCase( + name="ModelReplicas, ModelRoutes and ModelCaches compose the guard and the mirrored namespaces", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], + ), + ), + required_resources={ + "model-replicas": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "deploy-test-cluster-0", + "namespace": "team-a", + "labels": {"modelplane.ai/cluster": "test-cluster"}, + }, + } + ) + ), + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "deploy-test-cluster-0", + "namespace": "team-b", + "labels": {"modelplane.ai/cluster": "test-cluster"}, + }, + } + ) + ), + ], + ), + "model-routes": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelRoute", + "metadata": { + "name": "svc-eu", + "namespace": "team-b", + "labels": {"modelplane.ai/cluster": "test-cluster"}, + }, + } + ) + ), + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelRoute", + "metadata": { + "name": "svc-eu", + "namespace": "team-c", + "labels": {"modelplane.ai/cluster": "test-cluster"}, + }, + } + ) + ), + ], + ), + "model-caches": fnv1.Resources( + items=[ + _model_cache(name="qwen", namespace="team-d", cluster="test-cluster"), + _model_cache(name="kimi", namespace="team-e", cluster="other-cluster"), + ], + ), + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "usage-replicas": _guard_clusterusage( + reason="ModelReplicas, ModelRoutes and ModelCaches use this InferenceCluster" + ), + "namespace-team-a": _namespace_object(team="team-a", name="mp-team-a-bd964"), + "namespace-team-b": _namespace_object(team="team-b", name="mp-team-b-6bd62"), + "namespace-team-c": _namespace_object(team="team-c", name="mp-team-c-d79d9"), + "namespace-team-d": _namespace_object(team="team-d", name="mp-team-d-c2383"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], + ), + ), + # A ModelRoute composes its routing Objects through the cluster's + # ClusterProviderConfig, so it blocks deletion without any replica there, + # and its team's namespace is mirrored. + ComposeCase( + name="a ModelRoute on the cluster composes the guard", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], + ), + ), + required_resources={ + "model-routes": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelRoute", + "metadata": { + "name": "svc-eu", + "namespace": "team-c", + "labels": {"modelplane.ai/cluster": "test-cluster"}, + }, + } + ) + ) + ] + ), + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, + ), + ], + ), }, - }, - } - class_selector_vultr = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceClass", - match_name="gpu-l40s-vultr", - ) - - req14 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "usage-replicas": _guard_clusterusage(reason="ModelRoutes use this InferenceCluster"), + "namespace-team-c": _namespace_object(team="team-c", name="mp-team-c-d79d9"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], + ), + ), + # A ModelCache composes its PVC through the cluster's ClusterProviderConfig, + # so it blocks deletion without any replica there, and its team's namespace + # is mirrored. + ComposeCase( + name="a ModelCache staging onto the cluster composes the guard", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], + ), + ), + required_resources={ + "model-caches": fnv1.Resources( + items=[_model_cache(name="qwen", namespace="team-d", cluster="test-cluster")] + ), + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Vultr", - vultr=v1alpha1.Vultr(region="ewr"), - ), - nodePools=[ - v1alpha1.NodePool( - name="l40s-pool", - className="gpu-l40s-vultr", - nodeCount=2, - maxNodeCount=4, - ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, ], + }, + ], + cache=None, + gateway=None, + ), + resources={ + "usage-replicas": _guard_clusterusage(reason="ModelCaches use this InferenceCluster"), + "namespace-team-d": _namespace_object(team="team-d", name="mp-team-d-c2383"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], + ), + ), + # An InferenceGateway composes its Gateway and routing Objects through the + # cluster's ClusterProviderConfig just as a replica does, so it blocks + # deletion too. It is cluster scoped, so it mirrors no namespace. + ComposeCase( + name="an InferenceGateway on the cluster composes the guard", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], + ), + ), + required_resources={ + "gateways": fnv1.Resources( + items=[_inference_gateway(name="public", cluster="test-cluster", status=None)] + ), + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, + }, ), - ).model_dump(exclude_none=True, mode="json"), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, + }, + ], + }, + ], + cache=None, + gateway=None, ), + resources={ + "usage-replicas": _guard_clusterusage(reason="InferenceGateways use this InferenceCluster"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], ), - ) - req14.required_resources["class-gpu-l40s-vultr"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l40s_vultr)), - ) - - want14 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", + ), + # Every guard requirement resolves to nothing on this cluster, so there's + # no ClusterUsage. This is the teardown transition - the last user is gone, + # so the function stops composing the guard and the cluster becomes + # deletable. A cache staging onto another cluster and a gateway running on + # another cluster don't hold it. The replica and route requirements are + # empty but present, as Crossplane returns them when a selector matches + # nothing. + ComposeCase( + name="nothing on the cluster leaves it deletable", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], + ), + ), + required_resources={ + "model-replicas": fnv1.Resources(), + "model-routes": fnv1.Resources(), + "model-caches": fnv1.Resources( + items=[_model_cache(name="kimi", namespace="team-e", cluster="other-cluster")] + ), + "gateways": fnv1.Resources( + items=[_inference_gateway(name="elsewhere", cluster="other-cluster", status=None)] + ), + "class-gpu-l4": fnv1.Resources( + items=[ + _inference_class( + name="gpu-l4", + count=1, + memory="24Gi", + provisioning={ + "provider": "GKE", + "gke": { + "machineType": "g2-standard-48", + "diskSizeGb": 100, + "accelerator": {"type": "nvidia-l4", "count": 1}, + }, }, - "namespace": "modelplane-system", - "gpuPools": [ + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[ + { + "name": "l4-pool", + "nodes": 4, + "devices": [ { - "name": "l40s-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, - }, - ], + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "capacity": {"memory": {"value": "24Gi"}}, }, ], }, - }, + ], + cache=None, + gateway=None, ), - ), - resources={ - "vultr-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "ewr", - "kubernetesVersion": "v1.36.2+1", - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "plan": "vcg-l40s-16c-180g-48vram", - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-l40s", - }, - }, - ], - }, - }, + resources={ + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, ), - ], - context=structpb.Struct(), - ) - want14.requirements.resources["class-gpu-l40s-vultr"].CopyFrom(class_selector_vultr) - - # --- Case 14b: Vultr credentials pass through to the VultrCluster - # spec, mirroring the GKE/EKS/AKS passthrough. --- - req_creds_vultr = fnv1.RunFunctionRequest() - req_creds_vultr.CopyFrom(req14) - req_creds_vultr.observed.composite.CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Vultr", - vultr=v1alpha1.Vultr( - region="ewr", - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="my-vultr-account", - ), - ), - ), - nodePools=[ - v1alpha1.NodePool( - name="l40s-pool", - className="gpu-l40s-vultr", - nodeCount=2, - maxNodeCount=4, - ), - ], + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Installing"), + ], + ), + ), + # The guard is composed even when compose() returns early. resolve_classes() + # returns False whenever a referenced InferenceClass isn't observed yet - a + # routine transient. Here the class requirement is declared but not + # fulfilled. The function returns before composing the cluster, but the + # guard runs first, so a referencing replica still blocks deletion. This is + # the case that regresses if the guard is gated behind class resolution or + # cluster source. + # + # Only the guard and namespace are composed, both ready, so the function + # marks the XR not ready itself. Otherwise it would read ready while it + # waits for its classes. + ComposeCase( + name="guard is composed even when compose returns early", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), ), - ).model_dump(exclude_none=True, mode="json"), + node_pools=[ + v1alpha1.NodePool(name="l4-pool", className="gpu-l4", nodeCount=2, maxNodeCount=4), + ], + ), ), + required_resources={ + "model-replicas": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "deploy-test-cluster-0", + "namespace": "team-a", + "labels": {"modelplane.ai/cluster": "test-cluster"}, + }, + } + ) + ) + ] + ), + }, ), - ) - - want_creds_vultr = fnv1.RunFunctionResponse() - want_creds_vultr.CopyFrom(want14) - want_creds_vultr.desired.resources["vultr-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "ewr", - "kubernetesVersion": "v1.36.2+1", - "credentials": { - "type": "ProviderConfig", - "name": "my-vultr-account", - }, - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "plan": "vcg-l40s-16c-180g-48vram", - "nodeCount": 2, - "maxNodeCount": 4, - "gpu": { - "acceleratorType": "nvidia-l40s", - }, - }, - ], - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=fnv1.Resource(ready=fnv1.READY_FALSE), + resources={ + "usage-replicas": _guard_clusterusage(reason="ModelReplicas use this InferenceCluster"), + "namespace-team-a": _namespace_object(team="team-a", name="mp-team-a-bd964"), }, ), - ), - ) - - # --- Case 15: Vultr cluster ready - kubeconfig observed on the - # VultrCluster status. The VKE kubeconfig embeds static client - # certificates, so the ClusterProviderConfig carries no identity - # (unlike Nebius). The function composes the ServingStack backend - # with the kubeconfig and emits the Usage that blocks VultrCluster - # deletion until the ServingStack is gone. VultrCluster reports no - # cache StorageClass, so status.cache stays unset. --- - req15 = fnv1.RunFunctionRequest() - req15.CopyFrom(req14) - req15.observed.resources["vultr-cluster"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "metadata": {"name": "test-cluster", "namespace": "modelplane-system"}, - "spec": { - "region": "ewr", - "nodePools": [ - { - "name": "l40s-pool", - "role": "GPU", - "plan": "vcg-l40s-16c-180g-48vram", - "nodeCount": 2, - }, - ], - }, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - ], - }, - } + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for InferenceClasses: gpu-l4")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "class-gpu-l4": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceClass", match_name="gpu-l4" + ), + }, ), + conditions=[ + fnv1.Condition( + type="ClusterReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForClasses", + message="Waiting for InferenceClasses: gpu-l4", + ), + ], ), - ) - - want15 = fnv1.RunFunctionResponse() - want15.CopyFrom(want14) - want15.desired.resources["vultr-cluster"].ready = fnv1.READY_TRUE - want15.desired.composite.CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ - { - "name": "l40s-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "46068Mi"}}, - }, - ], - }, - ], - }, + ), + # The hostname gate, which is what keeps a cluster off the schedule until + # traffic to it is mutually authenticated in both directions. These cases + # observe an existing cluster's ServingStack ready, with whatever it and the + # fleet's InferenceGateways have published so far. The gateway name is + # Modelplane's own, so nothing configures it. An InferenceGateway running on + # test-cluster also composes the deletion guard. + # + # An address to reach, this cluster's CA so an InferenceGateway can tell it + # reached the right cluster, and an InferenceGateway CA so the cluster + # gateway demands a client certificate. + ComposeCase( + name="hostname is published once both directions are authenticated", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=None, + ), + resources={ + "serving-stack": _observed_serving_stack( + gateway={"address": "34.55.100.10", "caCertificate": "cluster-ca"} + ), }, ), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="fleet-0", cluster="test-cluster", status={"clientCACertificate": "fleet-ca"} + ) + ], + ), + }, ), - ) - want15.desired.resources["cluster-provider-config-kubernetes"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "test-cluster-cluster-kubeconfig-d0f89"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[], + cache=None, + gateway={ + "address": "34.55.100.10", + "caCertificate": "cluster-ca", + "hostname": "gateway-test-cluster-09532.modelplane-system.svc.cluster.local", }, - } + ), + resources={ + "usage-replicas": _guard_clusterusage(reason="InferenceGateways use this InferenceCluster"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=[{"name": "fleet-0", "certificate": "fleet-ca"}], + ready=fnv1.READY_TRUE, + ), + }, ), - ready=fnv1.READY_TRUE, - ), - ) - want15.desired.resources["serving-stack"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": { - "name": "test-cluster-serving-stack-fd00b", - "namespace": "modelplane-system", - }, - "spec": { - "cloud": "Vultr", - "gateway": {"hostname": _GATEWAY_HOSTNAME}, - "stack": "Standard", - "secrets": [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-abcde", - "key": "kubeconfig", - }, - ], - }, - } + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + }, ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_TRUE, reason="BackendHealthy"), + ], ), - ) - want15.desired.resources["usage-vultr-by-backend"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "of": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "VultrCluster", - "resourceSelector": {"matchControllerRef": True}, - }, - "by": { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "resourceSelector": {"matchControllerRef": True}, - }, - "replayDeletion": True, - }, - } + ), + # The case that matters: the cluster gateway only demands a client + # certificate when it has a CA to check against, and with none it serves no + # Gateway at all. Publishing the hostname anyway would make the cluster + # schedulable when nothing is listening on it, so every request routed there + # would be stranded. + ComposeCase( + name="no hostname without an InferenceGateway CA", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=None, + ), + resources={ + "serving-stack": _observed_serving_stack( + gateway={"address": "34.55.100.10", "caCertificate": "cluster-ca"} + ), + }, ), - ready=fnv1.READY_TRUE, ), - ) - del want15.conditions[:] - want15.conditions.extend( - [ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ClusterRunning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Installing", - ), - ] - ) - want15.results.append( - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Vultr cluster ready, composing backend", - ) - ) - - # Every compose path emits the ModelReplica guard requirement. - for want in ( - want1, - want2, - want3, - want4, - want5, - want6, - want7, - want8, - want9, - want10, - want11, - want12, - want13, - want14, - want_creds_vultr, - want15, - ): - want.requirements.resources["gateways"].CopyFrom(_gateways_selector()) - want.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) - want.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) - want.requirements.resources["model-caches"].CopyFrom(_caches_selector()) - - # The guard cases reuse case 1's request and response. - guard_cases = [ - Case( - "ModelReplicas, ModelRoutes and ModelCaches compose the guard and the mirrored namespaces", - *_guard_case(req1, want1), - ), - Case("a ModelRoute on the cluster composes the guard", *_route_guard_case(req1, want1)), - Case("a ModelCache staging onto the cluster composes the guard", *_cache_guard_case(req1, want1)), - Case("an InferenceGateway on the cluster composes the guard", *_gateway_guard_case(req1, want1)), - Case("nothing on the cluster leaves it deletable", *_unused_case(req1, want1)), - Case("guard is composed even when compose returns early", *_early_return_guard_case()), - ] - - # --- Case credentials: GKE with custom credentials passes them through to GKECluster. --- - req_creds = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="GKE", - gke=v1alpha1.Gke( - region="us-central1", - credentials=v1alpha1.Credentials( - type="ProviderConfig", - name="my-gcp-account", - ), - ), - ), - nodePools=[ - v1alpha1.NodePool( - name="l4-pool", - className="gpu-l4", - nodeCount=2, - maxNodeCount=4, - zones=["us-central1-a"], - ), - ], - ), - ).model_dump(exclude_none=True, mode="json") + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[], cache=None, gateway={"address": "34.55.100.10", "caCertificate": "cluster-ca"} ), + resources={ + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=None, + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + }, ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_TRUE, reason="BackendHealthy"), + ], ), - ) - req_creds.required_resources["class-gpu-l4"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(inference_class_l4)) - ) - - want_creds = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "providerConfigRef": { - "name": "test-cluster-cluster-kubeconfig-d0f89", - }, - "namespace": "modelplane-system", - "gpuPools": [ - { - "name": "l4-pool", - "nodes": 4, - "devices": [ - { - "name": "gpu", - "claim": "DRA", - "driver": "gpu.nvidia.com", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "capacity": {"memory": {"value": "24Gi"}}, - }, - ], - }, - ], - }, - } + ), + # Without this cluster's CA an InferenceGateway can't validate the cluster + # gateway it reaches, so it would have to fall back to the public trust + # store. + ComposeCase( + name="no hostname without this cluster's CA", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=None, ), + resources={ + "serving-stack": _observed_serving_stack(gateway={"address": "34.55.100.10"}), + }, ), - resources={ - "gke-cluster": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "GKECluster", - "metadata": { - "name": "test-cluster", - "namespace": "modelplane-system", - }, - "spec": { - "region": "us-central1", - "kubernetesVersion": "1.35", - "credentials": { - "type": "ProviderConfig", - "name": "my-gcp-account", - }, - "nodePools": [ - { - "name": "l4-pool", - "role": "GPU", - "machineType": "g2-standard-48", - "nodeCount": 2, - "minNodeCount": None, - "maxNodeCount": 4, - "diskSizeGb": 100, - "gpu": { - "acceleratorType": "nvidia-l4", - "acceleratorCount": 1, - }, - "zones": ["us-central1-a"], - }, - ], - }, - } - ), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="fleet-0", cluster="test-cluster", status={"clientCACertificate": "fleet-ca"} + ) + ], ), }, ), - conditions=[ - fnv1.Condition( - type="ClusterReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Provisioning", - ), - fnv1.Condition( - type="BackendReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster(gpu_pools=[], cache=None, gateway={"address": "34.55.100.10"}), + resources={ + "usage-replicas": _guard_clusterusage(reason="InferenceGateways use this InferenceCluster"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=[{"name": "fleet-0", "certificate": "fleet-ca"}], + ready=fnv1.READY_TRUE, + ), + }, ), - ], - context=structpb.Struct(), - ) - want_creds.requirements.resources["class-gpu-l4"].CopyFrom(class_selector) - want_creds.requirements.resources["gateways"].CopyFrom(_gateways_selector()) - want_creds.requirements.resources["model-replicas"].CopyFrom(_replicas_selector("test-cluster")) - want_creds.requirements.resources["model-routes"].CopyFrom(_routes_selector("test-cluster")) - want_creds.requirements.resources["model-caches"].CopyFrom(_caches_selector()) - - # Every cloud cluster composes an activation policy; with the policy - # observed Healthy the cluster XR is composed. - for req, want, kinds in [ - (req2, want2, fn._ACTIVATE_GCP), - (req_creds, want_creds, fn._ACTIVATE_GCP), - (req4, want4, fn._ACTIVATE_AWS), - (req5, want5, fn._ACTIVATE_AWS), - (req6, want6, fn._ACTIVATE_GCP), - (req7, want7, fn._ACTIVATE_AWS), - (req8, want8, fn._ACTIVATE_AWS), - (req9, want9, fn._ACTIVATE_AWS), - (req10, want10, fn._ACTIVATE_NEBIUS), - (req11, want11, fn._ACTIVATE_NEBIUS), - (req12, want12, fn._ACTIVATE_AZURE), - (req13, want13, fn._ACTIVATE_AZURE), - (req14, want14, fn._ACTIVATE_VULTR), - (req_creds_vultr, want_creds_vultr, fn._ACTIVATE_VULTR), - (req15, want15, fn._ACTIVATE_VULTR), - ]: - _observe_activated(req, kinds) - _want_activation(want, kinds) - - # While the policy is missing even one of the kinds from status.activated - # (e.g. a provider still installing), and with no cluster observed, the - # function composes only the activation policy (not marked ready, so the - # composite doesn't report ready), not the cluster XR. Observing the - # policy with a kind missing exercises the all-kinds check rather than - # the policy-absent branch. - req_unactivated = copy.deepcopy(req2) - _observe_activated(req_unactivated, fn._ACTIVATE_GCP[:-1]) - want_unactivated = copy.deepcopy(want2) - del want_unactivated.desired.resources["gke-cluster"] - want_unactivated.desired.resources["activation"].ClearField("ready") - - # Once the cluster is observed, the function keeps composing it even - # when the policy momentarily stops reporting the kinds active, so an - # activation blip never drops a provisioned cluster from desired state. - req_blip = copy.deepcopy(req6) - del req_blip.observed.resources["activation"] - want_blip = copy.deepcopy(want6) - - return [ - Case(name="existing cluster with secrets composes backend and CPC", req=req1, want=want1), - Case(name="existing cluster with a non-GCP identity threads the identity type", req=req1b, want=want1b), - Case(name="GKE cluster first pass composes GKECluster XR only", req=req2, want=want2), - Case(name="GKE credentials pass through to GKECluster spec", req=req_creds, want=want_creds), - Case(name="existing cluster second pass with backend ready", req=req3, want=want3), - Case(name="EKS cluster first pass composes EKSCluster XR only", req=req4, want=want4), - Case(name="EKS cluster not ready re-emits existing CPC unchanged", req=req5, want=want5), - Case(name="GKE cluster ready composes CPC, backend, usage, and RWX StorageClass", req=req6, want=want6), - Case(name="EKS cluster ready composes ServingStack and Usage", req=req7, want=want7), - Case( - name="EKS node pool with a Capacity Block sets capacityBlock on the EKSCluster pool", - req=req8, - want=want8, - ), - Case( - name="EKS node pool with fabric EFA sets fabric on the EKSCluster pool", - req=req9, - want=want9, - ), - Case(name="Nebius cluster first pass composes NebiusCluster XR only", req=req10, want=want10), - Case( - name="Nebius cluster ready composes CPC with Nebius identity, ServingStack, and Usage", - req=req11, - want=want11, - ), - Case(name="AKS cluster first pass composes AKSCluster XR only", req=req12, want=want12), - Case( - name="AKS cluster ready composes CPC without identity, ServingStack, and Usage", - req=req13, - want=want13, - ), - Case(name="cloud cluster not activated composes only the policy", req=req_unactivated, want=want_unactivated), - Case(name="observed cluster keeps composing through an activation blip", req=req_blip, want=want_blip), - Case(name="Vultr cluster first pass composes VultrCluster XR only", req=req14, want=want14), - Case( - name="Vultr credentials pass through to VultrCluster spec", - req=req_creds_vultr, - want=want_creds_vultr, - ), - Case( - name="Vultr cluster ready composes CPC without identity, ServingStack, and Usage", - req=req15, - want=want15, - ), - *guard_cases, - ] - - -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) -def test_compose(case: Case) -> None: - """RunFunction composes the resources an InferenceCluster needs.""" - got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) - - -# The hostname gate, which is what keeps a cluster off the schedule until -# traffic to it is mutually authenticated in both directions. - - -def _gateway_status_request(*, address: str | None, ca: str | None, gateway_cas: list[str]) -> fnv1.RunFunctionRequest: - """A cluster and whatever its serving stack and the fleet's gateways have - published so far. The gateway name is Modelplane's own, so nothing - configures it.""" - xr = v1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), - spec=v1alpha1.Spec( - cluster=v1alpha1.Cluster( - source="Existing", - existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + }, ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_TRUE, reason="BackendHealthy"), + ], ), - ) - stack_status: dict = {"conditions": [{"type": "Ready", "status": "True"}]} - gateway: dict = {} - if address: - gateway["address"] = address - if ca: - gateway["caCertificate"] = ca - if gateway: - stack_status["gateway"] = gateway - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json"))), - resources={ - "serving-stack": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": {"name": "test-cluster-serving-stack-fd00b"}, - "status": stack_status, - } + ), + # A hostname that resolves to nothing strands every request routed to it, + # and the CA is republished from the same status. + ComposeCase( + name="no gateway status before an address", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), ), + node_pools=None, + ), + resources={ + "serving-stack": _observed_serving_stack(gateway={"caCertificate": "cluster-ca"}), + }, + ), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="fleet-0", cluster="test-cluster", status={"clientCACertificate": "fleet-ca"} + ) + ], ), }, ), - ) - for i, cert in enumerate(gateway_cas): - req.required_resources["gateways"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceGateway", - "metadata": {"name": f"fleet-{i}"}, - "spec": {"clusterName": "test-cluster"}, - "status": {"clientCACertificate": cert}, - } - ), - ) - ) - return req - - -def _gateway_status(req: fnv1.RunFunctionRequest) -> dict: - """The status.gateway RunFunction writes to the InferenceCluster for req.""" - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - return resource.struct_to_dict(got.desired.composite.resource).get("status", {}).get("gateway", {}) - - -def test_gateway_status_hostname_published_once_both_directions_are_authenticated() -> None: - """The hostname is published once the address and both directions' CAs are.""" - # An address to reach, this cluster's CA so an InferenceGateway can tell it - # reached the right cluster, and an InferenceGateway CA so the cluster - # gateway demands a client certificate. - status = _gateway_status(_gateway_status_request(address="34.55.100.10", ca="cluster-ca", gateway_cas=["fleet-ca"])) - assert status == { - "address": "34.55.100.10", - "caCertificate": "cluster-ca", - "hostname": _GATEWAY_HOSTNAME, - } - - -def test_gateway_status_no_hostname_without_an_inference_gateway_ca() -> None: - """No hostname is published until an InferenceGateway has published a CA.""" - # The case that matters: the cluster gateway only demands a client - # certificate when it has a CA to check against, and with none it serves no - # Gateway at all. Publishing the hostname anyway would make the cluster - # schedulable when nothing is listening on it, so every request routed - # there would be stranded. - status = _gateway_status(_gateway_status_request(address="34.55.100.10", ca="cluster-ca", gateway_cas=[])) - assert status == {"address": "34.55.100.10", "caCertificate": "cluster-ca"} - - -def test_gateway_status_no_hostname_without_this_clusters_ca() -> None: - """No hostname is published until the cluster's own gateway CA is.""" - # Without it an InferenceGateway can't validate the cluster gateway it - # reaches, so it would have to fall back to the public trust store. - status = _gateway_status(_gateway_status_request(address="34.55.100.10", ca=None, gateway_cas=["fleet-ca"])) - assert status == {"address": "34.55.100.10"} - - -def test_gateway_status_no_gateway_status_before_an_address() -> None: - """No gateway status at all is published before the gateway has an address.""" - # A hostname that resolves to nothing strands every request routed to it, - # and the CA is republished from the same status. - status = _gateway_status(_gateway_status_request(address=None, ca="cluster-ca", gateway_cas=["fleet-ca"])) - assert status == {} - - -def test_gateway_status_serving_stack_accepts_every_inference_gateway_ca() -> None: - """The ServingStack's gateway accepts every published InferenceGateway CA, sorted.""" + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster(gpu_pools=[], cache=None, gateway=None), + resources={ + "usage-replicas": _guard_clusterusage(reason="InferenceGateways use this InferenceCluster"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=[{"name": "fleet-0", "certificate": "fleet-ca"}], + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_TRUE, reason="BackendHealthy"), + ], + ), + ), # Any InferenceGateway may forward to this cluster, so its gateway accepts # every published CA, whichever cluster the InferenceGateway runs on. These # CAs are what switches the cluster gateway's mTLS listener on. One that # hasn't published a CA yet is left out rather than holding the others # back, and the list is sorted so it doesn't churn. - req = _gateway_status_request(address="34.55.100.10", ca="cluster-ca", gateway_cas=["fleet-0-ca", "fleet-1-ca"]) - for name, status in (("aaa", {"clientCACertificate": "aaa-ca"}), ("pending", {})): - req.required_resources["gateways"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceGateway", - "metadata": {"name": name}, - "spec": {"clusterName": "elsewhere"}, - "status": status, - } + ComposeCase( + name="ServingStack accepts every InferenceGateway CA", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_inference_cluster( + cluster=v1alpha1.Cluster( + source="Existing", + existing=v1alpha1.Existing(secretRef=v1alpha1.SecretRef(name="my-kubeconfig")), + ), + node_pools=None, ), - ) - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - stack = resource.struct_to_dict(got.desired.resources[fn.BACKEND_RESOURCE_KEY].resource) - assert stack["spec"]["gateway"] == { - "hostname": _GATEWAY_HOSTNAME, - "clientCAs": [ - {"name": "aaa", "certificate": "aaa-ca"}, - {"name": "fleet-0", "certificate": "fleet-0-ca"}, - {"name": "fleet-1", "certificate": "fleet-1-ca"}, - ], - } + resources={ + "serving-stack": _observed_serving_stack( + gateway={"address": "34.55.100.10", "caCertificate": "cluster-ca"} + ), + }, + ), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="fleet-0", cluster="test-cluster", status={"clientCACertificate": "fleet-0-ca"} + ), + _inference_gateway( + name="fleet-1", cluster="test-cluster", status={"clientCACertificate": "fleet-1-ca"} + ), + _inference_gateway(name="aaa", cluster="elsewhere", status={"clientCACertificate": "aaa-ca"}), + _inference_gateway(name="pending", cluster="elsewhere", status={}), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_inference_cluster( + gpu_pools=[], + cache=None, + gateway={ + "address": "34.55.100.10", + "caCertificate": "cluster-ca", + "hostname": "gateway-test-cluster-09532.modelplane-system.svc.cluster.local", + }, + ), + resources={ + "usage-replicas": _guard_clusterusage(reason="InferenceGateways use this InferenceCluster"), + "cluster-provider-config-kubernetes": _cluster_provider_config( + kubeconfig="my-kubeconfig", identity=None + ), + "serving-stack": _serving_stack( + cloud="Existing", + secrets=[{"type": "Kubeconfig", "name": "my-kubeconfig", "key": "kubeconfig"}], + client_cas=[ + {"name": "aaa", "certificate": "aaa-ca"}, + {"name": "fleet-0", "certificate": "fleet-0-ca"}, + {"name": "fleet-1", "certificate": "fleet-1-ca"}, + ], + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "model-replicas": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelReplica", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-routes": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelRoute", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/cluster": "test-cluster"}), + ), + "model-caches": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelCache"), + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + }, + ), + conditions=[ + fnv1.Condition(type="ClusterReady", status=fnv1.STATUS_CONDITION_TRUE, reason="ClusterRunning"), + fnv1.Condition(type="BackendReady", status=fnv1.STATUS_CONDITION_TRUE, reason="BackendHealthy"), + ], + ), + ), +] -# The derived gateway hostname doubles as an SNI and a certificate SAN, so two -# clusters must never derive the same one. +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: ComposeCase) -> None: + """RunFunction composes the resources an InferenceCluster needs.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) -def test_gateway_hostname_dots_become_a_single_dns_label() -> None: - """A dotted cluster name still gives a hostname whose first label has no dots.""" +# The derived gateway hostname doubles as an SNI and a certificate SAN, so two +# clusters must never derive the same one. These cases call _gateway_hostname +# directly, because through RunFunction each cluster name would need a whole +# compose case, renaming every resource the function names after the cluster. +GATEWAY_HOSTNAME_CASES = [ # A dotted cluster name is a DNS-1123 subdomain, but the first segment of # the hostname has to be one DNS-1035 label. - hostname = fn._gateway_hostname("eu.example") - label = hostname.split(".")[0] - assert "." not in label - assert label.startswith("gateway-") - - -def test_gateway_hostname_names_differing_only_in_dots_do_not_collide() -> None: - """Cluster names that differ only in dots get different hostnames.""" - # The hash covers the raw cluster name, so 'eu.example' and 'eu-example' - # get different hostnames. Sharing one, a cluster's Service would shadow - # the other's under a certificate it accepts. - assert fn._gateway_hostname("eu.example") != fn._gateway_hostname("eu-example") + GatewayHostnameCase( + name="dots become a single DNS label", + cluster_name="eu.example", + want="gateway-eu-example-ad4f6.modelplane-system.svc.cluster.local", + ), + # With the case above, this pins that eu.example and eu-example hash + # differently: the hash covers the raw cluster name, before dots become + # dashes. Sharing one hostname, a cluster's Service would shadow the other's + # under a certificate it accepts. + GatewayHostnameCase( + name="the dashed twin of a dotted name gets its own hash", + cluster_name="eu-example", + want="gateway-eu-example-1a2b0.modelplane-system.svc.cluster.local", + ), +] + + +@pytest.mark.parametrize("case", GATEWAY_HOSTNAME_CASES, ids=lambda case: case.name) +def test_gateway_hostname(case: GatewayHostnameCase) -> None: + """_gateway_hostname derives a cluster's gateway hostname.""" + assert fn._gateway_hostname(case.cluster_name) == case.want diff --git a/functions/compose-inference-gateway/tests/test_fn.py b/functions/compose-inference-gateway/tests/test_fn.py index 9f4379031..214490bc3 100644 --- a/functions/compose-inference-gateway/tests/test_fn.py +++ b/functions/compose-inference-gateway/tests/test_fn.py @@ -15,7 +15,6 @@ """Tests for the compose-inference-gateway function.""" import asyncio -import base64 import dataclasses import json @@ -27,10 +26,7 @@ from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.inferencegateway import v1alpha1 - -_PC = "gw-eu-cluster-kubeconfig" -_CLUSTER = "gw-eu" -_ADDRESS = "34.56.129.3" +from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @dataclasses.dataclass @@ -42,130 +38,93 @@ class Case: want: fnv1.RunFunctionResponse -def _xr(*, name: str = "eu", **spec) -> dict: # noqa: ANN003 - """The InferenceGateway XR, built from the generated model so a field the - XRD doesn't define can't creep into a test.""" - xr = v1alpha1.InferenceGateway( - apiVersion="modelplane.ai/v1alpha1", - kind="InferenceGateway", - metadata={"name": name}, - spec=v1alpha1.Spec(clusterName=_CLUSTER, **spec), - ) - return xr.model_dump(exclude_none=True, mode="json", by_alias=True) - - -def _api_key_auth() -> v1alpha1.Auth: - """Caller auth by API key, against the Secrets _requirements(auth=True) selects.""" - return v1alpha1.Auth( - method="APIKey", - apiKey=v1alpha1.ApiKey( - secretSelector=v1alpha1.SecretSelector(matchLabels={"modelplane.ai/inference-keys": "true"}) - ), +def _xr(*, name: str, tls: bool, caller_secret_labels: dict[str, str] | None) -> fnv1.Resource: + """The XR on gw-eu, with TLS from eu-tls-0 if tls, and API-key auth by Secrets matching caller_secret_labels unless None.""" + tls_spec = v1alpha1.Tls(certificateRefs=[v1alpha1.CertificateRef(name="eu-tls-0")]) if tls else None + auth = None + if caller_secret_labels is not None: + auth = v1alpha1.Auth( + method="APIKey", + apiKey=v1alpha1.ApiKey(secretSelector=v1alpha1.SecretSelector(matchLabels=caller_secret_labels)), + ) + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.InferenceGateway( + apiVersion="modelplane.ai/v1alpha1", + kind="InferenceGateway", + metadata=metav1.ObjectMeta(name=name), + spec=v1alpha1.Spec(clusterName="gw-eu", tls=tls_spec, auth=auth), + ).model_dump(exclude_none=True, mode="json", by_alias=True) + ) ) -def _cluster(*, provider_config: str | None = _PC) -> dict: - """An observed InferenceCluster, optionally without a providerConfigRef. - - A registered cluster with no GPU pools, which is what a region with callers - but no accelerators looks like, and the least a gateway needs. - """ - status: dict = {} - if provider_config: - status["providerConfigRef"] = {"name": provider_config} - return { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceCluster", - "metadata": {"name": _CLUSTER}, - "spec": { - "cluster": { - "source": "Existing", - "existing": {"secretRef": {"name": f"{_CLUSTER}-kubeconfig", "key": "kubeconfig"}}, - } - }, - "status": status, - } +def _desired_xr(*, status: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The desired XR, carrying status, or only its readiness if status is None.""" + if status is None: + return fnv1.Resource(ready=ready) + return fnv1.Resource(resource=resource.dict_to_struct({"status": status}), ready=ready) -def _cluster_with_gateway(name: str, *, address: str, hostname: str) -> dict: - """An observed InferenceCluster whose gateway has published an address and - the internal name Modelplane derived for it.""" - return { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceCluster", - "metadata": {"name": name}, - "spec": { - "cluster": { - "source": "Existing", - "existing": {"secretRef": {"name": f"{name}-kubeconfig", "key": "kubeconfig"}}, +def _cluster(*, provider_config_ref: bool) -> fnv1.Resource: + """The gateway's InferenceCluster, gw-eu, as the clusters requirement returns it.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "gw-eu"}, + "spec": { + "cluster": { + "source": "Existing", + "existing": {"secretRef": {"name": "gw-eu-kubeconfig", "key": "kubeconfig"}}, + } + }, + "status": {"providerConfigRef": {"name": "gw-eu-cluster-kubeconfig"}} if provider_config_ref else {}, } - }, - "status": {"gateway": {"address": address, "hostname": hostname}}, - } + ) + ) -def _gateway_xr(name: str, cluster: str) -> dict: - """Another InferenceGateway, for the one-per-cluster contest.""" - return { +def _inference_gateway(*, name: str, address: str | None) -> fnv1.Resource: + """An InferenceGateway on gw-eu, as the gateways requirement returns it.""" + gateway: dict = { "apiVersion": "modelplane.ai/v1alpha1", "kind": "InferenceGateway", "metadata": {"name": name}, - "spec": {"clusterName": cluster}, - } - - -def _secret(name: str, data: dict[str, str]) -> dict: - """A control-plane Secret, with values base64 encoded as the API server - stores them, since the function copies data verbatim.""" - return { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": name, "namespace": fn.CONTROL_PLANE_NAMESPACE}, - "data": {k: base64.b64encode(v.encode()).decode() for k, v in data.items()}, + "spec": {"clusterName": "gw-eu"}, } + if address is not None: + gateway["status"] = {"address": address} + return fnv1.Resource(resource=resource.dict_to_struct(gateway)) -def _required(**resources) -> dict: # noqa: ANN003 - """Build the request's required_resources map.""" - return { - name: fnv1.Resources(items=[fnv1.Resource(resource=resource.dict_to_struct(r)) for r in items]) - for name, items in resources.items() - } - - -def _requirements(*, auth: bool = False, tls: int = 0) -> fnv1.Requirements: - """The requirements the function always emits, in the order it emits them.""" - reqs = { - "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), - "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), - } - if auth: - reqs["caller-secrets"] = fnv1.ResourceSelector( - api_version="v1", - kind="Secret", - namespace=fn.CONTROL_PLANE_NAMESPACE, - match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), - ) - for i in range(tls): - reqs[f"tls-secret-{i}"] = fnv1.ResourceSelector( - api_version="v1", kind="Secret", namespace=fn.CONTROL_PLANE_NAMESPACE, match_name=f"eu-tls-{i}" +def _caller_key_secret(*, name: str, data: dict[str, str]) -> fnv1.Resource: + """A caller-key Secret, as the caller-secrets requirement returns it.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": name, "namespace": "modelplane-system"}, + "data": data, + } ) - return fnv1.Requirements(resources=reqs) - + ) -def _observed_gateway(address: str | None, *, ready: bool) -> fnv1.Resource: - """The composed Gateway Object as observed, optionally with an address. - lastTransitionTime is fixed so the input is deterministic. - """ - manifest: dict = { - "apiVersion": "gateway.networking.k8s.io/v1", - "kind": "Gateway", - "metadata": {"name": fn._GATEWAY_NAME, "namespace": fn.REMOTE_NAMESPACE}, +def _observed_gateway(*, address: str, ready: bool) -> fnv1.Resource: + """The composed Gateway Object, observed once the Gateway has an address, with a Ready condition only if ready.""" + status: dict = { + "atProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "Gateway", + "metadata": {"name": "inference-gateway", "namespace": "modelplane-system"}, + "status": {"addresses": [{"type": "IPAddress", "value": address}]}, + } + } } - if address: - manifest["status"] = {"addresses": [{"type": "IPAddress", "value": address}]} - status: dict = {"atProvider": {"manifest": manifest}} if ready: status["conditions"] = [ { @@ -186,12 +145,8 @@ def _observed_gateway(address: str | None, *, ready: bool) -> fnv1.Resource: ) -def _observed_accepted() -> fnv1.Resource: - """A composed policy Object as observed once accepted. - - Its readiness comes from a CEL query on the policy's own Accepted condition, - so an Object that merely applied isn't enough. - """ +def _observed_caller_auth() -> fnv1.Resource: + """The composed caller-auth Object, observed Ready once its policy is accepted.""" return fnv1.Resource( resource=resource.dict_to_struct( { @@ -212,353 +167,156 @@ def _observed_accepted() -> fnv1.Resource: ) -def _not_ready(reason: str, message: str, requirements: fnv1.Requirements) -> fnv1.RunFunctionResponse: - """The whole response for a pass that composes nothing: no desired - resources, one GatewayReady=False condition, and the reason as a result.""" - return fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - context=structpb.Struct(), - requirements=requirements, - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=reason, - message=message, - ) - ], - results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message=message)], - ) - - -GATES_CASES = [ - Case( - name="unresolved requirements compose nothing", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - ), - want=_not_ready( - fn.CONDITION_REASON_WAITING_FOR_CLUSTER, - "Waiting for the gateway's cluster and the other gateways to resolve", - _requirements(), - ), - ), - Case( - name="a named cluster that does not exist", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required(clusters=[], gateways=[_gateway_xr("eu", _CLUSTER)]), - ), - want=_not_ready( - fn.CONDITION_REASON_WAITING_FOR_CLUSTER, - f"InferenceCluster {_CLUSTER} does not exist", - _requirements(), - ), - ), - Case( - name="a cluster that already hosts a lower-named gateway", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER), _gateway_xr("aaa", _CLUSTER)], - ), - ), - want=_not_ready( - fn.CONDITION_REASON_CLUSTER_TAKEN, - f"InferenceCluster {_CLUSTER} already hosts InferenceGateway aaa", - _requirements(), - ), - ), - Case( - name="a cluster with no providerConfigRef yet", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required( - clusters=[_cluster(provider_config=None)], gateways=[_gateway_xr("eu", _CLUSTER)] - ), - ), - want=_not_ready( - fn.CONDITION_REASON_WAITING_FOR_CLUSTER, - f"InferenceCluster {_CLUSTER} has not published a providerConfigRef", - _requirements(), - ), - ), -] - - -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - -@pytest.mark.parametrize("case", GATES_CASES, ids=lambda case: case.name) -def test_gates(case: Case) -> None: - """Passes where the gateway can't be composed compose nothing, and say - why. Asserting the whole response proves nothing is composed against a - cluster we can't reach, rather than a subset being applied.""" - got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) - - -def test_minimal_gateway() -> None: - """A gateway with no TLS or auth: the getting-started shape. +# The helpers below build the resources the function composes. Each is a +# provider-kubernetes Object that applies one manifest to the gateway's cluster +# through its ClusterProviderConfig, and every namespaced manifest lands in +# modelplane-system there. +# +# Each Object sets its own namespace too. An InferenceGateway is cluster-scoped, +# and Crossplane only defaults a composed namespaced resource's namespace from a +# namespaced composite. Without it every reconcile fails with "an empty +# namespace may not be set when a resource name is provided" and nothing is +# composed at all. - Composes the gateway objects and no auth policies, and reports no - endpoints until the Gateway has an address. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert sorted(got.desired.resources) == sorted( - [ - # The CA whose client certificates a cluster gateway trusts, - # published as a ClusterIssuer for compose-model-route to issue - # per-namespace client certificates from. - "client-ca-certificate", - "client-ca-issuer", - "client-ca-bundle", - "client-ca-configmap", - "client-selfsigned-issuer", - "client-traffic-policy", - "envoy-proxy", - "failover-policy", - "gateway", - "healthz-filter", - "healthz-route", - ] - ), "composes the gateway objects and its client PKI, and no caller auth" - for key, res in got.desired.resources.items(): - d = resource.struct_to_dict(res.resource) - assert d["kind"] == "Object", f"{key} targets the gateway's cluster" - assert d["spec"]["providerConfigRef"] == {"kind": "ClusterProviderConfig", "name": _PC}, ( - f"{key} uses the cluster's ClusterProviderConfig" - ) - # An InferenceGateway is cluster-scoped, and Crossplane only - # defaults a composed namespaced resource's namespace from a - # namespaced composite. Without this every reconcile fails with - # "an empty namespace may not be set when a resource name is - # provided" and nothing is composed at all. - assert d["metadata"]["namespace"] == fn.CONTROL_PLANE_NAMESPACE, ( - f"{key} sets its own namespace, which a cluster-scoped XR must" +def _envoy_proxy() -> fnv1.Resource: + """The composed EnvoyProxy, configuring the gateway's proxy pods and its usage-record access log.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "EnvoyProxy", + "metadata": {"name": "inference-gateway", "namespace": "modelplane-system"}, + "spec": { + # Two proxy pods spread softly across nodes and + # zones, a disruption budget so a drain can't + # evict both, and ndots:1. Without ndots:1 every + # backend hostname is resolved against each of + # the pod's search domains first, since they all + # have fewer than five dots. A cluster whose + # upstream resolver is slow then stalls + # resolution, and Envoy answers 503 with nothing + # but DNS timeouts to show for it. + "provider": { + "type": "Kubernetes", + "kubernetes": { + "envoyService": {"externalTrafficPolicy": "Cluster"}, + "envoyDeployment": { + "replicas": 2, + "patch": { + "type": "StrategicMerge", + "value": { + "spec": { + "template": { + "spec": { + "dnsConfig": { + "options": [{"name": "ndots", "value": "1"}] + } + } + } + } + }, + }, + "pod": { + "topologySpreadConstraints": [ + { + "maxSkew": 1, + "topologyKey": "kubernetes.io/hostname", + "whenUnsatisfiable": "ScheduleAnyway", + "labelSelector": { + "matchLabels": { + "gateway.envoyproxy.io/owning-gateway-name": "inference-gateway", + "gateway.envoyproxy.io/owning-gateway-namespace": "modelplane-system", + } + }, + }, + { + "maxSkew": 1, + "topologyKey": "topology.kubernetes.io/zone", + "whenUnsatisfiable": "ScheduleAnyway", + "labelSelector": { + "matchLabels": { + "gateway.envoyproxy.io/owning-gateway-name": "inference-gateway", + "gateway.envoyproxy.io/owning-gateway-namespace": "modelplane-system", + } + }, + }, + ] + }, + }, + "envoyPDB": {"maxUnavailable": 1}, + }, + }, + # A stopping pod drains for as long as a request + # may run by default, so a restart doesn't cut + # off streams in flight. + "shutdown": {"drainTimeout": "300s"}, + "telemetry": { + "accessLog": { + "settings": [ + { + "format": { + "type": "JSON", + # The caller and token fields + # read request metadata, not + # the response body or a + # header. The caller header + # is stripped before a + # third-party backend sees + # it, so a log reading the + # header loses the caller on + # exactly the records that + # attribute provider spend. + "json": { + "caller": "%DYNAMIC_METADATA(io.envoy.ai_gateway:caller)%", + "service": "%REQ(X-AI-EG-MODEL)%", + "endpoint": "%DYNAMIC_METADATA(io.envoy.ai_gateway:ai_service_backend_name)%", + "served_model": "%DYNAMIC_METADATA(io.envoy.ai_gateway:model_name_override)%", + "response_model": "%DYNAMIC_METADATA(io.envoy.ai_gateway:response_model)%", + "input_tokens": "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_input_token)%", + "output_tokens": "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_output_token)%", + "total_tokens": "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_total_token)%", + "status": "%RESPONSE_CODE%", + "duration_ms": "%DURATION%", + "start_time": "%START_TIME%", + }, + }, + "sinks": [{"type": "File", "file": {"path": "/dev/stdout"}}], + } + ] + } + }, + }, + } + }, + }, + } ) - manifest = d["spec"]["forProvider"]["manifest"] - if manifest["kind"] == "ClusterIssuer": - # Cluster-scoped: compose-model-route issues client certs from it - # into team namespaces, so it has no namespace of its own. - assert "namespace" not in manifest["metadata"], f"{key} is cluster-scoped, so it sets no namespace" - continue - if manifest["kind"] == "Bundle": - # A Bundle is cluster-scoped, so it has no namespace of its own. - # It picks the namespace it syncs its ConfigMap to by selector. - assert "namespace" not in manifest["metadata"], f"{key} is cluster-scoped, so it sets no namespace" - assert manifest["spec"]["target"]["namespaceSelector"] == { - "matchLabels": {"kubernetes.io/metadata.name": fn.REMOTE_NAMESPACE} - }, f"{key} syncs only to the remote namespace" - continue - assert manifest["metadata"]["namespace"] == fn.REMOTE_NAMESPACE, f"{key} lands in the remote namespace" - - # Two attempts per priority, so a retry tries another endpoint at the - # same priority before moving down. At one, a single transient failure - # on one replica would send the request to the next priority, which may - # be a paid provider. - failover = resource.struct_to_dict(got.desired.resources["failover-policy"].resource) - assert failover["spec"]["forProvider"]["manifest"]["spec"]["retry"] == { - "numAttemptsPerPriority": 2, - "numRetries": 3, - "retryOn": { - # retriable-status-codes has to be present for the status - # codes below to do anything: Envoy Gateway replaces retry_on - # wholesale with this list, and Envoy only consults - # retriable_status_codes when retry_on names it. Without it a - # provider answering 503 or 429 is never retried, which is - # the case failover exists for. - "triggers": [ - "connect-failure", - "refused-stream", - "reset", - "retriable-status-codes", - ], - # 429 so a rate-limited provider's traffic overflows to - # another endpoint rather than failing back to the caller. - "httpStatusCodes": [429, 503], - }, - } - # Panic mode defaults to 50%, above which Envoy ignores health and - # spreads traffic over every endpoint including the ejected ones. Every - # endpoint of a ModelService shares one cluster, so ejecting a whole - # priority tier usually crosses it and failover stops working. - # - # Asserted on the whole healthCheck, because panicThreshold is a sibling - # of passive rather than a field inside it, and nested wrongly the API - # server prunes it while the policy still applies. - assert failover["spec"]["forProvider"]["manifest"]["spec"]["healthCheck"] == { - "passive": { - "baseEjectionTime": "30s", - "consecutive5XxErrors": 5, - "interval": "5s", - "maxEjectionPercent": 100, - }, - "panicThreshold": 0, - } - assert failover["spec"]["forProvider"]["manifest"]["spec"]["targetRefs"] == [ - {"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": fn._GATEWAY_NAME} - ], "targets the Gateway, so it covers every ModelService's route" + ) - # AI Gateway buffers whole bodies, and Envoy Gateway's 32KiB default - # buffer limit answers 413 to a long prompt or non-streamed completion. - assert resource.struct_to_dict(got.desired.resources["client-traffic-policy"].resource)["spec"]["forProvider"][ - "manifest" - ] == { - "apiVersion": "gateway.envoyproxy.io/v1alpha1", - "kind": "ClientTrafficPolicy", - "metadata": {"name": "inference-gateway-client-traffic", "namespace": "modelplane-system"}, - "spec": { - "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": "inference-gateway"}], - "connection": {"bufferLimit": "50Mi"}, - "http2": {"initialStreamWindowSize": "16Mi", "initialConnectionWindowSize": "24Mi"}, - }, - } - # The token fields must read request metadata, not the response body or - # a header. The caller header is stripped before a third-party backend - # sees it, so a log reading the header loses the caller on exactly the - # records that attribute provider spend. - log = resource.struct_to_dict(got.desired.resources["envoy-proxy"].resource) - fields = log["spec"]["forProvider"]["manifest"]["spec"]["telemetry"]["accessLog"]["settings"][0]["format"]["json"] - # Two proxy pods spread softly across nodes and zones, a disruption - # budget so a drain can't evict both, and ndots:1. Without ndots:1 every - # backend hostname is resolved against each of the pod's search domains - # first, since they all have fewer than five dots. A cluster whose - # upstream resolver is slow then stalls resolution, and Envoy answers 503 - # with nothing but DNS timeouts to show for it. - proxy_labels = { - "gateway.envoyproxy.io/owning-gateway-name": "inference-gateway", - "gateway.envoyproxy.io/owning-gateway-namespace": "modelplane-system", - } - assert log["spec"]["forProvider"]["manifest"]["spec"]["provider"] == { - "type": "Kubernetes", - "kubernetes": { - "envoyService": {"externalTrafficPolicy": "Cluster"}, - "envoyDeployment": { - "replicas": 2, - "patch": {"type": "StrategicMerge", "value": fn._NDOTS_PATCH}, - "pod": { - "topologySpreadConstraints": [ - { - "maxSkew": 1, - "topologyKey": "kubernetes.io/hostname", - "whenUnsatisfiable": "ScheduleAnyway", - "labelSelector": {"matchLabels": proxy_labels}, - }, - { - "maxSkew": 1, - "topologyKey": "topology.kubernetes.io/zone", - "whenUnsatisfiable": "ScheduleAnyway", - "labelSelector": {"matchLabels": proxy_labels}, - }, - ] - }, - }, - "envoyPDB": {"maxUnavailable": 1}, +def _gateway(*, https: bool, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Gateway: an HTTP listener, joined by an HTTPS one terminating eu-tls-0 when https is set.""" + http_listener = { + "name": "http", + "protocol": "HTTP", + "port": 80, + "allowedRoutes": { + "namespaces": { + "from": "Selector", + "selector": {"matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}]}, + } }, } - # A stopping pod drains for as long as a request may run by default, so - # a restart doesn't cut off streams in flight. - assert log["spec"]["forProvider"]["manifest"]["spec"]["shutdown"] == {"drainTimeout": "300s"} - assert fn._NDOTS_PATCH["spec"]["template"]["spec"]["dnsConfig"]["options"] == [{"name": "ndots", "value": "1"}] - - assert fields["caller"] == "%DYNAMIC_METADATA(io.envoy.ai_gateway:caller)%" - assert fields["input_tokens"] == "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_input_token)%" - assert fields["output_tokens"] == "%DYNAMIC_METADATA(io.envoy.ai_gateway:llm_output_token)%" - - gw = resource.struct_to_dict(got.desired.resources["gateway"].resource) - manifest = gw["spec"]["forProvider"]["manifest"] - assert manifest["spec"]["listeners"] == [ - { - "name": "http", - "protocol": "HTTP", - "port": 80, - "allowedRoutes": { - "namespaces": { - "from": "Selector", - "selector": {"matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}]}, - } - }, - } - ], "one HTTP listener, no hostname, accepting routes from the mirrored namespaces" - assert manifest["spec"]["infrastructure"]["parametersRef"] == { - "group": "gateway.envoyproxy.io", - "kind": "EnvoyProxy", - "name": fn._GATEWAY_NAME, - }, "its own EnvoyProxy, not the GatewayClass's" - assert resource.struct_to_dict(got.desired.composite.resource).get("status") == {}, ( - "nothing to report until the Gateway has an address" - ) - - -def test_full_gateway() -> None: - """A gateway with TLS and auth, whose Gateway has an address. - - Checks the things a caller depends on: the HTTPS listener, the Secrets - copied to the cluster, the caller policy naming them, /healthz exempted - from that policy, and a status publishing no URLs, since a caller - reaches a TLS gateway on a DNS name only its owner knows. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr( - tls=v1alpha1.Tls(certificateRefs=[v1alpha1.CertificateRef(name="eu-tls-0")]), - auth=_api_key_auth(), - ) - ) - ), - resources={ - "gateway": _observed_gateway(_ADDRESS, ready=True), - "caller-auth": _observed_accepted(), - }, - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{ - "caller-secrets": [_secret("ml-team-keys", {"ml-team-assistant": "sk-mp-a1b2c3"})], - "tls-secret-0": [_secret("eu-tls-0", {"tls.crt": "cert", "tls.key": "key"})], - }, - ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - assert sorted(got.desired.resources) == [ - "caller-auth", - "caller-secret-ml-team-keys", - "client-ca-bundle", - "client-ca-certificate", - "client-ca-configmap", - "client-ca-issuer", - "client-selfsigned-issuer", - "client-traffic-policy", - "envoy-proxy", - "failover-policy", - "gateway", - "healthz-auth", - "healthz-filter", - "healthz-route", - "redirect-auth", - "redirect-route", - "tls-secret-eu-tls-0", - ] - - def manifest(key: str) -> dict: - return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] - - assert manifest("gateway")["spec"]["listeners"][1] == { + https_listener = { "name": "https", "protocol": "HTTPS", "port": 443, @@ -570,596 +328,2496 @@ def manifest(key: str) -> dict: } }, } - assert _to_dict(got.requirements) == _to_dict(_requirements(auth=True, tls=1)) - assert manifest("tls-secret-eu-tls-0") == { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": "eu-tls-0", "namespace": fn.REMOTE_NAMESPACE}, - "type": "kubernetes.io/tls", - "data": { - "tls.crt": base64.b64encode(b"cert").decode(), - "tls.key": base64.b64encode(b"key").decode(), - }, - }, "the certificate is copied verbatim, keeping the name the Gateway refers to it by" - assert manifest("caller-auth")["spec"]["apiKeyAuth"] == { - "credentialRefs": [{"name": "callers-ml-team-keys"}], - # Authorization for OpenAI clients, x-api-key for Anthropic ones. - "extractFrom": [{"headers": ["Authorization", "x-api-key"]}], - "forwardClientIDHeader": fn._CALLER_HEADER, - "sanitize": True, - } - assert manifest("healthz-auth")["spec"] == { - "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "HTTPRoute", "name": fn._HEALTHZ_NAME}], - "authorization": {"defaultAction": "Allow"}, - }, "/healthz overrides the Gateway-level policy so a health check needs no credential" - # Inference binds to the HTTPS listener alone, so :80 carries only - # /healthz and this catch-all redirect to it. /healthz is an Exact match, - # so it still answers a plain-HTTP health check. - assert manifest("healthz-route")["spec"]["parentRefs"] == [ - { - "group": "gateway.networking.k8s.io", - "kind": "Gateway", - "name": fn._GATEWAY_NAME, - "sectionName": "http", - } - ] - assert manifest("redirect-route")["spec"] == { - "parentRefs": [ - { - "group": "gateway.networking.k8s.io", - "kind": "Gateway", - "name": fn._GATEWAY_NAME, - "sectionName": "http", - } - ], - "rules": [ + return fnv1.Resource( + resource=resource.dict_to_struct( { - "matches": [{"path": {"type": "PathPrefix", "value": "/"}}], - "filters": [{"type": "RequestRedirect", "requestRedirect": {"scheme": "https", "statusCode": 301}}], + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status) && has(object.status.addresses) && object.status.addresses.size() > 0", + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "Gateway", + "metadata": {"name": "inference-gateway", "namespace": "modelplane-system"}, + "spec": { + "gatewayClassName": "envoy", + "infrastructure": { + "parametersRef": { + "group": "gateway.envoyproxy.io", + "kind": "EnvoyProxy", + "name": "inference-gateway", + } + }, + "listeners": [http_listener, https_listener] if https else [http_listener], + }, + } + }, + }, } - ], - } - assert manifest("redirect-auth")["spec"] == { - "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "HTTPRoute", "name": fn._REDIRECT_NAME}], - "authorization": {"defaultAction": "Allow"}, - }, "the redirect must happen before auth, or an unauthenticated caller gets 401 instead of being sent to HTTPS" - assert resource.struct_to_dict(got.desired.composite.resource)["status"] == {"address": _ADDRESS} - assert [_to_dict(c) for c in got.conditions] == [ - _to_dict( - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_TRUE, - reason=fn.CONDITION_REASON_GATEWAY_PROGRAMMED, - ) - ) - ] - - -def test_endpoints_are_built_from_the_address() -> None: - """A gateway serving plain HTTP publishes URLs on its address, which is - something a caller can actually put in an SDK's base_url.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={"gateway": _observed_gateway(_ADDRESS, ready=False)}, ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert resource.struct_to_dict(got.desired.composite.resource)["status"] == { - "address": _ADDRESS, - "endpoints": { - "openAI": f"http://{_ADDRESS}/v1", - "anthropic": f"http://{_ADDRESS}/anthropic/v1", - }, - } - assert next(iter(got.conditions)).reason == fn.CONDITION_REASON_WAITING_FOR_GATEWAY, ( - "an address alone isn't readiness; the Gateway must be programmed" + ready=ready, ) -def test_an_ipv6_address_is_bracketed_in_the_endpoints() -> None: - """A bare IPv6 literal collides with the port separator in a URL, so an - SDK given http://2001:db8::1/v1 as a base_url can't use it.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={"gateway": _observed_gateway("2001:db8::1", ready=False)}, - ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), +def _client_selfsigned_issuer() -> fnv1.Resource: + """The composed self-signed Issuer that signs the client CA.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "Issuer", + "metadata": {"name": "inference-gateway-selfsigned", "namespace": "modelplane-system"}, + "spec": {"selfSigned": {}}, + } + }, + }, + } + ) ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert resource.struct_to_dict(got.desired.composite.resource)["status"]["endpoints"] == { - "openAI": "http://[2001:db8::1]/v1", - "anthropic": "http://[2001:db8::1]/anthropic/v1", - } - -def test_resolves_each_cluster_gateway_name() -> None: - """A Service per cluster gateway, resolving its name to its address here. - - A ModelService's backends address a cluster gateway by the name - compose-inference-cluster derived, and this gateway's Envoy resolves it, - so its cluster needs a Service of that name. An IP is served by a - headless Service and an EndpointSlice; a hostname, which is how a cloud - load balancer names itself, by an ExternalName Service. A cluster that - hasn't published both an address and a name gets neither. - """ - ipv4 = "prod-ipv4-gateway-aaaaa.modelplane-system.svc.cluster.local" - ipv6 = "prod-ipv6-gateway-bbbbb.modelplane-system.svc.cluster.local" - dns = "prod-dns-gateway-ccccc.modelplane-system.svc.cluster.local" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required( - gateways=[_gateway_xr("eu", _CLUSTER)], - clusters=[ - _cluster(), # this gateway's own cluster, no gateway published yet - _cluster_with_gateway("prod-ipv4", address="203.0.113.7", hostname=ipv4), - _cluster_with_gateway("prod-ipv6", address="2001:db8::1", hostname=ipv6), - _cluster_with_gateway("prod-dns", address="lb-x.elb.amazonaws.com", hostname=dns), - ], - ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - resolvers = { - key: resource.struct_to_dict(res.resource) - for key, res in got.desired.resources.items() - if key.startswith("cluster-name") - } - for key, obj in resolvers.items(): - assert obj["spec"]["providerConfigRef"] == {"kind": "ClusterProviderConfig", "name": _PC}, ( - f"{key} is composed against this gateway's own cluster" +def _client_ca_certificate(*, common_name: str) -> fnv1.Resource: + """The composed Certificate for the client CA, whose client certificates a cluster gateway trusts.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.conditions) && " + "object.status.conditions.exists(c, c.type == 'Ready' && c.status == 'True')" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "Certificate", + "metadata": {"name": "inference-gateway-ca", "namespace": "modelplane-system"}, + "spec": { + "isCA": True, + "commonName": common_name, + "secretName": "inference-gateway-ca", + "duration": "87600h", + "renewBefore": "8760h", + "privateKey": {"algorithm": "ECDSA", "size": 256}, + "issuerRef": { + "name": "inference-gateway-selfsigned", + "kind": "Issuer", + "group": "cert-manager.io", + }, + }, + } + }, + }, + } ) - manifests = {key: obj["spec"]["forProvider"]["manifest"] for key, obj in resolvers.items()} - assert manifests == { - "cluster-name-prod-ipv4-gateway-aaaaa": { - "apiVersion": "v1", - "kind": "Service", - "metadata": {"name": "prod-ipv4-gateway-aaaaa", "namespace": fn.REMOTE_NAMESPACE}, - "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, - }, - "cluster-name-slice-prod-ipv4-gateway-aaaaa": { - "apiVersion": "discovery.k8s.io/v1", - "kind": "EndpointSlice", - "metadata": { - "name": "prod-ipv4-gateway-aaaaa", - "namespace": fn.REMOTE_NAMESPACE, - "labels": {"kubernetes.io/service-name": "prod-ipv4-gateway-aaaaa"}, - }, - "addressType": "IPv4", - "ports": [{"name": "https", "port": 443}], - "endpoints": [{"addresses": ["203.0.113.7"], "conditions": {"ready": True}}], - }, - "cluster-name-prod-ipv6-gateway-bbbbb": { - "apiVersion": "v1", - "kind": "Service", - "metadata": {"name": "prod-ipv6-gateway-bbbbb", "namespace": fn.REMOTE_NAMESPACE}, - "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, - }, - "cluster-name-slice-prod-ipv6-gateway-bbbbb": { - "apiVersion": "discovery.k8s.io/v1", - "kind": "EndpointSlice", - "metadata": { - "name": "prod-ipv6-gateway-bbbbb", - "namespace": fn.REMOTE_NAMESPACE, - "labels": {"kubernetes.io/service-name": "prod-ipv6-gateway-bbbbb"}, - }, - "addressType": "IPv6", - "ports": [{"name": "https", "port": 443}], - "endpoints": [{"addresses": ["2001:db8::1"], "conditions": {"ready": True}}], - }, - "cluster-name-prod-dns-gateway-ccccc": { - "apiVersion": "v1", - "kind": "Service", - "metadata": {"name": "prod-dns-gateway-ccccc", "namespace": fn.REMOTE_NAMESPACE}, - "spec": {"type": "ExternalName", "externalName": "lb-x.elb.amazonaws.com"}, - }, - }, ( - "IP clusters get a headless Service + EndpointSlice, the hostname cluster an ExternalName, " - "and the own cluster with nothing published gets neither" ) -def test_certificate_common_names_fit_the_x509_limit() -> None: - """A long gateway name must not push a certificate commonName past the - 64-byte X.509 limit, which cert-manager's webhook rejects. A gateway name - is a cluster-scoped resource name, so it can be up to 253 characters.""" - long_name = "g" + "a" * 62 - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(name=long_name)))), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr(long_name, _CLUSTER)]), +def _client_ca_issuer() -> fnv1.Resource: + """The composed ClusterIssuer, backed by the client CA, that compose-model-route issues client certificates from.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "ClusterIssuer", + "metadata": {"name": "inference-gateway-ca"}, + "spec": {"ca": {"secretName": "inference-gateway-ca"}}, + } + }, + }, + } + ) ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - manifest = resource.struct_to_dict(got.desired.resources["client-ca-certificate"].resource)["spec"]["forProvider"][ - "manifest" - ] - cn = manifest["spec"]["commonName"] - assert len(cn.encode()) <= 64, "client-ca-certificate commonName exceeds the 64-byte X.509 limit" -def test_a_rejected_caller_policy_is_not_ready() -> None: - """A gateway whose caller policy was rejected refuses every request with - a 500 while its Gateway still has an address. Envoy Gateway rejects the - policy when two selected Secrets share a key value, so this is reachable - by writing two Secrets.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth()))), - # The Gateway is programmed; the policy is not accepted. - resources={"gateway": _observed_gateway(_ADDRESS, ready=True)}, - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{"caller-secrets": [_secret("ml-team-keys", {"a": "sk-1"})]}, - ), +def _client_ca_bundle() -> fnv1.Resource: + """The composed trust-manager Bundle copying the client CA's certificate, without its key, into a ConfigMap.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.conditions) && " + "object.status.conditions.exists(c, c.type == 'Synced' && c.status == 'True')" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "trust.cert-manager.io/v1alpha1", + "kind": "Bundle", + "metadata": {"name": "inference-gateway-ca"}, + "spec": { + "sources": [{"secret": {"name": "inference-gateway-ca", "key": "ca.crt"}}], + "target": { + "configMap": {"key": "ca.crt"}, + "namespaceSelector": { + "matchLabels": {"kubernetes.io/metadata.name": "modelplane-system"} + }, + }, + }, + } + }, + }, + } + ) ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - cond = next(iter(got.conditions)) - assert cond.status == fnv1.STATUS_CONDITION_FALSE - assert cond.reason == fn.CONDITION_REASON_AUTH_NOT_ACCEPTED -SHARED_CALLER_KEY_CASES = [ - ( - "two Secrets share a key", - [_secret("team-a-keys", {"a": "sk-1"}), _secret("team-b-keys", {"b": "sk-1"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, - message="Caller team-b-keys/b has the same key as team-a-keys/a, so Envoy Gateway rejects the " - "caller authentication policy and every request is refused", - ), - ), - ( - "one Secret shares a key", - [_secret("team-a-keys", {"a": "sk-1", "z": "sk-1"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, - message="Caller team-a-keys/z has the same key as team-a-keys/a, so Envoy Gateway rejects the " - "caller authentication policy and every request is refused", - ), - ), - ( - # Listed out of order. Walked in name order, team-a's x is seen - # first, so team-b's x is skipped and its y repeats x's key. - # Walked as listed, team-a's x would be the skipped one. - "Secrets are walked in name order", - [_secret("team-b-keys", {"x": "sk-2", "y": "sk-1"}), _secret("team-a-keys", {"x": "sk-1"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_AUTH_NOT_ACCEPTED, - message="Caller team-b-keys/y has the same key as team-a-keys/x, so Envoy Gateway rejects the " - "caller authentication policy and every request is refused", - ), - ), - ( - "a repeated caller name is skipped, whatever its key", - [_secret("team-a-keys", {"a": "sk-1"}), _secret("team-b-keys", {"a": "sk-1", "b": "sk-2"})], - fnv1.Condition( - type=fn.CONDITION_TYPE_GATEWAY_READY, - status=fnv1.STATUS_CONDITION_TRUE, - reason=fn.CONDITION_REASON_GATEWAY_PROGRAMMED, - ), - ), -] +def _client_ca_configmap() -> fnv1.Resource: + """The composed Object observing, without writing, the ConfigMap the client CA Bundle syncs.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "managementPolicies": ["Observe"], + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "inference-gateway-ca", "namespace": "modelplane-system"}, + } + }, + }, + } + ) + ) -@pytest.mark.parametrize("case", SHARED_CALLER_KEY_CASES, ids=lambda case: case[0]) -def test_a_shared_caller_key_is_not_ready_before_the_policy_is_observed( - case: tuple[str, list[dict], fnv1.Condition], -) -> None: - """Envoy Gateway rejects the caller policy when two callers share a key, - but the policy's Object still reads as accepted until provider-kubernetes - next observes it. The gateway reports the outage from the Secrets - themselves, and skips a repeated caller name before comparing its key, - as Envoy Gateway does.""" - _, secrets, want = case - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth()))), - resources={ - "gateway": _observed_gateway(_ADDRESS, ready=True), - "caller-auth": _observed_accepted(), - }, - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{"caller-secrets": secrets}, - ), +def _caller_secret(*, name: str, data: dict[str, str]) -> fnv1.Resource: + """A composed copy of a caller-key Secret, with its data copied verbatim.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": name, "namespace": "modelplane-system"}, + "type": "Opaque", + "data": data, + } + }, + }, + } + ) ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert [_to_dict(c) for c in got.conditions] == [_to_dict(want)] -def test_caller_secrets_are_listed_in_name_order() -> None: - """Envoy Gateway keeps the first Secret listed when two hold the same - caller name, so the policy lists them by name rather than in the order - they resolved in, and the winner doesn't change between reconciles.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth())))), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{ - "caller-secrets": [ - _secret("team-b-keys", {"b": "sk-2"}), - _secret("team-a-keys", {"a": "sk-1"}), - ] - }, +def _caller_auth(*, credential_refs: list[dict], ready: fnv1.Ready) -> fnv1.Resource: + """The composed SecurityPolicy authenticating callers by API key, against the Secrets credential_refs names.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "SecurityPolicy", + "metadata": {"name": "inference-gateway-callers", "namespace": "modelplane-system"}, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + } + ], + "apiKeyAuth": { + "credentialRefs": credential_refs, + # Authorization for OpenAI clients, x-api-key + # for Anthropic ones. + "extractFrom": [{"headers": ["Authorization", "x-api-key"]}], + "forwardClientIDHeader": "x-modelplane-caller", + "sanitize": True, + }, + }, + } + }, + }, + } ), + ready=ready, ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - policy = resource.struct_to_dict(got.desired.resources["caller-auth"].resource) - assert policy["spec"]["forProvider"]["manifest"]["spec"]["apiKeyAuth"]["credentialRefs"] == [ - {"name": "callers-team-a-keys"}, - {"name": "callers-team-b-keys"}, - ] -MISSING_CALLER_SECRET_CASES = [ - ( - "the selector matches no Secret", - {"caller-secrets": []}, - "spec.auth.apiKey.secretSelector matches no Secret, so no caller could authenticate", - ), - ( - "the caller Secrets have not resolved yet", - {}, - "Waiting for caller key Secrets to resolve", - ), -] +def _failover_policy() -> fnv1.Resource: + """The composed BackendTrafficPolicy that retries a failed request and ejects an endpoint that keeps failing.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "BackendTrafficPolicy", + "metadata": {"name": "inference-gateway-failover", "namespace": "modelplane-system"}, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + } + ], + "retry": { + # Two attempts per priority, so a retry tries + # another endpoint at the same priority + # before moving down. At one, a single + # transient failure on one replica would + # send the request to the next priority, + # which may be a paid provider. + "numAttemptsPerPriority": 2, + "numRetries": 3, + "retryOn": { + # retriable-status-codes has to be + # present for the status codes below to + # do anything: Envoy Gateway replaces + # retry_on wholesale with this list, and + # Envoy only consults + # retriable_status_codes when retry_on + # names it. Without it a provider + # answering 503 or 429 is never retried, + # which is the case failover exists for. + "triggers": [ + "connect-failure", + "refused-stream", + "reset", + "retriable-status-codes", + ], + # 429 so a rate-limited provider's + # traffic overflows to another endpoint + # rather than failing back to the caller. + "httpStatusCodes": [429, 503], + }, + }, + # Panic mode defaults to 50%, above which Envoy + # ignores health and spreads traffic over every + # endpoint including the ejected ones. Every + # endpoint of a ModelService shares one cluster, + # so ejecting a whole priority tier usually + # crosses it and failover stops working. + # panicThreshold is a sibling of passive rather + # than a field inside it: nested wrongly, the API + # server prunes it while the policy still + # applies. + "healthCheck": { + "passive": { + "baseEjectionTime": "30s", + "consecutive5XxErrors": 5, + "interval": "5s", + "maxEjectionPercent": 100, + }, + "panicThreshold": 0, + }, + }, + } + }, + }, + } + ) + ) -@pytest.mark.parametrize("case", MISSING_CALLER_SECRET_CASES, ids=lambda case: case[0]) -def test_a_missing_caller_secret_denies_but_keeps_the_gateway(case: tuple[str, dict, str]) -> None: - """Auth is asked for but no caller Secret has resolved. The Gateway is - still composed, so its load balancer and address survive, and its caller - policy denies every request rather than leaving the door open. Two states - reach this, the selector matching no Secret and the requirement not having - resolved yet, differing only in the reason reported.""" - _, extra, want_message = case - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(auth=_api_key_auth())))), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)], **extra), +def _client_traffic_policy() -> fnv1.Resource: + """The composed ClientTrafficPolicy that lets a whole request or response body fit in the proxy's buffer.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "ClientTrafficPolicy", + "metadata": {"name": "inference-gateway-client-traffic", "namespace": "modelplane-system"}, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + } + ], + "connection": {"bufferLimit": "50Mi"}, + "http2": {"initialStreamWindowSize": "16Mi", "initialConnectionWindowSize": "24Mi"}, + }, + } + }, + }, + } + ) ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert "gateway" in got.desired.resources, "the Gateway is kept, so its address survives" - assert not any(key.startswith("caller-secret-") for key in got.desired.resources), ( - "no caller Secret resolved, so none is copied to the cluster" + +def _healthz_filter() -> fnv1.Resource: + """The composed HTTPRouteFilter answering /healthz with a 200.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "HTTPRouteFilter", + "metadata": {"name": "inference-gateway-healthz", "namespace": "modelplane-system"}, + "spec": { + "directResponse": { + "statusCode": 200, + "contentType": "application/json", + "body": {"type": "Inline", "inline": '{"status":"ok"}'}, + } + }, + } + }, + }, + } + ) ) - spec = resource.struct_to_dict(got.desired.resources["caller-auth"].resource)["spec"]["forProvider"]["manifest"][ - "spec" - ] - assert spec == { - "targetRefs": [{"group": "gateway.networking.k8s.io", "kind": "Gateway", "name": fn._GATEWAY_NAME}], - "authorization": {"defaultAction": "Deny"}, - }, "with no caller key the policy denies every request rather than authenticating nobody by omission" - cond = next(iter(got.conditions)) - assert cond.status == fnv1.STATUS_CONDITION_FALSE - assert cond.reason == fn.CONDITION_REASON_SECRETS_MISSING - assert cond.message == want_message -def test_a_missing_tls_secret_keeps_the_gateway() -> None: - """A referenced TLS Secret hasn't resolved. The Gateway is still composed, - so its address survives; the HTTPS listener is left without a certificate - on the cluster until the Secret appears, rather than the whole Gateway - withdrawn and its load balancer moved.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr(tls=v1alpha1.Tls(certificateRefs=[v1alpha1.CertificateRef(name="eu-tls-0")])) - ) - ) - ), - required_resources=_required( - clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)], **{"tls-secret-0": []} - ), +def _healthz_route() -> fnv1.Resource: + """The composed HTTPRoute serving /healthz on the HTTP listener, as an Exact match that outranks any redirect.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "HTTPRoute", + "metadata": {"name": "inference-gateway-healthz", "namespace": "modelplane-system"}, + "spec": { + "parentRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + "sectionName": "http", + } + ], + "rules": [ + { + "matches": [{"path": {"type": "Exact", "value": "/healthz"}}], + "filters": [ + { + "type": "ExtensionRef", + "extensionRef": { + "group": "gateway.envoyproxy.io", + "kind": "HTTPRouteFilter", + "name": "inference-gateway-healthz", + }, + } + ], + } + ], + }, + } + }, + }, + } + ) ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - assert "gateway" in got.desired.resources, "the Gateway is kept, so its address survives" - assert "tls-secret-eu-tls-0" not in got.desired.resources, "the missing Secret isn't copied to the cluster" - listeners = resource.struct_to_dict(got.desired.resources["gateway"].resource)["spec"]["forProvider"]["manifest"][ - "spec" - ]["listeners"] - assert [ln["name"] for ln in listeners] == ["http", "https"], "the HTTPS listener is still declared" - cond = next(iter(got.conditions)) - assert cond.status == fnv1.STATUS_CONDITION_FALSE - assert cond.reason == fn.CONDITION_REASON_SECRETS_MISSING - assert cond.message == "Waiting for TLS Secrets: eu-tls-0" -def test_the_incumbent_keeps_its_cluster() -> None: - """A gateway created later must not take a cluster off one already - serving traffic. Doing so would delete the incumbent's Gateway and bring - its load balancer back on a different address.""" - # "aaa" sorts before "zzz" but "zzz" already has an address. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceGateway", - "metadata": {"name": "aaa"}, - "spec": {"clusterName": _CLUSTER}, - } - ) - ) - ), - required_resources=_required( - clusters=[_cluster()], - gateways=[ - _gateway_xr("aaa", _CLUSTER), - {**_gateway_xr("zzz", _CLUSTER), "status": {"address": _ADDRESS}}, - ], - ), +def _healthz_auth() -> fnv1.Resource: + """The composed SecurityPolicy letting /healthz past caller auth, so a health check needs no credential.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "SecurityPolicy", + "metadata": {"name": "inference-gateway-healthz-open", "namespace": "modelplane-system"}, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "HTTPRoute", + "name": "inference-gateway-healthz", + } + ], + "authorization": {"defaultAction": "Allow"}, + }, + } + }, + }, + } + ) ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert len(got.desired.resources) == 0, "the newcomer composes nothing" - cond = next(iter(got.conditions)) - assert cond.reason == fn.CONDITION_REASON_CLUSTER_TAKEN - assert "zzz" in cond.message -def test_no_composed_object_observes_a_secret() -> None: - """No composed Object reads a Secret, which is what keeps this gateway's - client CA private key off the control plane. +def _redirect_route() -> fnv1.Resource: + """The composed HTTPRoute redirecting everything on :80 but /healthz to HTTPS.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-cluster-kubeconfig"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "HTTPRoute", + "metadata": {"name": "inference-gateway-redirect", "namespace": "modelplane-system"}, + "spec": { + "parentRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + "sectionName": "http", + } + ], + "rules": [ + { + "matches": [{"path": {"type": "PathPrefix", "value": "/"}}], + "filters": [ + { + "type": "RequestRedirect", + "requestRedirect": {"scheme": "https", "statusCode": 301}, + } + ], + } + ], + }, + } + }, + }, + } + ) + ) - provider-kubernetes copies an observed object's whole manifest into the - Object's status, and its --sanitize-secrets flag defaults to false, so - observing a Secret publishes every key in it to anyone who can get - objects. This CA signs the certificate every cluster gateway in the fleet - accepts, so leaking its key means anyone can reach any engine. - Asserted over everything composed rather than over the PKI, because the - cost of reintroducing this anywhere is the same. +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + - Observing is the case that matters here. The Secrets this function - *writes* also end up in status, because provider-kubernetes reports what - it observes of what it manages, so this alone doesn't keep their contents - off the control plane. Those hold caller keys and serving certificates - that came from control-plane Secrets to begin with, so the exposure is a - wider audience for data already present rather than data that would - otherwise never be there, and prerequisites.yaml runs - provider-kubernetes with --sanitize-secrets to redact it. A CA private - key is different in kind: it is generated on the workload cluster and - observing it is the only way it could ever reach the control plane. - """ - # Auth and TLS both on, so the Secret-copying path is exercised: without - # them this function composes no Secret at all and the assertion holds - # vacuously. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr( - tls={"certificateRefs": [{"name": "eu-tls-0"}]}, - auth={"method": "APIKey", "apiKey": {"secretSelector": {"matchLabels": {"team": "ml"}}}}, - ) +# Secret data is base64 encoded, as the API server stores it: c2stMQ== is +# "sk-1", c2stMg== "sk-2", c2stbXAtYTFiMmMz "sk-mp-a1b2c3", Y2VydA== "cert" and +# a2V5 "key". +COMPOSE_CASES = [ + # Passes where the gateway can't be composed compose nothing, and say why. + # The whole response shows nothing is composed against a cluster the + # gateway can't reach or doesn't own, rather than a subset being applied. + Case( + name="unresolved requirements compose nothing", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=False, caller_secret_labels=None)), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_xr(status=None, ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for the gateway's cluster and the other gateways to resolve", ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + message="Waiting for the gateway's cluster and the other gateways to resolve", + ) + ], ), - required_resources=_required( - clusters=[_cluster()], - gateways=[_gateway_xr("eu", _CLUSTER)], - **{ - "caller-secrets": [_secret("ml-team-keys", {"alice": "key"})], - "tls-secret-0": [_secret("eu-tls-0", {"tls.crt": "cert", "tls.key": "key"})], + ), + Case( + name="a named cluster that does not exist composes nothing", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=False, caller_secret_labels=None)), + required_resources={ + "clusters": fnv1.Resources(), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), }, ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - composed_secrets = [] - observed_secrets = [] - for key, res in got.desired.resources.items(): - d = resource.struct_to_dict(res.resource) - manifest = d["spec"]["forProvider"]["manifest"] - if manifest["kind"] != "Secret": - continue - composed_secrets.append(key) - if "Observe" in d["spec"].get("managementPolicies", []): - observed_secrets.append(key) - assert observed_secrets == [], "these observe a Secret, so its private keys reach the control plane" - assert composed_secrets != [], "no Secret composed, so the assertion above proves nothing" - - -def test_client_pki_publishes_the_ca_without_its_key() -> None: - """The client CA's certificate reaches the control plane through a - trust-manager Bundle, which copies one named key into a ConfigMap, rather - than through the Secret that also holds the private key.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - def manifest(key: str) -> dict: - return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] - - assert manifest("client-ca-bundle") == { - "apiVersion": "trust.cert-manager.io/v1alpha1", - "kind": "Bundle", - "metadata": {"name": "inference-gateway-ca"}, - "spec": { - "sources": [{"secret": {"name": "inference-gateway-ca", "key": "ca.crt"}}], - "target": { - "configMap": {"key": "ca.crt"}, - "namespaceSelector": {"matchLabels": {"kubernetes.io/metadata.name": "modelplane-system"}}, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_xr(status=None, ready=fnv1.READY_FALSE)), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="InferenceCluster gw-eu does not exist")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + message="InferenceCluster gw-eu does not exist", + ) + ], + ), + ), + Case( + name="a cluster that already hosts a lower-named gateway composes nothing", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=False, caller_secret_labels=None)), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources( + items=[_inference_gateway(name="eu", address=None), _inference_gateway(name="aaa", address=None)] + ), }, - }, - } - # Named after the Bundle, because that's the ConfigMap a Bundle syncs. - assert manifest("client-ca-configmap") == { - "apiVersion": "v1", - "kind": "ConfigMap", - "metadata": {"name": "inference-gateway-ca", "namespace": "modelplane-system"}, - } - assert resource.struct_to_dict(got.desired.resources["client-ca-configmap"].resource)["spec"][ - "managementPolicies" - ] == ["Observe"], "trust-manager owns this ConfigMap; Crossplane must not write it" - - -def test_client_ca_published_from_the_observed_configmap() -> None: - """status.clientCACertificate comes from the ConfigMap trust-manager - syncs, as plain text rather than base64. A cluster only trusts this - gateway once it has it, so nothing reaches an engine before it appears. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={ - "gateway": _observed_gateway("gw.example.org", ready=True), - "client-ca-configmap": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "status": { - "atProvider": { - "manifest": { - "apiVersion": "v1", - "kind": "ConfigMap", - "data": {"ca.crt": "-----BEGIN CERTIFICATE-----\nclient\n"}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_xr(status=None, ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="InferenceCluster gw-eu already hosts InferenceGateway aaa" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ClusterAlreadyHasGateway", + message="InferenceCluster gw-eu already hosts InferenceGateway aaa", + ) + ], + ), + ), + # A gateway must not take a cluster off one already serving traffic. Doing + # so would delete the incumbent's Gateway and bring its load balancer back + # on a different address. "aaa" sorts before "zzz", but "zzz" already has an + # address. + Case( + name="the incumbent keeps its cluster", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="aaa", tls=False, caller_secret_labels=None)), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources( + items=[ + _inference_gateway(name="aaa", address=None), + _inference_gateway(name="zzz", address="34.56.129.3"), + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_xr(status=None, ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="InferenceCluster gw-eu already hosts InferenceGateway zzz" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ClusterAlreadyHasGateway", + message="InferenceCluster gw-eu already hosts InferenceGateway zzz", + ) + ], + ), + ), + Case( + name="a cluster with no providerConfigRef yet composes nothing", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=False, caller_secret_labels=None)), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=False)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_xr(status=None, ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="InferenceCluster gw-eu has not published a providerConfigRef", + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + message="InferenceCluster gw-eu has not published a providerConfigRef", + ) + ], + ), + ), + # The getting-started shape: no TLS or auth. Composes the gateway objects + # and its client PKI, and no caller auth. There's nothing to report in + # status until the Gateway has an address. + Case( + name="a gateway with no TLS or auth composes no caller auth or redirect", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=False, caller_secret_labels=None)), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="Waiting for the Gateway on cluster gw-eu to be programmed" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="Waiting for the Gateway on cluster gw-eu to be programmed", + ) + ], + ), + ), + # A gateway serving plain HTTP publishes URLs on its address, which is + # something a caller can actually put in an SDK's base_url. An address + # alone isn't readiness: the Gateway must be programmed. + Case( + name="endpoints are built from the address", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels=None), + resources={"gateway": _observed_gateway(address="34.56.129.3", ready=False)}, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "34.56.129.3", + "endpoints": { + "openAI": "http://34.56.129.3/v1", + "anthropic": "http://34.56.129.3/anthropic/v1", + }, + }, + ready=fnv1.READY_FALSE, + ), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="Waiting for the Gateway on cluster gw-eu to be programmed" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="Waiting for the Gateway on cluster gw-eu to be programmed", + ) + ], + ), + ), + # A bare IPv6 literal collides with the port separator in a URL, so an SDK + # given http://2001:db8::1/v1 as a base_url can't use it. + Case( + name="an IPv6 address is bracketed in the endpoints", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels=None), + resources={"gateway": _observed_gateway(address="2001:db8::1", ready=False)}, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "2001:db8::1", + "endpoints": { + "openAI": "http://[2001:db8::1]/v1", + "anthropic": "http://[2001:db8::1]/anthropic/v1", + }, + }, + ready=fnv1.READY_FALSE, + ), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="Waiting for the Gateway on cluster gw-eu to be programmed" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="Waiting for the Gateway on cluster gw-eu to be programmed", + ) + ], + ), + ), + # A long gateway name must not push the client CA's commonName past the + # 64-byte X.509 limit, which cert-manager's webhook rejects. A gateway name + # is a cluster-scoped resource name, so it can be up to 253 characters. This + # one is 63, and the commonName is cut to 64 bytes. + Case( + name="the client CA commonName fits the X.509 limit", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + name="gaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + tls=False, + caller_secret_labels=None, + ) + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="gaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", address=None + ) + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate( + common_name="Modelplane InferenceGateway CA gaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + ), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="Waiting for the Gateway on cluster gw-eu to be programmed" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="Waiting for the Gateway on cluster gw-eu to be programmed", + ) + ], + ), + ), + # status.clientCACertificate comes from the ConfigMap trust-manager syncs, + # as plain text rather than base64. A cluster only trusts this gateway once + # it has it, so nothing reaches an engine before it appears. + Case( + name="the client CA is published from the observed ConfigMap", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels=None), + resources={ + "gateway": _observed_gateway(address="gw.example.org", ready=True), + "client-ca-configmap": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "status": { + "atProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "data": {"ca.crt": "-----BEGIN CERTIFICATE-----\nclient\n"}, + } } + }, + } + ), + ), + }, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "gw.example.org", + "clientCACertificate": "-----BEGIN CERTIFICATE-----\nclient\n", + "endpoints": { + "openAI": "http://gw.example.org/v1", + "anthropic": "http://gw.example.org/anthropic/v1", + }, + }, + ready=fnv1.READY_UNSPECIFIED, + ), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition(type="GatewayReady", status=fnv1.STATUS_CONDITION_TRUE, reason="GatewayProgrammed"), + ], + ), + ), + # With no observed ConfigMap the gateway publishes no CA, so no cluster + # trusts it yet and no cluster publishes a hostname on its account. + Case( + name="no client CA before the Bundle syncs", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels=None), + resources={"gateway": _observed_gateway(address="gw.example.org", ready=True)}, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "gw.example.org", + "endpoints": { + "openAI": "http://gw.example.org/v1", + "anthropic": "http://gw.example.org/anthropic/v1", + }, + }, + ready=fnv1.READY_UNSPECIFIED, + ), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition(type="GatewayReady", status=fnv1.STATUS_CONDITION_TRUE, reason="GatewayProgrammed"), + ], + ), + ), + # A Service per cluster gateway, resolving its name to its address here. A + # ModelService's backends address a cluster gateway by the name + # compose-inference-cluster derived, and this gateway's Envoy resolves it, + # so its cluster needs a Service of that name. An IP is served by a headless + # Service and an EndpointSlice; a hostname, which is how a cloud load + # balancer names itself, by an ExternalName Service. This gateway's own + # cluster has published no gateway address or name, so it gets no Service. + # Every one is composed against this gateway's own cluster. + Case( + name="resolves each cluster gateway's name", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=False, caller_secret_labels=None)), + required_resources={ + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "clusters": fnv1.Resources( + items=[ + _cluster(provider_config_ref=True), + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "prod-ipv4"}, + "spec": { + "cluster": { + "source": "Existing", + "existing": { + "secretRef": {"name": "prod-ipv4-kubeconfig", "key": "kubeconfig"} + }, + } + }, + "status": { + "gateway": { + "address": "203.0.113.7", + "hostname": "prod-ipv4-gateway-aaaaa.modelplane-system.svc.cluster.local", + } + }, } - }, - } + ) + ), + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "prod-ipv6"}, + "spec": { + "cluster": { + "source": "Existing", + "existing": { + "secretRef": {"name": "prod-ipv6-kubeconfig", "key": "kubeconfig"} + }, + } + }, + "status": { + "gateway": { + "address": "2001:db8::1", + "hostname": "prod-ipv6-gateway-bbbbb.modelplane-system.svc.cluster.local", + } + }, + } + ) + ), + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "prod-dns"}, + "spec": { + "cluster": { + "source": "Existing", + "existing": { + "secretRef": {"name": "prod-dns-kubeconfig", "key": "kubeconfig"} + }, + } + }, + "status": { + "gateway": { + "address": "lb-x.elb.amazonaws.com", + "hostname": "prod-dns-gateway-ccccc.modelplane-system.svc.cluster.local", + } + }, + } + ) + ), + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "cluster-name-prod-ipv4-gateway-aaaaa": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Service", + "metadata": { + "name": "prod-ipv4-gateway-aaaaa", + "namespace": "modelplane-system", + }, + "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, + } + }, + }, + } + ) + ), + "cluster-name-slice-prod-ipv4-gateway-aaaaa": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "discovery.k8s.io/v1", + "kind": "EndpointSlice", + "metadata": { + "name": "prod-ipv4-gateway-aaaaa", + "namespace": "modelplane-system", + "labels": {"kubernetes.io/service-name": "prod-ipv4-gateway-aaaaa"}, + }, + "addressType": "IPv4", + "ports": [{"name": "https", "port": 443}], + "endpoints": [ + {"addresses": ["203.0.113.7"], "conditions": {"ready": True}} + ], + } + }, + }, + } + ) + ), + "cluster-name-prod-ipv6-gateway-bbbbb": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Service", + "metadata": { + "name": "prod-ipv6-gateway-bbbbb", + "namespace": "modelplane-system", + }, + "spec": {"clusterIP": "None", "ports": [{"name": "https", "port": 443}]}, + } + }, + }, + } + ) + ), + "cluster-name-slice-prod-ipv6-gateway-bbbbb": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "discovery.k8s.io/v1", + "kind": "EndpointSlice", + "metadata": { + "name": "prod-ipv6-gateway-bbbbb", + "namespace": "modelplane-system", + "labels": {"kubernetes.io/service-name": "prod-ipv6-gateway-bbbbb"}, + }, + "addressType": "IPv6", + "ports": [{"name": "https", "port": 443}], + "endpoints": [ + {"addresses": ["2001:db8::1"], "conditions": {"ready": True}} + ], + } + }, + }, + } + ) + ), + "cluster-name-prod-dns-gateway-ccccc": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Service", + "metadata": { + "name": "prod-dns-gateway-ccccc", + "namespace": "modelplane-system", + }, + "spec": {"type": "ExternalName", "externalName": "lb-x.elb.amazonaws.com"}, + } + }, + }, + } + ) + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="Waiting for the Gateway on cluster gw-eu to be programmed" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="Waiting for the Gateway on cluster gw-eu to be programmed", + ) + ], + ), + ), + # A referenced TLS Secret doesn't exist. The Gateway is still composed, + # HTTPS listener and all, so its address survives. The listener is left + # without a certificate on the cluster until the Secret appears, rather + # than the whole Gateway withdrawn and its load balancer moved. + Case( + name="a missing TLS Secret keeps the Gateway", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=True, caller_secret_labels=None)), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "tls-secret-0": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=True, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "redirect-route": _redirect_route(), + }, + ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for TLS Secrets: eu-tls-0")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "tls-secret-0": fnv1.ResourceSelector( + api_version="v1", kind="Secret", namespace="modelplane-system", match_name="eu-tls-0" ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="SecretsMissing", + message="Waiting for TLS Secrets: eu-tls-0", + ) + ], + ), + ), + # A gateway with TLS and auth, whose Gateway has an address. A caller + # depends on the HTTPS listener, the Secrets copied to the cluster, the + # caller policy naming them, /healthz and the :80 redirect exempted from + # that policy, and a status publishing no URLs, since a caller reaches a + # TLS gateway on a DNS name only its owner knows. The certificate is copied + # verbatim, keeping the name the Gateway refers to it by. The redirect is + # exempted because it must happen before auth, or an unauthenticated caller + # would get a 401 instead of being sent to HTTPS. + Case( + name="a gateway with TLS and auth serves authenticated HTTPS and publishes no URLs", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=True, caller_secret_labels={"modelplane.ai/inference-keys": "true"}), + resources={ + "gateway": _observed_gateway(address="34.56.129.3", ready=True), + "caller-auth": _observed_caller_auth(), + }, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[_caller_key_secret(name="ml-team-keys", data={"ml-team-assistant": "c2stbXAtYTFiMmMz"})] + ), + "tls-secret-0": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "eu-tls-0", "namespace": "modelplane-system"}, + "data": {"tls.crt": "Y2VydA==", "tls.key": "a2V5"}, + } + ) + ) + ] ), }, ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - assert ( - resource.struct_to_dict(got.desired.composite.resource)["status"]["clientCACertificate"] - == "-----BEGIN CERTIFICATE-----\nclient\n" - ) - - -def test_no_client_ca_before_the_bundle_syncs() -> None: - """With no observed ConfigMap the gateway publishes no CA, so no cluster - trusts it yet and no cluster publishes a hostname on its account.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr())), - resources={"gateway": _observed_gateway("gw.example.org", ready=True)}, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={"address": "34.56.129.3"}, ready=fnv1.READY_UNSPECIFIED), + resources={ + "caller-secret-ml-team-keys": _caller_secret( + name="callers-ml-team-keys", data={"ml-team-assistant": "c2stbXAtYTFiMmMz"} + ), + "tls-secret-eu-tls-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "eu-tls-0", "namespace": "modelplane-system"}, + "type": "kubernetes.io/tls", + "data": {"tls.crt": "Y2VydA==", "tls.key": "a2V5"}, + } + }, + }, + } + ) + ), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=True, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-ml-team-keys"}], ready=fnv1.READY_TRUE + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + "redirect-route": _redirect_route(), + "redirect-auth": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "SecurityPolicy", + "metadata": { + "name": "inference-gateway-redirect-open", + "namespace": "modelplane-system", + }, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "HTTPRoute", + "name": "inference-gateway-redirect", + } + ], + "authorization": {"defaultAction": "Allow"}, + }, + } + }, + }, + } + ) + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + "tls-secret-0": fnv1.ResourceSelector( + api_version="v1", kind="Secret", namespace="modelplane-system", match_name="eu-tls-0" + ), + } + ), + conditions=[ + fnv1.Condition(type="GatewayReady", status=fnv1.STATUS_CONDITION_TRUE, reason="GatewayProgrammed"), + ], ), - required_resources=_required(clusters=[_cluster()], gateways=[_gateway_xr("eu", _CLUSTER)]), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) + ), + # A gateway whose caller policy was rejected refuses every request with a + # 500 while its Gateway still has an address. Envoy Gateway rejects the + # policy when two selected Secrets share a key value, so this is reachable + # by writing two Secrets. Here the Gateway is programmed, but the policy's + # Object hasn't been observed at all, which the function treats the same as + # a rejected policy. + Case( + name="a caller policy not yet accepted is not ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}), + resources={"gateway": _observed_gateway(address="34.56.129.3", ready=True)}, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[_caller_key_secret(name="ml-team-keys", data={"a": "c2stMQ=="})] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "34.56.129.3", + "endpoints": { + "openAI": "http://34.56.129.3/v1", + "anthropic": "http://34.56.129.3/anthropic/v1", + }, + }, + ready=fnv1.READY_FALSE, + ), + resources={ + "caller-secret-ml-team-keys": _caller_secret(name="callers-ml-team-keys", data={"a": "c2stMQ=="}), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-ml-team-keys"}], ready=fnv1.READY_UNSPECIFIED + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="The gateway's caller authentication policy has not been accepted, so every request is refused", + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="CallerAuthNotAccepted", + message="The gateway's caller authentication policy has not been accepted, so every request is refused", + ) + ], + ), + ), + # Envoy Gateway rejects the caller policy when two callers share a key, but + # the policy's Object still reads as accepted until provider-kubernetes + # next observes it. The gateway reports the outage from the Secrets + # themselves, and skips a repeated caller name before comparing its key, as + # Envoy Gateway does. These four cases observe the policy as accepted. + Case( + name="two Secrets sharing a key leave the gateway not ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}), + resources={ + "gateway": _observed_gateway(address="34.56.129.3", ready=True), + "caller-auth": _observed_caller_auth(), + }, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[ + _caller_key_secret(name="team-a-keys", data={"a": "c2stMQ=="}), + _caller_key_secret(name="team-b-keys", data={"b": "c2stMQ=="}), + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "34.56.129.3", + "endpoints": { + "openAI": "http://34.56.129.3/v1", + "anthropic": "http://34.56.129.3/anthropic/v1", + }, + }, + ready=fnv1.READY_FALSE, + ), + resources={ + "caller-secret-team-a-keys": _caller_secret(name="callers-team-a-keys", data={"a": "c2stMQ=="}), + "caller-secret-team-b-keys": _caller_secret(name="callers-team-b-keys", data={"b": "c2stMQ=="}), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-team-a-keys"}, {"name": "callers-team-b-keys"}], + ready=fnv1.READY_TRUE, + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Caller team-b-keys/b has the same key as team-a-keys/a, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="CallerAuthNotAccepted", + message="Caller team-b-keys/b has the same key as team-a-keys/a, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ) + ], + ), + ), + Case( + name="a key repeated within one Secret leaves the gateway not ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}), + resources={ + "gateway": _observed_gateway(address="34.56.129.3", ready=True), + "caller-auth": _observed_caller_auth(), + }, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[_caller_key_secret(name="team-a-keys", data={"a": "c2stMQ==", "z": "c2stMQ=="})] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "34.56.129.3", + "endpoints": { + "openAI": "http://34.56.129.3/v1", + "anthropic": "http://34.56.129.3/anthropic/v1", + }, + }, + ready=fnv1.READY_FALSE, + ), + resources={ + "caller-secret-team-a-keys": _caller_secret( + name="callers-team-a-keys", data={"a": "c2stMQ==", "z": "c2stMQ=="} + ), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-team-a-keys"}], ready=fnv1.READY_TRUE + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Caller team-a-keys/z has the same key as team-a-keys/a, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="CallerAuthNotAccepted", + message="Caller team-a-keys/z has the same key as team-a-keys/a, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ) + ], + ), + ), + # The Secrets are listed out of order. Walked in name order, team-a's x is + # seen first, so team-b's x is skipped and its y repeats x's key. Walked as + # listed, team-a's x would be the skipped one. + Case( + name="Secrets are walked in name order", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}), + resources={ + "gateway": _observed_gateway(address="34.56.129.3", ready=True), + "caller-auth": _observed_caller_auth(), + }, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[ + _caller_key_secret(name="team-b-keys", data={"x": "c2stMg==", "y": "c2stMQ=="}), + _caller_key_secret(name="team-a-keys", data={"x": "c2stMQ=="}), + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "34.56.129.3", + "endpoints": { + "openAI": "http://34.56.129.3/v1", + "anthropic": "http://34.56.129.3/anthropic/v1", + }, + }, + ready=fnv1.READY_FALSE, + ), + resources={ + "caller-secret-team-a-keys": _caller_secret(name="callers-team-a-keys", data={"x": "c2stMQ=="}), + "caller-secret-team-b-keys": _caller_secret( + name="callers-team-b-keys", data={"x": "c2stMg==", "y": "c2stMQ=="} + ), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-team-a-keys"}, {"name": "callers-team-b-keys"}], + ready=fnv1.READY_TRUE, + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Caller team-b-keys/y has the same key as team-a-keys/x, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="CallerAuthNotAccepted", + message="Caller team-b-keys/y has the same key as team-a-keys/x, so Envoy Gateway rejects the " + "caller authentication policy and every request is refused", + ) + ], + ), + ), + Case( + name="a repeated caller name is skipped, whatever its key", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}), + resources={ + "gateway": _observed_gateway(address="34.56.129.3", ready=True), + "caller-auth": _observed_caller_auth(), + }, + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[ + _caller_key_secret(name="team-a-keys", data={"a": "c2stMQ=="}), + _caller_key_secret(name="team-b-keys", data={"a": "c2stMQ==", "b": "c2stMg=="}), + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr( + status={ + "address": "34.56.129.3", + "endpoints": { + "openAI": "http://34.56.129.3/v1", + "anthropic": "http://34.56.129.3/anthropic/v1", + }, + }, + ready=fnv1.READY_UNSPECIFIED, + ), + resources={ + "caller-secret-team-a-keys": _caller_secret(name="callers-team-a-keys", data={"a": "c2stMQ=="}), + "caller-secret-team-b-keys": _caller_secret( + name="callers-team-b-keys", data={"a": "c2stMQ==", "b": "c2stMg=="} + ), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_TRUE), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-team-a-keys"}, {"name": "callers-team-b-keys"}], + ready=fnv1.READY_TRUE, + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition(type="GatewayReady", status=fnv1.STATUS_CONDITION_TRUE, reason="GatewayProgrammed"), + ], + ), + ), + # Envoy Gateway keeps the first Secret listed when two hold the same caller + # name, so the policy lists them by name rather than in the order they + # resolved in, and the winner doesn't change between reconciles. + Case( + name="caller Secrets are listed in name order", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}) + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[ + _caller_key_secret(name="team-b-keys", data={"b": "c2stMg=="}), + _caller_key_secret(name="team-a-keys", data={"a": "c2stMQ=="}), + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "caller-secret-team-a-keys": _caller_secret(name="callers-team-a-keys", data={"a": "c2stMQ=="}), + "caller-secret-team-b-keys": _caller_secret(name="callers-team-b-keys", data={"b": "c2stMg=="}), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-team-a-keys"}, {"name": "callers-team-b-keys"}], + ready=fnv1.READY_UNSPECIFIED, + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="Waiting for the Gateway on cluster gw-eu to be programmed" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="Waiting for the Gateway on cluster gw-eu to be programmed", + ) + ], + ), + ), + # Auth is asked for but no caller Secret has resolved. The Gateway is still + # composed, so its load balancer and address survive. No caller Secret is + # copied to the cluster, and the caller policy denies every request rather + # than authenticating nobody by omission. Two states reach this, the + # selector matching no Secret and the requirement not having resolved yet, + # differing only in the message reported. + Case( + name="a caller selector that matches no Secret denies but keeps the Gateway", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}) + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "SecurityPolicy", + "metadata": { + "name": "inference-gateway-callers", + "namespace": "modelplane-system", + }, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + } + ], + "authorization": {"defaultAction": "Deny"}, + }, + } + }, + }, + } + ) + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="spec.auth.apiKey.secretSelector matches no Secret, so no caller could authenticate", + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="SecretsMissing", + message="spec.auth.apiKey.secretSelector matches no Secret, so no caller could authenticate", + ) + ], + ), + ), + Case( + name="unresolved caller Secrets deny but keep the Gateway", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr(name="eu", tls=False, caller_secret_labels={"modelplane.ai/inference-keys": "true"}) + ), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=False, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "SecurityPolicy", + "metadata": { + "name": "inference-gateway-callers", + "namespace": "modelplane-system", + }, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + } + ], + "authorization": {"defaultAction": "Deny"}, + }, + } + }, + }, + } + ) + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + }, + ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for caller key Secrets to resolve")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/inference-keys": "true"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="SecretsMissing", + message="Waiting for caller key Secrets to resolve", + ) + ], + ), + ), + # No composed Object observes a Secret, which is what keeps this gateway's + # client CA private key off the control plane. provider-kubernetes copies + # an observed object's whole manifest into the Object's status, and its + # --sanitize-secrets flag defaults to false, so observing a Secret publishes + # every key in it to anyone who can get objects. This CA signs the + # certificate every cluster gateway in the fleet accepts, so leaking its + # key means anyone can reach any engine. + # + # Auth and TLS are both on, so the Secret-copying path is exercised: without + # them this function composes no Secret at all. Neither copy carries an + # Observe management policy, and the whole response covers every composed + # Object rather than only the PKI, because the cost of reintroducing this + # anywhere is the same. + # + # Observing is what matters here. The Secrets this function writes also end + # up in status, because provider-kubernetes reports what it observes of what + # it manages, so this alone doesn't keep their contents off the control + # plane. Those hold caller keys and serving certificates that came from + # control-plane Secrets to begin with, so the exposure is a wider audience + # for data already present rather than data that would otherwise never be + # there, and prerequisites.yaml runs provider-kubernetes with + # --sanitize-secrets to redact it. A CA private key is different in kind: it + # is generated on the workload cluster and observing it is the only way it + # could ever reach the control plane. + Case( + name="no composed Object observes a Secret", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_xr(name="eu", tls=True, caller_secret_labels={"team": "ml"})), + required_resources={ + "clusters": fnv1.Resources(items=[_cluster(provider_config_ref=True)]), + "gateways": fnv1.Resources(items=[_inference_gateway(name="eu", address=None)]), + "caller-secrets": fnv1.Resources( + items=[_caller_key_secret(name="ml-team-keys", data={"alice": "a2V5"})] + ), + "tls-secret-0": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "eu-tls-0", "namespace": "modelplane-system"}, + "data": {"tls.crt": "Y2VydA==", "tls.key": "a2V5"}, + } + ) + ) + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_xr(status={}, ready=fnv1.READY_FALSE), + resources={ + "caller-secret-ml-team-keys": _caller_secret(name="callers-ml-team-keys", data={"alice": "a2V5"}), + "tls-secret-eu-tls-0": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "eu-tls-0", "namespace": "modelplane-system"}, + "type": "kubernetes.io/tls", + "data": {"tls.crt": "Y2VydA==", "tls.key": "a2V5"}, + } + }, + }, + } + ) + ), + "envoy-proxy": _envoy_proxy(), + "gateway": _gateway(https=True, ready=fnv1.READY_UNSPECIFIED), + "client-selfsigned-issuer": _client_selfsigned_issuer(), + "client-ca-certificate": _client_ca_certificate(common_name="Modelplane InferenceGateway CA eu"), + "client-ca-issuer": _client_ca_issuer(), + "client-ca-bundle": _client_ca_bundle(), + "client-ca-configmap": _client_ca_configmap(), + "caller-auth": _caller_auth( + credential_refs=[{"name": "callers-ml-team-keys"}], ready=fnv1.READY_UNSPECIFIED + ), + "failover-policy": _failover_policy(), + "client-traffic-policy": _client_traffic_policy(), + "healthz-filter": _healthz_filter(), + "healthz-route": _healthz_route(), + "healthz-auth": _healthz_auth(), + "redirect-route": _redirect_route(), + "redirect-auth": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "providerConfigRef": { + "kind": "ClusterProviderConfig", + "name": "gw-eu-cluster-kubeconfig", + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.ancestors) && " + "object.status.ancestors.exists(a, has(a.conditions) && " + "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "SecurityPolicy", + "metadata": { + "name": "inference-gateway-redirect-open", + "namespace": "modelplane-system", + }, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "HTTPRoute", + "name": "inference-gateway-redirect", + } + ], + "authorization": {"defaultAction": "Allow"}, + }, + } + }, + }, + } + ) + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="Waiting for the Gateway on cluster gw-eu to be programmed" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "caller-secrets": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + namespace="modelplane-system", + match_labels=fnv1.MatchLabels(labels={"team": "ml"}), + ), + "tls-secret-0": fnv1.ResourceSelector( + api_version="v1", kind="Secret", namespace="modelplane-system", match_name="eu-tls-0" + ), + } + ), + conditions=[ + fnv1.Condition( + type="GatewayReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="Waiting for the Gateway on cluster gw-eu to be programmed", + ) + ], + ), + ), +] - assert "clientCACertificate" not in resource.struct_to_dict(got.desired.composite.resource)["status"] + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes an InferenceGateway and reports its readiness.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-model-cache/tests/test_fn.py b/functions/compose-model-cache/tests/test_fn.py index b1f607ff5..d3c749fd1 100644 --- a/functions/compose-model-cache/tests/test_fn.py +++ b/functions/compose-model-cache/tests/test_fn.py @@ -30,9 +30,6 @@ from models.ai.modelplane.modelcache import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -# A fixed transition time keeps observed conditions deterministic. -_TRANSITION_TIME = datetime.datetime(2026, 6, 8, tzinfo=datetime.UTC) - @dataclasses.dataclass class Case: @@ -43,843 +40,1146 @@ class Case: want: fnv1.RunFunctionResponse -# The XR used across cases: a HuggingFace ModelCache in the ml-team namespace. -# Both the PVC and Job derive their names from -# resource.child_name("modelcache", "ml-team", "qwen", ...). -def _cache_xr(**hf_extra: Any) -> v1alpha1.ModelCache: - return v1alpha1.ModelCache( - metadata=metav1.ObjectMeta(name="qwen", namespace="ml-team"), - spec=v1alpha1.Spec( - source="HuggingFace", - huggingFace=v1alpha1.HuggingFace(repo="Qwen/Qwen3-0.6B", sizeGiB=20, **hf_extra), +def _model_cache(*, revision: str | None, auth_secret: v1alpha1.AuthSecret | None, ready: fnv1.Ready) -> fnv1.Resource: + """The qwen ModelCache XR, with a status reporting cluster-a staged and the XR Ready only if ready is READY_TRUE.""" + status = None + if ready == fnv1.READY_TRUE: + status = v1alpha1.Status( + summary=v1alpha1.Summary(ready="1/1"), + clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], + conditions=[ + v1alpha1.Condition( + type="Ready", + status="True", + reason="Available", + lastTransitionTime=datetime.datetime(2026, 6, 8, tzinfo=datetime.UTC), + ), + ], + ) + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ModelCache( + metadata=metav1.ObjectMeta(name="qwen", namespace="ml-team"), + spec=v1alpha1.Spec( + source="HuggingFace", + huggingFace=v1alpha1.HuggingFace( + repo="Qwen/Qwen3-0.6B", + sizeGiB=20, + revision=revision, + authSecret=auth_secret, + ), + ), + status=status, + ).model_dump(exclude_none=True, mode="json", by_alias=True) ), ) -def _cluster_dict(name: str, pc: str, *, source: str = "GKE", storage_class: str | None = None) -> dict: - """An InferenceCluster as Crossplane returns it in a required-resource set. - - The cache PVC's StorageClass comes from status.cache.storageClassName, which - the InferenceCluster relays from its backing cluster. The match gate requires - both providerConfigRef AND status.cache, so every matchable fixture reports - one. Defaults to the source's effective class (GKE -> modelplane-rwx, - EKS -> modelplane-rwx-efs) unless overridden. - """ - blocks = { - "GKE": {"gke": {"project": "my-project", "region": "us-central1"}}, - "EKS": {"eks": {"region": "us-west-2"}}, - } - if storage_class is None: - storage_class = "modelplane-rwx-efs" if source == "EKS" else "modelplane-rwx" - return { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceCluster", - "metadata": {"name": name}, - "spec": {"cluster": {"source": source, **blocks[source]}}, - "status": { - "providerConfigRef": {"name": pc}, - "cache": {"storageClassName": storage_class}, - }, - } - +def _desired_model_cache(*, summary: str, clusters: list[dict], ready: fnv1.Ready) -> fnv1.Resource: + """The desired ModelCache XR, reporting summary as its ready count and each cluster's phase.""" + return fnv1.Resource( + resource=resource.dict_to_struct({"status": {"summary": {"ready": summary}, "clusters": clusters}}), + ready=ready, + ) -def _observed_object(manifest_status: dict, *, ready: bool = False) -> dict: - """A full Object envelope as provider-kubernetes observes it. - Carries the echoed remote status under status.atProvider.manifest.status - (read by derive_cluster_phase) and, when `ready`, the Object's own Ready - condition (read by mark_ready_resources, populated by DeriveFromCelQuery). - """ - status: dict[str, Any] = {"atProvider": {"manifest": {"status": manifest_status}}} - if ready: - status["conditions"] = [ +def _inference_cluster(*, name: str, provider_config: str, cluster: dict, storage_class: str) -> fnv1.Resource: + """An InferenceCluster reporting the providerConfigRef and cache StorageClass the function matches on.""" + return fnv1.Resource( + resource=resource.dict_to_struct( { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2026-06-08T00:00:00Z", - }, - ] - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": {"forProvider": {"manifest": {}}}, - "status": status, - } - - -def _auth_secret(*, data: dict[str, str] | None = None) -> dict: - """The control-plane authSecret as Crossplane returns it in a required- - resource set: a core/v1 Secret with base64 `data`.""" - return { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": "hf-token", "namespace": "ml-team"}, - "type": "Opaque", - "data": {"HF_TOKEN": _TOKEN_B64} if data is None else data, - } - - -def _req( - xr: v1alpha1.ModelCache, - clusters: list[dict], - observed: dict[str, dict] | None = None, - auth: dict | None = None, -) -> fnv1.RunFunctionRequest: - """Build a request the way the repo's other function tests do. - - - XR goes in observed.composite via dict_to_struct(model_dump(mode="json")). - - Resolved clusters go in the `clusters` required-resource set. - - `auth`, when given, is the resolved control-plane Secret in the - `auth-secret` required-resource set. - - `observed` maps a desired-resource key -> an observed Object envelope. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json")), - ), + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": name}, + "spec": {"cluster": cluster}, + "status": { + "providerConfigRef": {"name": provider_config}, + "cache": {"storageClassName": storage_class}, + }, + } ), ) - # Touch the key so a resolved-but-empty match (clusters == []) is present - # with no items, the way Crossplane returns it - distinct from an unresolved - # requirement, whose key is absent. - req.required_resources["clusters"].items.extend( - fnv1.Resource(resource=resource.dict_to_struct(c)) for c in clusters - ) - if auth is not None: - req.required_resources["auth-secret"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(auth)), - ) - for key, obj in (observed or {}).items(): - req.observed.resources[key].resource.update(obj) - return req - -# The function requires every InferenceCluster with a bare selector, so every -# response echoes this selector under requirements. -_CLUSTERS_SELECTOR = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceCluster", -) -# When the cache references an authSecret, the function requires that Secret by -# name from the XR's namespace; the response echoes this selector. -_AUTH_SELECTOR = fnv1.ResourceSelector( - api_version="v1", - kind="Secret", - match_name="hf-token", - namespace="ml-team", -) - -# The hydration shell script the Job runs (no revision, no auth secret). No -# --local-dir: HF_HUB_CACHE (below) points `hf download` at the mount, so it -# stages in HuggingFace's cache layout and a serving pod can load by repo id. -_HYDRATE_CMD = ( - "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " - "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B; " - "touch /mnt/artifact/.modelplane-hydrated" -) -# With a pinned revision (case 2 wires --revision and the HF_TOKEN env). -_HYDRATE_CMD_REVISION = ( - "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " - "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B --revision main; " - "touch /mnt/artifact/.modelplane-hydrated" -) -# Set on the hydration Job so `hf download` writes the hub cache layout into -# the mount rather than the container's default ~/.cache. -_HYDRATE_CACHE_ENV = {"name": "HF_HUB_CACHE", "value": "/mnt/artifact"} - -_PVC_NAME = "modelcache-ml-team-qwen-17db2" -_JOB_NAME = "modelcache-ml-team-qwen-hydrate-256ec" -_AUTH_NAME = "modelcache-ml-team-qwen-auth-ae01b" -_LABELS = {"modelplane.ai/modelcache": "qwen"} - -# The token data the control-plane authSecret carries, base64 as the API server -# stores it. Propagated verbatim into the workload-cluster Secret's data. -_TOKEN_B64 = "aGYtdG9rZW4tdmFsdWU=" +def _auth_secret(*, data: dict[str, str]) -> fnv1.Resource: + """The ModelCache's control-plane authSecret, with base64 data.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "hf-token", "namespace": "ml-team"}, + "type": "Opaque", + "data": data, + } + ), + ) -def _pvc_object(pc: str, *, storage_class: str = "modelplane-rwx") -> dict: - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": { - "forProvider": { - "manifest": { - "apiVersion": "v1", - "kind": "PersistentVolumeClaim", - "metadata": {"name": _PVC_NAME, "namespace": "mp-ml-team-51733", "labels": _LABELS}, - "spec": { - "accessModes": ["ReadWriteMany"], - "resources": {"requests": {"storage": "20Gi"}}, - "storageClassName": storage_class, +def _pvc_object(*, provider_config: str, storage_class: str, ready: fnv1.Ready) -> fnv1.Resource: + """The Object wrapping the qwen ModelCache's cache PVC on a workload cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "PersistentVolumeClaim", + "metadata": { + "name": "modelcache-ml-team-qwen-17db2", + "namespace": "mp-ml-team-51733", + "labels": {"modelplane.ai/modelcache": "qwen"}, + }, + "spec": { + "accessModes": ["ReadWriteMany"], + "resources": {"requests": {"storage": "20Gi"}}, + "storageClassName": storage_class, + }, + }, }, + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": provider_config}, + "readiness": {"celQuery": 'object.status.phase == "Bound"', "policy": "DeriveFromCelQuery"}, }, - }, - "providerConfigRef": {"kind": "ClusterProviderConfig", "name": pc}, - "readiness": {"celQuery": 'object.status.phase == "Bound"', "policy": "DeriveFromCelQuery"}, - }, - } + } + ), + ready=ready, + ) -def _job_object(pc: str, *, command: str = _HYDRATE_CMD, env: list | None = None) -> dict: - # HF_HUB_CACHE leads the Job's env on every source; anything the caller - # passes (an authSecret's HF_TOKEN) follows it. - env = [_HYDRATE_CACHE_ENV, *(env or [])] - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": { - "forProvider": { - "manifest": { - "apiVersion": "batch/v1", - "kind": "Job", - "metadata": {"name": _JOB_NAME, "namespace": "mp-ml-team-51733", "labels": _LABELS}, - "spec": { - "backoffLimit": 3, - "ttlSecondsAfterFinished": 180, - "template": { - "metadata": {"labels": _LABELS}, +def _job_object(*, provider_config: str, command: str, env: list[dict]) -> fnv1.Resource: + """The Object wrapping the qwen ModelCache's hydration Job, which mounts the cache PVC.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "forProvider": { + "manifest": { + "apiVersion": "batch/v1", + "kind": "Job", + "metadata": { + "name": "modelcache-ml-team-qwen-hydrate-256ec", + "namespace": "mp-ml-team-51733", + "labels": {"modelplane.ai/modelcache": "qwen"}, + }, "spec": { - "restartPolicy": "OnFailure", - "containers": [ - { - "name": "hydrate", - "image": "python:3.11-slim", - "command": ["/bin/sh", "-c", command], - "env": env, - "volumeMounts": [{"name": "artifact", "mountPath": "/mnt/artifact"}], + "backoffLimit": 3, + "ttlSecondsAfterFinished": 180, + "template": { + "metadata": {"labels": {"modelplane.ai/modelcache": "qwen"}}, + "spec": { + "restartPolicy": "OnFailure", + "containers": [ + { + "name": "hydrate", + "image": "python:3.11-slim", + "command": ["/bin/sh", "-c", command], + "env": env, + "volumeMounts": [{"name": "artifact", "mountPath": "/mnt/artifact"}], + }, + ], + "volumes": [ + { + "name": "artifact", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}, + }, + ], }, - ], - "volumes": [ - {"name": "artifact", "persistentVolumeClaim": {"claimName": _PVC_NAME}}, - ], + }, }, }, }, + "managementPolicies": ["Observe", "Create", "Update", "LateInitialize"], + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": provider_config}, + "readiness": { + "celQuery": 'object.status.conditions.exists(c, c.type == "Complete" && c.status == "True")', + "policy": "DeriveFromCelQuery", + }, }, - }, - "managementPolicies": ["Observe", "Create", "Update", "LateInitialize"], - "providerConfigRef": {"kind": "ClusterProviderConfig", "name": pc}, - "readiness": { - "celQuery": 'object.status.conditions.exists(c, c.type == "Complete" && c.status == "True")', - "policy": "DeriveFromCelQuery", - }, - }, - } + } + ), + ) -def _auth_object(pc: str) -> dict: - """The workload-cluster Secret Object propagating the HF token. No readiness - block: a Secret has no status, so the Object uses default readiness.""" - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": { - "forProvider": { - "manifest": { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": _AUTH_NAME, "namespace": "mp-ml-team-51733", "labels": _LABELS}, - "data": {"HF_TOKEN": _TOKEN_B64}, +def _observed_pvc_object() -> fnv1.Resource: + """The cache PVC's Object as observed, with the PVC Bound and the Object Ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": {"forProvider": {"manifest": {}}}, + "status": { + "atProvider": {"manifest": {"status": {"phase": "Bound"}}}, + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + }, + ], }, - }, - "providerConfigRef": {"kind": "ClusterProviderConfig", "name": pc}, - }, + } + ), + ) + + +def _observed_job_object(*, condition: str, ready: fnv1.Ready) -> fnv1.Resource: + """The hydration Job's Object as observed, the Job reporting condition, and Ready only if ready is READY_TRUE.""" + status: dict[str, Any] = { + "atProvider": {"manifest": {"status": {"conditions": [{"type": condition, "status": "True"}]}}}, } + if ready == fnv1.READY_TRUE: + status["conditions"] = [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + }, + ] + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": {"forProvider": {"manifest": {}}}, + "status": status, + } + ), + ) -def _compose_cases() -> list[Case]: - """The cases for test_compose, built by code that derives them from shared parts.""" - # --- Case 1: GKE cluster, first pass. Composes the RWX PVC + hydration - # Job per matched cluster; nothing observed yet so phase is Pending and - # ArtifactReady is Hydrating. Emits the one-time "Staging" event. --- - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Pending"}], - }, - }, +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +# Every case is a HuggingFace ModelCache named qwen in the ml-team namespace. The +# function requires every InferenceCluster with a bare selector, so every +# response carries that requirement. +COMPOSE_CASES = [ + # Nothing is observed yet, so the one-time Staging event fires. The Job sets + # HF_HUB_CACHE rather than passing --local-dir, so `hf download` writes + # HuggingFace's cache layout to the mount and a serving pod can load the + # model by repo id. + Case( + name="GKE cluster first pass composes RWX PVC and hydration Job", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache(revision=None, auth_secret=None, ready=fnv1.READY_UNSPECIFIED), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _inference_cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="0/1", + clusters=[{"name": "cluster-a", "phase": "Pending"}], + ready=fnv1.READY_UNSPECIFIED, ), + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_UNSPECIFIED, + ), + "hydrate-cluster-a": _job_object( + provider_config="cluster-a-pc", + command=( + "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " + "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B; " + "touch /mnt/artifact/.modelplane-hydrated" + ), + env=[{"name": "HF_HUB_CACHE", "value": "/mnt/artifact"}], + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), - "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + ), + ), + # The function copies the token's base64 verbatim to a workload-cluster + # Secret, and the Job's HF_TOKEN references that Secret, because the + # control-plane Secret isn't on the workload cluster. A Secret has no + # status, so its Object has no readiness block and uses default readiness. + Case( + name="HuggingFace revision and auth secret wire --revision and HF_TOKEN", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision="main", + auth_secret=v1alpha1.AuthSecret(name="hf-token"), + ready=fnv1.READY_UNSPECIFIED, + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _inference_cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], + ), + "auth-secret": fnv1.Resources(items=[_auth_secret(data={"HF_TOKEN": "aGYtdG9rZW4tdmFsdWU="})]), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - want1.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 2: GKE cluster with a pinned revision + auth secret. The Job - # command gains --revision, and the function propagates the token to a - # workload-cluster Secret (auth-cluster-a) whose name the Job's HF_TOKEN - # env references - not the user's control-plane Secret name. --- - xr2 = _cache_xr(revision="main", authSecret=v1alpha1.AuthSecret(name="hf-token")) - env2 = [ - { - "name": "HF_TOKEN", - "valueFrom": {"secretKeyRef": {"name": _AUTH_NAME, "key": "HF_TOKEN"}}, - }, - ] - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Pending"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="0/1", + clusters=[{"name": "cluster-a", "phase": "Pending"}], + ready=fnv1.READY_UNSPECIFIED, ), + resources={ + "auth-cluster-a": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Secret", + "metadata": { + "name": "modelcache-ml-team-qwen-auth-ae01b", + "namespace": "mp-ml-team-51733", + "labels": {"modelplane.ai/modelcache": "qwen"}, + }, + "data": {"HF_TOKEN": "aGYtdG9rZW4tdmFsdWU="}, + }, + }, + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + }, + } + ), + ), + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_UNSPECIFIED, + ), + "hydrate-cluster-a": _job_object( + provider_config="cluster-a-pc", + command=( + "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " + "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B --revision main; " + "touch /mnt/artifact/.modelplane-hydrated" + ), + env=[ + {"name": "HF_HUB_CACHE", "value": "/mnt/artifact"}, + { + "name": "HF_TOKEN", + "valueFrom": { + "secretKeyRef": {"name": "modelcache-ml-team-qwen-auth-ae01b", "key": "HF_TOKEN"}, + }, + }, + ], + ), + }, ), - resources={ - "auth-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_auth_object("cluster-a-pc"))), - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), - "hydrate-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct( - _job_object("cluster-a-pc", command=_HYDRATE_CMD_REVISION, env=env2), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "auth-secret": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="hf-token", + namespace="ml-team", ), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + ), + ), + Case( + name="EKS cluster PVC sources the EFS class from status.cache", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache(revision=None, auth_secret=None, ready=fnv1.READY_UNSPECIFIED), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _inference_cluster( + name="eks-a", + provider_config="eks-a-pc", + cluster={"source": "EKS", "eks": {"region": "us-west-2"}}, + storage_class="modelplane-rwx-efs", + ), + ], ), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - want2.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want2.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 3: EKS cluster reporting an EFS RWX class on status.cache. The - # PVC sources its storageClassName from status.cache (modelplane-rwx-efs), - # not the GKE/Filestore one. --- - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "eks-a", "phase": "Pending"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="0/1", + clusters=[{"name": "eks-a", "phase": "Pending"}], + ready=fnv1.READY_UNSPECIFIED, + ), + resources={ + "pvc-eks-a": _pvc_object( + provider_config="eks-a-pc", + storage_class="modelplane-rwx-efs", + ready=fnv1.READY_UNSPECIFIED, + ), + "hydrate-eks-a": _job_object( + provider_config="eks-a-pc", + command=( + "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " + "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B; " + "touch /mnt/artifact/.modelplane-hydrated" + ), + env=[{"name": "HF_HUB_CACHE", "value": "/mnt/artifact"}], + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: eks-a", ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, ), - resources={ - "pvc-eks-a": fnv1.Resource( - resource=resource.dict_to_struct( - _pvc_object("eks-a-pc", storage_class="modelplane-rwx-efs"), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + ), + ), + # The observed PVC suppresses the Staging event, and the XR becoming Ready + # emits the staged one. Once the cluster is Ready its Job is dropped, so only + # the PVC is composed. + Case( + name="PVC bound and Job complete reports Ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache(revision=None, auth_secret=None, ready=fnv1.READY_UNSPECIFIED), + resources={ + "pvc-cluster-a": _observed_pvc_object(), + "hydrate-cluster-a": _observed_job_object(condition="Complete", ready=fnv1.READY_TRUE), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _inference_cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="1/1", + clusters=[{"name": "cluster-a", "phase": "Ready"}], + ready=fnv1.READY_TRUE, + ), + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_TRUE, ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + ), + ), + # The observed PVC suppresses the Staging event. + Case( + name="PVC bound with no Job observed reports Hydrating", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache(revision=None, auth_secret=None, ready=fnv1.READY_UNSPECIFIED), + resources={ + "pvc-cluster-a": _observed_pvc_object(), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _inference_cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], ), - "hydrate-eks-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("eks-a-pc"))), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: eks-a", - ), - ], - context=structpb.Struct(), - ) - want3.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 4: ready. The observed PVC + Job Objects each carry their own - # Ready condition (from DeriveFromCelQuery), and the wrapped manifest - # status shows PVC Bound + Job succeeded. Phase Ready, both Objects - # marked ready, summary 1/1, XR ready, ArtifactReady Staged. The - # already-composed PVC suppresses the "Staging" event; the - # not-previously-ready -> ready transition emits the "staged" event. --- - observed4 = { - "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), - } - pvc_ready = _pvc_object("cluster-a-pc") - want4 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="0/1", + clusters=[{"name": "cluster-a", "phase": "Hydrating"}], + ready=fnv1.READY_UNSPECIFIED, ), - ready=fnv1.READY_TRUE, + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_TRUE, + ), + "hydrate-cluster-a": _job_object( + provider_config="cluster-a-pc", + command=( + "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " + "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B; " + "touch /mnt/artifact/.modelplane-hydrated" + ), + env=[{"name": "HF_HUB_CACHE", "value": "/mnt/artifact"}], + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, ), - resources={ - # Job dropped once Ready; only the PVC remains composed. - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(pvc_ready), ready=fnv1.READY_TRUE), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), + ], + ), + ), + Case( + name="failed Job reports Failed and takes precedence over PVC binding", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache(revision=None, auth_secret=None, ready=fnv1.READY_UNSPECIFIED), + resources={ + "pvc-cluster-a": _observed_pvc_object(), + "hydrate-cluster-a": _observed_job_object(condition="Failed", ready=fnv1.READY_UNSPECIFIED), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _inference_cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], + ), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - results=[ - fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), - ], - context=structpb.Struct(), - ) - want4.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 5: hydrating. PVC Bound (Object Ready) but the Job hasn't - # completed, so phase is Hydrating, only the PVC is marked ready, summary - # 0/1, and the XR is not ready. No transition event fires. --- - observed5 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} - want5 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Hydrating"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="0/1", + clusters=[{"name": "cluster-a", "phase": "Failed"}], + ready=fnv1.READY_UNSPECIFIED, ), + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_TRUE, + ), + "hydrate-cluster-a": _job_object( + provider_config="cluster-a-pc", + command=( + "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " + "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B; " + "touch /mnt/artifact/.modelplane-hydrated" + ), + env=[{"name": "HF_HUB_CACHE", "value": "/mnt/artifact"}], + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Failed"), + ], + ), + ), + # Cluster a is Ready, so its Job is dropped, while b, with only its PVC + # observed, is still Hydrating. + Case( + name="one of two clusters ready reports partial", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache(revision=None, auth_secret=None, ready=fnv1.READY_UNSPECIFIED), + resources={ + "pvc-a": _observed_pvc_object(), + "hydrate-a": _observed_job_object(condition="Complete", ready=fnv1.READY_TRUE), + "pvc-b": _observed_pvc_object(), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _inference_cluster( + name="a", + provider_config="a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + _inference_cluster( + name="b", + provider_config="b-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], ), - "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Hydrating"), - ], - context=structpb.Struct(), - ) - want5.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 6: failed. The Job reports a Failed condition (and is NOT - # Ready). A Failed Job takes precedence over PVC binding, so phase is - # Failed, only the PVC is marked ready, summary 0/1, XR not ready, and - # ArtifactReady is False with reason Failed. --- - observed6 = { - "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Failed", "status": "True"}]}), - } - want6 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Failed"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="1/2", + clusters=[{"name": "a", "phase": "Ready"}, {"name": "b", "phase": "Hydrating"}], + ready=fnv1.READY_UNSPECIFIED, ), + resources={ + "pvc-a": _pvc_object(provider_config="a-pc", storage_class="modelplane-rwx", ready=fnv1.READY_TRUE), + "pvc-b": _pvc_object(provider_config="b-pc", storage_class="modelplane-rwx", ready=fnv1.READY_TRUE), + "hydrate-b": _job_object( + provider_config="b-pc", + command=( + "set -e; if [ -f /mnt/artifact/.modelplane-hydrated ]; then echo 'already hydrated, skipping'; exit 0; fi; " + "pip install --quiet huggingface_hub; hf download Qwen/Qwen3-0.6B; " + "touch /mnt/artifact/.modelplane-hydrated" + ), + env=[{"name": "HF_HUB_CACHE", "value": "/mnt/artifact"}], + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Partial"), + ], + ), + ), + # The XR's status already reports the cluster Ready, and only its PVC is + # observed, as after the TTL controller cleans up the Job. The status latch + # keeps the cluster Ready, so the Job isn't composed again, and the XR was + # already Ready, so no event fires. + Case( + name="hydrated cluster stays Ready after its Job is TTL-cleaned", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache(revision=None, auth_secret=None, ready=fnv1.READY_TRUE), + resources={ + "pvc-cluster-a": _observed_pvc_object(), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _inference_cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], ), - "hydrate-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_job_object("cluster-a-pc"))), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Failed"), - ], - context=structpb.Struct(), - ) - want6.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 7: partial (1/2). Cluster a is Ready (PVC Bound + Job - # succeeded), cluster b is still Hydrating (PVC Bound only). Summary - # 1/2, ArtifactReady False with reason Partial, XR not ready. --- - observed7 = { - "pvc-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), - "pvc-b": _observed_object({"phase": "Bound"}, ready=True), - } - want7 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/2"}, - "clusters": [ - {"name": "a", "phase": "Ready"}, - {"name": "b", "phase": "Hydrating"}, - ], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="1/1", + clusters=[{"name": "cluster-a", "phase": "Ready"}], + ready=fnv1.READY_TRUE, + ), + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + ), + ), + # The function waits for Crossplane to resolve the authSecret before it + # composes anything. + Case( + name="authSecret unresolved requires it and returns early", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision=None, + auth_secret=v1alpha1.AuthSecret(name="hf-token"), + ready=fnv1.READY_UNSPECIFIED, ), ), - resources={ - # Cluster a is Ready, so its Job is dropped; b is still Hydrating. - "pvc-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("a-pc")), ready=fnv1.READY_TRUE), - "pvc-b": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("b-pc")), ready=fnv1.READY_TRUE), - "hydrate-b": fnv1.Resource(resource=resource.dict_to_struct(_job_object("b-pc"))), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _inference_cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], + ), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="Partial"), - ], - context=structpb.Struct(), - ) - want7.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 8: latch. A previously-Ready cluster whose hydration Job was - # dropped (and TTL-cleaned), so only the PVC is observed now. The status - # latch keeps phase Ready and the Job is not re-composed, PVC marked - # ready, summary 1/1, XR ready. Already-ready, so no transition event. --- - xr_ready = _cache_xr() - xr_ready.status = v1alpha1.Status( - summary=v1alpha1.Summary(ready="1/1"), - clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], - conditions=[ - v1alpha1.Condition( - type="Ready", - status="True", - reason="Available", - lastTransitionTime=_TRANSITION_TIME, - ), - ], - ) - observed8 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} - want8 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "auth-secret": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="hf-token", + namespace="ml-team", + ), + }, + ), + ), + ), + # The Secret carries OTHER rather than HF_TOKEN. The PVC doesn't need the + # token, so it still composes and a cache isn't pruned for a missing one, but + # the Job and token Secret are held back. The XR is marked not ready, since + # the PVC alone would make it ready once it binds. + Case( + name="authSecret resolved without the referenced key composes PVC and warns", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision=None, + auth_secret=v1alpha1.AuthSecret(name="hf-token"), + ready=fnv1.READY_UNSPECIFIED, ), - ready=fnv1.READY_TRUE, ), - resources={ - # Latched Ready with the Job already dropped: only the PVC. - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE + required_resources={ + "clusters": fnv1.Resources( + items=[ + _inference_cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], ), + "auth-secret": fnv1.Resources(items=[_auth_secret(data={"OTHER": "aGYtdG9rZW4tdmFsdWU="})]), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - context=structpb.Struct(), - ) - want8.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - - # --- Case 9: authSecret referenced but not yet resolved. The function - # requires both the clusters and the auth Secret, then returns early - # (no resources, status, or conditions) until Crossplane resolves the - # Secret and re-calls it. --- - xr9 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - want9 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(), - context=structpb.Struct(), - ) - want9.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want9.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 10: authSecret resolved but the Secret lacks the referenced - # key (here it carries OTHER, not HF_TOKEN). The PVC still composes - it - # doesn't depend on the token, so a cache isn't pruned for a missing one - # - but the hydration Job and token Secret are held back. ArtifactReady - # is False with reason AuthSecretMissing, and a warning names the Secret - # and key so the user can fix it instead of seeing the XR stall. The XR - # is marked not ready, since the PVC alone would make it ready once it - # binds. --- - want10 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "0/1"}, - "clusters": [{"name": "cluster-a", "phase": "Pending"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="0/1", + clusters=[{"name": "cluster-a", "phase": "Pending"}], + ready=fnv1.READY_FALSE, + ), + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="authSecret ml-team/hf-token is missing or has no key 'HF_TOKEN'", + ), + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", ), - ready=fnv1.READY_FALSE, + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "auth-secret": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="hf-token", + namespace="ml-team", + ), + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource(resource=resource.dict_to_struct(_pvc_object("cluster-a-pc"))), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="AuthSecretMissing"), + ], + ), + ), + # An empty token value is as broken as a missing key: the Job would run with + # an empty HF_TOKEN. + Case( + name="authSecret resolved with an empty token value composes PVC and warns", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision=None, + auth_secret=v1alpha1.AuthSecret(name="hf-token"), + ready=fnv1.READY_UNSPECIFIED, + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _inference_cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], + ), + "auth-secret": fnv1.Resources(items=[_auth_secret(data={"HF_TOKEN": ""})]), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="AuthSecretMissing"), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_WARNING, - message="authSecret ml-team/hf-token is missing or has no key 'HF_TOKEN'", - ), - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) - want10.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want10.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 11: Ready cluster with an authSecret. The token is only needed - # while hydrating, so once the cluster is Ready the auth Secret is dropped - # alongside the Job (only the PVC remains composed), even though the - # control-plane Secret still resolves. Keeps the token from lingering on - # the inference cluster after hydration. --- - xr11 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - observed11 = { - "auth-cluster-a": _observed_object({}, ready=True), - "pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True), - "hydrate-cluster-a": _observed_object({"conditions": [{"type": "Complete", "status": "True"}]}, ready=True), - } - want11 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="0/1", + clusters=[{"name": "cluster-a", "phase": "Pending"}], + ready=fnv1.READY_FALSE, ), - ready=fnv1.READY_TRUE, + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_UNSPECIFIED, + ), + }, ), - resources={ - # Ready: the auth Secret and Job are both dropped, only the PVC remains. - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="authSecret ml-team/hf-token is missing or has no key 'HF_TOKEN'", + ), + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Staging Qwen/Qwen3-0.6B to 1 clusters: cluster-a", ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "auth-secret": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="hf-token", + namespace="ml-team", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="AuthSecretMissing"), + ], + ), + ), + # The token is only needed while hydrating, so once the cluster is Ready the + # auth Secret is dropped with the Job, even though the control-plane Secret + # still resolves. That keeps the token off the inference cluster after + # hydration. + Case( + name="Ready cluster drops the auth Secret with the Job", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision=None, + auth_secret=v1alpha1.AuthSecret(name="hf-token"), + ready=fnv1.READY_UNSPECIFIED, + ), + resources={ + "auth-cluster-a": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": {"forProvider": {"manifest": {}}}, + "status": { + "atProvider": {"manifest": {"status": {}}}, + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + }, + ], + }, + } + ), + ), + "pvc-cluster-a": _observed_pvc_object(), + "hydrate-cluster-a": _observed_job_object(condition="Complete", ready=fnv1.READY_TRUE), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _inference_cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], + ), + "auth-secret": fnv1.Resources(items=[_auth_secret(data={"HF_TOKEN": "aGYtdG9rZW4tdmFsdWU="})]), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - results=[ - fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), - ], - context=structpb.Struct(), - ) - want11.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want11.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 12: token rotated away after a cache is Ready. A latched-Ready - # cluster whose authSecret now resolves without the key. The PVC keeps - # composing (and stays Ready via the status latch) rather than being - # pruned, and because hydration is already done the missing token is - # neither reported (ArtifactReady stays Staged) nor warned. --- - xr12 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - xr12.status = v1alpha1.Status( - summary=v1alpha1.Summary(ready="1/1"), - clusters=[v1alpha1.Cluster(name="cluster-a", phase="Ready")], - conditions=[ - v1alpha1.Condition( - type="Ready", - status="True", - reason="Available", - lastTransitionTime=_TRANSITION_TIME, - ), - ], - ) - observed12 = {"pvc-cluster-a": _observed_object({"phase": "Bound"}, ready=True)} - want12 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "summary": {"ready": "1/1"}, - "clusters": [{"name": "cluster-a", "phase": "Ready"}], - }, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="1/1", + clusters=[{"name": "cluster-a", "phase": "Ready"}], + ready=fnv1.READY_TRUE, + ), + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_TRUE, + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Artifact staged on all 1 clusters"), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "auth-secret": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="hf-token", + namespace="ml-team", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + ), + ), + # The XR's status already reports the cluster Ready, and its authSecret now + # lacks the key. The PVC doesn't need the token, so it isn't pruned, and + # hydration is done, so the missing token is neither reported nor warned. + Case( + name="token rotated away after Ready keeps the PVC and stays Ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision=None, + auth_secret=v1alpha1.AuthSecret(name="hf-token"), + ready=fnv1.READY_TRUE, ), - ready=fnv1.READY_TRUE, + resources={ + "pvc-cluster-a": _observed_pvc_object(), + }, ), - resources={ - "pvc-cluster-a": fnv1.Resource( - resource=resource.dict_to_struct(_pvc_object("cluster-a-pc")), ready=fnv1.READY_TRUE + required_resources={ + "clusters": fnv1.Resources( + items=[ + _inference_cluster( + name="cluster-a", + provider_config="cluster-a-pc", + cluster={"source": "GKE", "gke": {"project": "my-project", "region": "us-central1"}}, + storage_class="modelplane-rwx", + ), + ], ), + "auth-secret": fnv1.Resources(items=[_auth_secret(data={"OTHER": "aGYtdG9rZW4tdmFsdWU="})]), }, ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), - ], - context=structpb.Struct(), - ) - want12.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want12.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - # --- Case 13: authSecret missing AND no clusters matched. NoClusters is - # the dominant signal - the cache can't progress regardless of the token - # - so both conditions report NoClusters and the missing token is neither - # reported nor warned. --- - xr13 = _cache_xr(authSecret=v1alpha1.AuthSecret(name="hf-token")) - want13 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"summary": {"ready": "0/0"}, "clusters": []}}), - ready=fnv1.READY_FALSE, - ), - ), - conditions=[ - fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), - fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), - ], - context=structpb.Struct(), - ) - want13.requirements.resources["clusters"].CopyFrom(_CLUSTERS_SELECTOR) - want13.requirements.resources["auth-secret"].CopyFrom(_AUTH_SELECTOR) - - return [ - Case( - name="GKE cluster first pass composes RWX PVC and hydration Job", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")]), - want=want1, - ), - Case( - name="HuggingFace revision and auth secret wire --revision and HF_TOKEN", - req=_req(xr2, [_cluster_dict("cluster-a", "cluster-a-pc")], auth=_auth_secret()), - want=want2, - ), - Case( - name="EKS cluster PVC sources the EFS class from status.cache", - req=_req(_cache_xr(), [_cluster_dict("eks-a", "eks-a-pc", source="EKS")]), - want=want3, - ), - Case( - name="PVC bound and Job complete reports Ready", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed4), - want=want4, - ), - Case( - name="PVC bound but Job running reports Hydrating", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed5), - want=want5, - ), - Case( - name="failed Job reports Failed and takes precedence over PVC binding", - req=_req(_cache_xr(), [_cluster_dict("cluster-a", "cluster-a-pc")], observed6), - want=want6, - ), - Case( - name="one of two clusters ready reports partial", - req=_req(_cache_xr(), [_cluster_dict("a", "a-pc"), _cluster_dict("b", "b-pc")], observed7), - want=want7, - ), - Case( - name="hydrated cluster stays Ready after its Job is TTL-cleaned", - req=_req(xr_ready, [_cluster_dict("cluster-a", "cluster-a-pc")], observed8), - want=want8, - ), - Case( - name="authSecret unresolved requires it and returns early", - req=_req(xr9, [_cluster_dict("cluster-a", "cluster-a-pc")]), - want=want9, - ), - Case( - name="authSecret resolved without the referenced key composes PVC and warns", - req=_req( - xr9, - [_cluster_dict("cluster-a", "cluster-a-pc")], - auth=_auth_secret(data={"OTHER": _TOKEN_B64}), - ), - want=want10, - ), - Case( - name="authSecret resolved with an empty token value composes PVC and warns", - req=_req( - xr9, - [_cluster_dict("cluster-a", "cluster-a-pc")], - auth=_auth_secret(data={"HF_TOKEN": ""}), - ), - want=want10, - ), - Case( - name="Ready cluster drops the auth Secret with the Job", - req=_req( - xr11, - [_cluster_dict("cluster-a", "cluster-a-pc")], - observed11, - auth=_auth_secret(), - ), - want=want11, - ), - Case( - name="token rotated away after Ready keeps the PVC and stays Ready", - req=_req( - xr12, - [_cluster_dict("cluster-a", "cluster-a-pc")], - observed12, - auth=_auth_secret(data={"OTHER": _TOKEN_B64}), - ), - want=want12, - ), - Case( - name="authSecret missing with no clusters reports NoClusters not AuthSecretMissing", - req=_req(xr13, [], auth=_auth_secret(data={"OTHER": _TOKEN_B64})), - want=want13, - ), - ] - - -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache( + summary="1/1", + clusters=[{"name": "cluster-a", "phase": "Ready"}], + ready=fnv1.READY_TRUE, + ), + resources={ + "pvc-cluster-a": _pvc_object( + provider_config="cluster-a-pc", + storage_class="modelplane-rwx", + ready=fnv1.READY_TRUE, + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "auth-secret": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="hf-token", + namespace="ml-team", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_TRUE, reason="Matched"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Staged"), + ], + ), + ), + # Crossplane reports a selector that matches nothing as a present but empty + # entry, unlike an unresolved one, whose key is absent. NoClusters dominates, + # since the cache can't progress whatever the token, so the missing token is + # neither reported nor warned. + Case( + name="authSecret without the key and no clusters reports NoClusters not AuthSecretMissing", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_cache( + revision=None, + auth_secret=v1alpha1.AuthSecret(name="hf-token"), + ready=fnv1.READY_UNSPECIFIED, + ), + ), + required_resources={ + "clusters": fnv1.Resources(), + "auth-secret": fnv1.Resources(items=[_auth_secret(data={"OTHER": "aGYtdG9rZW4tdmFsdWU="})]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_cache(summary="0/0", clusters=[], ready=fnv1.READY_FALSE), + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "auth-secret": fnv1.ResourceSelector( + api_version="v1", + kind="Secret", + match_name="hf-token", + namespace="ml-team", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ClustersMatched", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), + fnv1.Condition(type="ArtifactReady", status=fnv1.STATUS_CONDITION_FALSE, reason="NoClusters"), + ], + ), + ), +] -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction composes a ModelCache's PVC and hydration Job per cluster and reports their progress.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) diff --git a/functions/compose-model-deployment/tests/test_cel.py b/functions/compose-model-deployment/tests/test_cel.py index 7a2e3fd9b..7af975397 100644 --- a/functions/compose-model-deployment/tests/test_cel.py +++ b/functions/compose-model-deployment/tests/test_cel.py @@ -26,183 +26,269 @@ from function import cel -def _device( - driver: str = "gpu.nvidia.com", - attributes: dict | None = None, - capacity: dict | None = None, - **extra: object, -) -> dict: - """A pool device in the raw dict shape cel.Program.matches expects.""" - return { - "driver": driver, - "attributes": attributes or {}, - "capacity": capacity or {}, - **extra, - } - - @dataclasses.dataclass class Case: + """A test case for matching a DRA CEL selector against a device.""" + name: str expr: str device: dict want: bool -# Reusable device fixtures. -_GPU = _device( - driver="gpu.nvidia.com", - attributes={ - "architecture": {"string": "Hopper"}, - "cudaComputeCapability": {"version": "9.5.3"}, - # A qualified name lands under its own domain, not the driver's. - "resource.kubernetes.io/pcieRoot": {"string": "pci0"}, - }, - capacity={"memory": {"value": "141Gi"}}, -) -_NIC = _device(driver="nic.nvidia.com", attributes={"linkType": {"string": "infiniband"}}) +def _hopper_gpu() -> dict: + """A gpu.nvidia.com Hopper GPU with 141Gi of memory, on PCIe root pci0.""" + return { + "driver": "gpu.nvidia.com", + "attributes": { + "architecture": {"string": "Hopper"}, + "cudaComputeCapability": {"version": "9.5.3"}, + "resource.kubernetes.io/pcieRoot": {"string": "pci0"}, + }, + "capacity": {"memory": {"value": "141Gi"}}, + } + -_ATTR = 'device.attributes["gpu.nvidia.com"]' -_CAP = 'device.capacity["gpu.nvidia.com"]' +def _scalar_device(*, x: dict) -> dict: + """A gpu.nvidia.com device whose one attribute, x, is the given typed scalar.""" + return {"driver": "gpu.nvidia.com", "attributes": {"x": x}, "capacity": {}} -# Device fixtures for the verbatim selector examples in the DRA docs. -# (k8s.io/docs concept page and the allocate-devices-dra task page.) -_LARGE_BLACK = _device( - driver="resource-driver.example.com", - attributes={"color": {"string": "black"}, "size": {"string": "large"}}, -) -_SMALL_WHITE = _device( - driver="resource-driver.example.com", - attributes={"color": {"string": "white"}, "size": {"string": "small"}}, -) -_EXAMPLE_GPU = _device( - driver="gpu.example.com", - attributes={"type": {"string": "gpu"}}, -) -_GPU_64GI = _device( - driver="driver.example.com", - attributes={"type": {"string": "gpu"}}, - capacity={"memory": {"value": "64Gi"}}, -) + +def _example_device(*, color: str, size: str) -> dict: + """A resource-driver.example.com device, as in the DRA docs' examples.""" + return { + "driver": "resource-driver.example.com", + "attributes": {"color": {"string": color}, "size": {"string": size}}, + "capacity": {}, + } MATCHES_CASES = [ # driver. - Case(name="driver equals", expr='device.driver == "gpu.nvidia.com"', device=_GPU, want=True), - Case(name="driver not equals", expr='device.driver == "nic.nvidia.com"', device=_GPU, want=False), + Case( + name="driver equals", + expr='device.driver == "gpu.nvidia.com"', + device=_hopper_gpu(), + want=True, + ), + Case( + name="driver not equals", + expr='device.driver == "nic.nvidia.com"', + device=_hopper_gpu(), + want=False, + ), # Quantity comparison + methods. Case( name="quantity compareTo ge", - expr=f'{_CAP}.memory.compareTo(quantity("141Gi")) >= 0', - device=_GPU, + expr='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("141Gi")) >= 0', + device=_hopper_gpu(), want=True, ), Case( name="quantity compareTo too big", - expr=f'{_CAP}.memory.compareTo(quantity("200Gi")) >= 0', - device=_GPU, + expr='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("200Gi")) >= 0', + device=_hopper_gpu(), want=False, ), Case( name="quantity isGreaterThan", - expr=f'{_CAP}.memory.isGreaterThan(quantity("80Gi"))', - device=_GPU, + expr='device.capacity["gpu.nvidia.com"].memory.isGreaterThan(quantity("80Gi"))', + device=_hopper_gpu(), + want=True, + ), + Case( + name="quantity isLessThan", + expr='device.capacity["gpu.nvidia.com"].memory.isLessThan(quantity("200Gi"))', + device=_hopper_gpu(), + want=True, + ), + # Upstream sign is global-only, so q.sign() is a compile error there. This + # pins the member form we accept anyway, a documented divergence in cel.py + # that only makes us more permissive. + Case( + name="quantity sign", + expr='device.capacity["gpu.nvidia.com"].memory.sign() == 1', + device=_hopper_gpu(), + want=True, + ), + Case( + name="quantity asInteger", + expr='device.capacity["gpu.nvidia.com"].memory.asInteger() == 151397597184', + device=_hopper_gpu(), + want=True, + ), + Case( + name="quantity isInteger", + expr='device.capacity["gpu.nvidia.com"].memory.isInteger()', + device=_hopper_gpu(), want=True, ), - Case(name="quantity isLessThan", expr=f'{_CAP}.memory.isLessThan(quantity("200Gi"))', device=_GPU, want=True), - Case(name="quantity sign", expr=f"{_CAP}.memory.sign() == 1", device=_GPU, want=True), - Case(name="quantity asInteger", expr=f"{_CAP}.memory.asInteger() == {141 * 2**30}", device=_GPU, want=True), - Case(name="quantity isInteger", expr=f"{_CAP}.memory.isInteger()", device=_GPU, want=True), Case( name="quantity add", - expr=f'{_CAP}.memory.add(quantity("1Gi")).compareTo(quantity("142Gi")) == 0', - device=_GPU, + expr='device.capacity["gpu.nvidia.com"].memory.add(quantity("1Gi")).compareTo(quantity("142Gi")) == 0', + device=_hopper_gpu(), want=True, ), - Case(name="isQuantity true", expr='isQuantity("1.3Gi")', device=_GPU, want=True), - Case(name="isQuantity false", expr='isQuantity("200K")', device=_GPU, want=False), + Case( + name="isQuantity true", + expr='isQuantity("1.3Gi")', + device=_hopper_gpu(), + want=True, + ), + Case( + name="isQuantity false", + expr='isQuantity("200K")', + device=_hopper_gpu(), + want=False, + ), # Semver comparison + methods. Case( name="semver isGreaterThan", - expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.0.0"))', - device=_GPU, + expr='device.attributes["gpu.nvidia.com"].cudaComputeCapability.isGreaterThan(semver("9.0.0"))', + device=_hopper_gpu(), want=True, ), Case( name="semver not greater", - expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.9.0"))', - device=_GPU, + expr='device.attributes["gpu.nvidia.com"].cudaComputeCapability.isGreaterThan(semver("9.9.0"))', + device=_hopper_gpu(), + want=False, + ), + Case( + name="semver major", + expr='device.attributes["gpu.nvidia.com"].cudaComputeCapability.major() == 9', + device=_hopper_gpu(), + want=True, + ), + Case( + name="semver minor", + expr='device.attributes["gpu.nvidia.com"].cudaComputeCapability.minor() == 5', + device=_hopper_gpu(), + want=True, + ), + Case( + name="semver patch", + expr='device.attributes["gpu.nvidia.com"].cudaComputeCapability.patch() == 3', + device=_hopper_gpu(), + want=True, + ), + Case( + name="semver equality", + expr='device.attributes["gpu.nvidia.com"].cudaComputeCapability == semver("9.5.3")', + device=_hopper_gpu(), + want=True, + ), + Case( + name="isSemver strict true", + expr='isSemver("1.0.0")', + device=_hopper_gpu(), + want=True, + ), + Case( + name="isSemver strict rejects short", + expr='isSemver("1.0")', + device=_hopper_gpu(), want=False, ), - Case(name="semver major", expr=f"{_ATTR}.cudaComputeCapability.major() == 9", device=_GPU, want=True), - Case(name="semver minor", expr=f"{_ATTR}.cudaComputeCapability.minor() == 5", device=_GPU, want=True), - Case(name="semver patch", expr=f"{_ATTR}.cudaComputeCapability.patch() == 3", device=_GPU, want=True), - Case(name="semver equality", expr=f'{_ATTR}.cudaComputeCapability == semver("9.5.3")', device=_GPU, want=True), - Case(name="isSemver strict true", expr='isSemver("1.0.0")', device=_GPU, want=True), - Case(name="isSemver strict rejects short", expr='isSemver("1.0")', device=_GPU, want=False), - Case(name="isSemver normalize accepts short", expr='isSemver("1.0", true)', device=_GPU, want=True), - Case(name="semver normalize overload", expr='semver("v1.0", true).major() == 1', device=_GPU, want=True), + Case( + name="isSemver normalize accepts short", + expr='isSemver("1.0", true)', + device=_hopper_gpu(), + want=True, + ), + Case( + name="semver normalize overload", + expr='semver("v1.0", true).major() == 1', + device=_hopper_gpu(), + want=True, + ), # Typed scalar attributes (resolve straight to the value, no .string). - Case(name="string attribute", expr=f'{_ATTR}.architecture == "Hopper"', device=_GPU, want=True), Case( - name="string attribute mismatch", + name="string attribute", + expr='device.attributes["gpu.nvidia.com"].architecture == "Hopper"', + device=_hopper_gpu(), + want=True, + ), + Case( + name="string attribute under a non-GPU driver's domain", expr='device.attributes["nic.nvidia.com"].linkType == "infiniband"', - device=_NIC, + device={"driver": "nic.nvidia.com", "attributes": {"linkType": {"string": "infiniband"}}, "capacity": {}}, want=True, ), Case( name="bool attribute true", - expr=f"{_ATTR}.x", - device=_device(attributes={"x": {"bool": True}}), + expr='device.attributes["gpu.nvidia.com"].x', + device=_scalar_device(x={"bool": True}), want=True, ), Case( name="bool attribute false", - expr=f"{_ATTR}.x", - device=_device(attributes={"x": {"bool": False}}), + expr='device.attributes["gpu.nvidia.com"].x', + device=_scalar_device(x={"bool": False}), want=False, ), Case( name="int attribute", - expr=f"{_ATTR}.x >= 8", - device=_device(attributes={"x": {"int": 8}}), + expr='device.attributes["gpu.nvidia.com"].x >= 8', + device=_scalar_device(x={"int": 8}), want=True, ), Case( name="int attribute below", - expr=f"{_ATTR}.x >= 8", - device=_device(attributes={"x": {"int": 4}}), + expr='device.attributes["gpu.nvidia.com"].x >= 8', + device=_scalar_device(x={"int": 4}), want=False, ), # Qualified names split into their own domain. Case( name="qualified name under its domain", expr='device.attributes["resource.kubernetes.io"].pcieRoot == "pci0"', - device=_GPU, + device=_hopper_gpu(), + want=True, + ), + # The same input as "string attribute", kept to mirror upstream's + # separate driver-name-qualifier row. + Case( + name="bare name under driver domain", + expr='device.attributes["gpu.nvidia.com"].architecture == "Hopper"', + device=_hopper_gpu(), want=True, ), - Case(name="bare name under driver domain", expr=f'{_ATTR}.architecture == "Hopper"', device=_GPU, want=True), # Non-matches that must not raise. Case( name="two-component version is non-match", - expr=f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("8.0.0"))', - device=_device(attributes={"cudaComputeCapability": {"version": "9.0"}}), + expr='device.attributes["gpu.nvidia.com"].cudaComputeCapability.isGreaterThan(semver("8.0.0"))', + device={ + "driver": "gpu.nvidia.com", + "attributes": {"cudaComputeCapability": {"version": "9.0"}}, + "capacity": {}, + }, want=False, ), Case( name="malformed quantity is non-match", - expr=f'{_CAP}.memory.compareTo(quantity("1Gi")) >= 0', - device=_device(capacity={"memory": {"value": "10Mo"}}), + expr='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("1Gi")) >= 0', + device={"driver": "gpu.nvidia.com", "attributes": {}, "capacity": {"memory": {"value": "10Mo"}}}, + want=False, + ), + Case( + name="unknown id is non-match", + expr='device.attributes["gpu.nvidia.com"].nope == "x"', + device=_hopper_gpu(), want=False, ), - Case(name="unknown id is non-match", expr=f'{_ATTR}.nope == "x"', device=_GPU, want=False), # A non-bool selector must not spuriously match. Upstream rejects it # at compile time; we treat a non-bool result as a non-match. - Case(name="non-bool string selector is non-match", expr='"5"', device=_GPU, want=False), + Case( + name="non-bool string selector is non-match", + expr='"5"', + device=_hopper_gpu(), + want=False, + ), Case( name="non-bool int selector is non-match", - expr=f"{_ATTR}.x", - device=_device(attributes={"x": {"int": 5}}), + expr='device.attributes["gpu.nvidia.com"].x', + device=_scalar_device(x={"int": 5}), want=False, ), # Domain presence. Upstream's domain-presence idiom is "" in @@ -211,42 +297,53 @@ class Case: # compile error on a real cluster (celpy accepts it - see cel.py's # documented divergences). An unknown domain is simply absent (False), # not present-but-empty. - Case(name="unknown domain absent", expr='"other.com" in device.attributes', device=_GPU, want=False), - Case(name="known domain present", expr='"gpu.nvidia.com" in device.attributes', device=_GPU, want=True), + Case( + name="unknown domain absent", + expr='"other.com" in device.attributes', + device=_hopper_gpu(), + want=False, + ), + Case( + name="known domain present", + expr='"gpu.nvidia.com" in device.attributes', + device=_hopper_gpu(), + want=True, + ), # Reading an unknown domain resolves to an empty map (not an error), # so an id lookup under it is a non-match rather than a failure. Case( name="unknown domain id is non-match", expr='device.attributes["other.com"].x == "y"', - device=_GPU, + device=_hopper_gpu(), want=False, ), # Guard a domain read with the in idiom before indexing it. Case( name="guarded known domain", - expr=f'"gpu.nvidia.com" in device.attributes && {_ATTR}.architecture == "Hopper"', - device=_GPU, + expr='"gpu.nvidia.com" in device.attributes && device.attributes["gpu.nvidia.com"].architecture == "Hopper"', + device=_hopper_gpu(), want=True, ), # The full design selector. Case( name="full design expression", expr=( - f'{_ATTR}.cudaComputeCapability.isGreaterThan(semver("9.0.0")) && ' - f'{_CAP}.memory.compareTo(quantity("141Gi")) >= 0' + 'device.attributes["gpu.nvidia.com"].cudaComputeCapability.isGreaterThan(semver("9.0.0")) && ' + 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("141Gi")) >= 0' ), - device=_GPU, + device=_hopper_gpu(), want=True, ), - # Verbatim selector examples from the DRA docs, each against a device - # that should and should not match. + # Verbatim selector examples from the DRA docs (the k8s.io concept page + # and the allocate-devices-dra task page), each against a device that + # should match, and all but small-white against one that shouldn't. Case( name="docs: large-black subrequest matches", expr=( 'device.attributes["resource-driver.example.com"].color == "black" && ' 'device.attributes["resource-driver.example.com"].size == "large"' ), - device=_LARGE_BLACK, + device=_example_device(color="black", size="large"), want=True, ), Case( @@ -255,7 +352,7 @@ class Case: 'device.attributes["resource-driver.example.com"].color == "black" && ' 'device.attributes["resource-driver.example.com"].size == "large"' ), - device=_SMALL_WHITE, + device=_example_device(color="white", size="small"), want=False, ), Case( @@ -264,19 +361,19 @@ class Case: 'device.attributes["resource-driver.example.com"].color == "white" && ' 'device.attributes["resource-driver.example.com"].size == "small"' ), - device=_SMALL_WHITE, + device=_example_device(color="white", size="small"), want=True, ), Case( name="docs: extended-resource DeviceClass selector matches", expr="device.driver == 'gpu.example.com' && device.attributes['gpu.example.com'].type == 'gpu'", - device=_EXAMPLE_GPU, + device={"driver": "gpu.example.com", "attributes": {"type": {"string": "gpu"}}, "capacity": {}}, want=True, ), Case( name="docs: extended-resource DeviceClass selector rejects other driver", expr="device.driver == 'gpu.example.com' && device.attributes['gpu.example.com'].type == 'gpu'", - device=_NIC, + device={"driver": "nic.nvidia.com", "attributes": {"linkType": {"string": "infiniband"}}, "capacity": {}}, want=False, ), Case( @@ -285,7 +382,11 @@ class Case: 'device.attributes["driver.example.com"].type == "gpu" && ' 'device.capacity["driver.example.com"].memory == quantity("64Gi")' ), - device=_GPU_64GI, + device={ + "driver": "driver.example.com", + "attributes": {"type": {"string": "gpu"}}, + "capacity": {"memory": {"value": "64Gi"}}, + }, want=True, ), Case( @@ -294,11 +395,11 @@ class Case: 'device.attributes["driver.example.com"].type == "gpu" && ' 'device.capacity["driver.example.com"].memory == quantity("64Gi")' ), - device=_device( - driver="driver.example.com", - attributes={"type": {"string": "gpu"}}, - capacity={"memory": {"value": "32Gi"}}, - ), + device={ + "driver": "driver.example.com", + "attributes": {"type": {"string": "gpu"}}, + "capacity": {"memory": {"value": "32Gi"}}, + }, want=False, ), ] diff --git a/functions/compose-model-deployment/tests/test_fn.py b/functions/compose-model-deployment/tests/test_fn.py index f937959ab..ec43c75f5 100644 --- a/functions/compose-model-deployment/tests/test_fn.py +++ b/functions/compose-model-deployment/tests/test_fn.py @@ -16,9 +16,8 @@ import asyncio import dataclasses -import datetime import json -from typing import Any +from typing import Any, Literal import pytest from crossplane.function import resource @@ -27,1698 +26,2278 @@ from google.protobuf import duration_pb2 as durationpb from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb -from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 -from models.ai.modelplane.modelcache import v1alpha1 as mcv1alpha1 from models.ai.modelplane.modeldeployment import v1alpha1 from models.ai.modelplane.modelreplica import v1alpha1 as mrv1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -# The selector used on the deployment's single GPU request, echoed verbatim -# into each resolved device request. -_GPU_CEL = 'device.driver == "gpu.nvidia.com"' -# A fixed transition time keeps observed conditions deterministic. -_TRANSITION_TIME = datetime.datetime(2025, 1, 1, tzinfo=datetime.UTC) - -# The resolved DRA device requests the scheduler stamps onto each ModelReplica -# member, derived from the deployment's nodeSelector matched against the -# cluster's GPU device (deviceClassName gpu.nvidia.com). -_DEVICE_REQUESTS = [ - { - "name": "gpu", - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "selectors": [{"cel": _GPU_CEL}], - } -] +@dataclasses.dataclass +class ComposeCase: + """A test case for RunFunction.""" -# The single Standalone-member engine every fixture deployment uses. -_ENGINE = v1alpha1.Engine( - name="main", - members=[ - v1alpha1.Member( - role="Standalone", - nodeSelector=v1alpha1.NodeSelector( - devices=[v1alpha1.Device(name="gpu", count=1, selectors=[v1alpha1.Selector(cel=_GPU_CEL)])], - ), - template=v1alpha1.Template( - spec=v1alpha1.Spec( - containers=[ - v1alpha1.Container( - name="engine", - image="vllm/vllm-openai:latest", - args=["--model=Qwen/Qwen3-0.6B"], - ), - ], - ), - ), - ), - ], -) + name: str + req: fnv1.RunFunctionRequest + want: fnv1.RunFunctionResponse -# A Standalone engine with no container args, for the co-location case. -_ENGINE_NO_ARGS = v1alpha1.Engine( - name="main", - members=[ - v1alpha1.Member( - role="Standalone", - nodeSelector=v1alpha1.NodeSelector( - devices=[v1alpha1.Device(name="gpu", count=1, selectors=[v1alpha1.Selector(cel=_GPU_CEL)])], - ), - template=v1alpha1.Template( - spec=v1alpha1.Spec(containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")]), - ), - ), - ], -) +@dataclasses.dataclass +class ResolveRequiredCase: + """A test case for fn.resolve_required.""" -def _replica_engines(*, args: bool = True) -> list: - """The composed ModelReplica's spec.engines: one engine whose Standalone - member carries the matched pool and its resolved device requests. + name: str + req: fnv1.RunFunctionRequest + # resolve_required's name parameter, renamed so it doesn't collide with + # the case's own name. + requirement: str + want: tuple[fn.Resolution, dict | None] - args toggles the engine container's --model arg, matching the fixture - deployment a want is built from. - Every container carries MODELPLANE_SERVED_MODEL_NAME, ahead of any env the - user wrote, so an arg can reference it. It's how an engine comes up under the - name Modelplane routes to instead of Modelplane having to be told what the - engine was started with. - """ - container: dict[str, Any] = { - "name": "engine", - "image": "vllm/vllm-openai:latest", - "env": [{"name": "MODELPLANE_SERVED_MODEL_NAME", "value": "ml-team/my-model"}], - } - if args: - container["args"] = ["--model=Qwen/Qwen3-0.6B"] - return [ - { - "name": "main", - "copies": 1, - "members": [ - { - # No worker block: it's only set on Worker members, and - # the XRD deliberately has no schema default (defaults - # apply before CEL validation, which forbids worker on a - # Standalone). - "role": "Standalone", - "nodePoolName": "default", - "deviceRequests": _DEVICE_REQUESTS, - "template": {"spec": {"containers": [container]}}, - } - ], - } - ] +@dataclasses.dataclass +class InjectServedModelNameCase: + """A test case for fn._inject_served_model_name.""" + name: str + template: mrv1alpha1.Template + served: str + want: mrv1alpha1.Template -# The composed spec.engines for the args-bearing fixture deployment, shared by -# most wants below. -_REPLICA_ENGINES = _replica_engines() -_REPLICA_ENGINES_NO_ARGS = _replica_engines(args=False) -# The composed spec.engines for a PrefillDecode deployment: the standard -# single-GPU Standalone engine, one marked Prefill and one Decode. -_PD_REPLICA_ENGINES = [ - {**_replica_engines()[0], "name": "prefill", "phase": "Prefill"}, - {**_replica_engines()[0], "name": "decode", "phase": "Decode"}, -] +@dataclasses.dataclass +class ServedModelNameCase: + """A test case for fn.served_model_name.""" -# A one-replica deployment requesting a single GPU. Reused across most cases. -_XR = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE])), - ), -).model_dump(exclude_none=True, mode="json") + name: str + namespace: str + deployment: str + want: str -def _cluster( - name: str, +def _model_deployment( *, - ready: bool = True, - hostname: str | None = "cluster.clusters.example.com", - nodes: int = 2, - placement_labels: dict[str, str] | None = None, - cache_storage: bool = False, -) -> dict: - """An InferenceCluster input fixture, dumped to a dict. - - A ready cluster has a Ready=True condition and a gateway hostname. ready=False - flips the condition to Unavailable; hostname=None drops the gateway entirely - (mirroring an offline cluster). nodes=0 yields a pool with no capacity. - cache_storage reports an RWX StorageClass in status.cache, which a cluster - needs to stage a ModelCache. - """ - return icv1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta(name=name), - spec=icv1alpha1.Spec( - cluster=icv1alpha1.Cluster( - source="Existing", - existing=icv1alpha1.Existing(secretRef=icv1alpha1.SecretRef(name="k")), - ), - placement=( - icv1alpha1.Placement(metadata=icv1alpha1.Metadata(labels=placement_labels)) - if placement_labels - else None + replicas: int, + template_labels: dict[str, str] | None, + cluster_selector: v1alpha1.ClusterSelector | None, + model_cache: str | None, + serving_mode: Literal["Unified", "PrefillDecode"] | None, + engines: list[dict[str, Any]], + args: list[str] | None, +) -> fnv1.Resource: + """The observed ModelDeployment my-model, each of its engines running one Standalone GPU member.""" + member = v1alpha1.Member( + role="Standalone", + nodeSelector=v1alpha1.NodeSelector( + devices=[ + v1alpha1.Device( + name="gpu", + count=1, + selectors=[v1alpha1.Selector(cel='device.driver == "gpu.nvidia.com"')], + ), + ], + ), + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest", args=args)], ), ), - status=icv1alpha1.Status( - conditions=[ - icv1alpha1.Condition( - type="Ready", - status="True" if ready else "False", - reason="Available" if ready else "Unavailable", - lastTransitionTime=_TRANSITION_TIME, - ) - ], - gateway=icv1alpha1.Gateway(address="10.0.0.1", hostname=hostname) if hostname else None, - cache=icv1alpha1.CacheModel(storageClassName="rwx") if cache_storage else None, - providerConfigRef=icv1alpha1.ProviderConfigRef(name=name), - gpuPools=[ - icv1alpha1.GpuPool( - name="default", - nodes=nodes, - devices=[ - icv1alpha1.Device( - name="gpu", - claim="DRA", - driver="gpu.nvidia.com", - deviceClassName="gpu.nvidia.com", - count=1, - ) - ], - ) - ], + ) + xr = v1alpha1.ModelDeployment( + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), + spec=v1alpha1.SpecModel1( + replicas=replicas, + template=v1alpha1.TemplateModel( + metadata=v1alpha1.Metadata(labels=template_labels) if template_labels is not None else None, + spec=v1alpha1.SpecModel( + clusterSelector=cluster_selector, + modelCacheRef=v1alpha1.ModelCacheRef(name=model_cache) if model_cache is not None else None, + serving=v1alpha1.Serving(mode=serving_mode) if serving_mode is not None else None, + engines=[v1alpha1.Engine(**engine, members=[member]) for engine in engines], + ), + ), ), - ).model_dump(exclude_none=True, mode="json") - - -# A ready cluster with a two-node GPU pool. Reused across most cases. -_CLUSTER_A = _cluster("cluster-a") - -# cluster-a with the RWX storage a ModelCache needs. -_CLUSTER_A_CACHE = _cluster("cluster-a", cache_storage=True) - - -def _cache(name: str, *, match_labels: dict[str, str] | None = None) -> dict: - """Build an observed ModelCache in the ml-team namespace. + ) + return fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json", by_alias=True))) - match_labels, when given, sets spec.clusterSelector.matchLabels - the - footprint the deployment scheduler intersects with its own selector. - """ - selector = mcv1alpha1.ClusterSelector(matchLabels=match_labels) if match_labels else None - return mcv1alpha1.ModelCache( - metadata=metav1.ObjectMeta(name=name, namespace="ml-team"), - spec=mcv1alpha1.Spec( - source="HuggingFace", - huggingFace=mcv1alpha1.HuggingFace(repo="Qwen/Qwen2.5-7B", sizeGiB=20), - clusterSelector=selector, - ), - ).model_dump(exclude_none=True, mode="json") +def _desired_model_deployment(*, total_replicas: int, ready_replicas: int, ready: fnv1.Ready) -> fnv1.Resource: + """The desired ModelDeployment, with how many replicas it scheduled and how many are ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct({"status": {"replicas": {"total": total_replicas, "ready": ready_replicas}}}), + ready=ready, + ) -def _replica_status(replica: dict, *, ready: bool) -> dict: - """Return a copy of an observed ModelReplica with a Ready condition. - compose_endpoints gates each ModelEndpoint on its ModelReplica reporting - Ready=True - the replica's engines are serving and its remote Service and - HTTPRoute exist - so the endpoint never advertises a backend still warming - up (#102). Tests stamp the condition on the observed replica to drive that - gate. - """ - replica = dict(replica) - replica["status"] = { +def _cluster( + *, + name: str, + ready: bool, + gateway_hostname: str | None, + nodes: int, + cache_storage: bool, + placement_labels: dict[str, str] | None, +) -> fnv1.Resource: + """An observed InferenceCluster with one pool of single-GPU nodes.""" + spec: dict[str, Any] = { + "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k", "key": "kubeconfig"}}}, + "stack": "Standard", + } + if placement_labels is not None: + spec["placement"] = {"metadata": {"labels": placement_labels}} + status: dict[str, Any] = { "conditions": [ { "type": "Ready", "status": "True" if ready else "False", - "reason": "Available" if ready else "Creating", + "reason": "Available" if ready else "Unavailable", "lastTransitionTime": "2025-01-01T00:00:00Z", } - ] + ], + "providerConfigRef": {"name": name}, + "gpuPools": [ + { + "name": "default", + "nodes": nodes, + "devices": [ + { + "name": "gpu", + "claim": "DRA", + "driver": "gpu.nvidia.com", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + } + ], + } + ], + } + if gateway_hostname is not None: + status["gateway"] = {"address": "10.0.0.1", "hostname": gateway_hostname} + if cache_storage: + status["cache"] = {"storageClassName": "rwx"} + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": name}, + "spec": spec, + "status": status, + } + ) + ) + + +def _cache(*, cluster_selector: dict | None) -> fnv1.Resource: + """The ModelCache qwen, which a deployment references by modelCacheRef.""" + spec: dict[str, Any] = { + "source": "HuggingFace", + "huggingFace": {"repo": "Qwen/Qwen2.5-7B", "sizeGiB": 20}, } - return replica + if cluster_selector is not None: + spec["clusterSelector"] = cluster_selector + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelCache", + "metadata": {"name": "qwen", "namespace": "ml-team"}, + "spec": spec, + } + ) + ) -# An existing ModelReplica pinned to cluster-a, observed across cases 5 and 6. -_EXISTING_REPLICA = mrv1alpha1.ModelReplica( - metadata=metav1.ObjectMeta( - name="my-model-5ab63", - namespace="ml-team", - labels={ - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", +def _observed_replica(*, ready: bool | None) -> fnv1.Resource: + """The ModelReplica my-model has on cluster-a, with no Ready condition if ready is None.""" + replica: dict[str, Any] = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": { + "name": "my-model-5ab63", + "namespace": "ml-team", + "labels": { + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, }, - ), - spec=mrv1alpha1.SpecModel( - clusterName="cluster-a", - engines=[ - mrv1alpha1.Engine( - name="main", - copies=1, - members=[ - mrv1alpha1.Member( - role="Standalone", - nodePoolName="default", - deviceRequests=[ - mrv1alpha1.DeviceRequest( - name="gpu", - deviceClassName="gpu.nvidia.com", - count=1, - selectors=[mrv1alpha1.Selector(cel=_GPU_CEL)], - ), - ], - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")], - ), - ), - ), - ], - ), - ], - ), -).model_dump(exclude_none=True, mode="json") - -# The ModelReplica a deployment that references ModelCache qwen composes on -# cluster-b, as the first replica there. -_CACHED_REPLICA_B = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-f0b76", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-b", - "modelplane.ai/replica-index": "0", + "spec": { + "clusterName": "cluster-a", + "engines": [ + { + "name": "main", + "copies": 1, + "members": [ + { + "role": "Standalone", + "nodePoolName": "default", + "deviceRequests": [ + { + "name": "gpu", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [{"cel": 'device.driver == "gpu.nvidia.com"'}], + } + ], + "template": { + "spec": {"containers": [{"name": "engine", "image": "vllm/vllm-openai:latest"}]} + }, + } + ], + } + ], }, - }, - "spec": {"clusterName": "cluster-b", "modelCacheRef": {"name": "qwen"}, "engines": _REPLICA_ENGINES}, -} + } + if ready is not None: + replica["status"] = { + "conditions": [ + { + "type": "Ready", + "status": "True" if ready else "False", + "reason": "Available" if ready else "Creating", + "lastTransitionTime": "2025-01-01T00:00:00Z", + } + ] + } + return fnv1.Resource(resource=resource.dict_to_struct(replica)) -# The requirements selectors every want echoes back. Both are bare selectors -# matching all resources of the kind. -_CLUSTER_SEL = fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster") -_REPLICA_SEL = fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica") + +def _observed_endpoint() -> fnv1.Resource: + """The ModelEndpoint my-model has for its replica on cluster-a, as observed.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, + } + ) + ) -def _req( - xr: dict, +def _composed_replica( *, - clusters: list[dict] | None = None, - replicas: list[dict] | None = None, - observed: dict | None = None, - cache: dict | None = None, - cache_resolved_empty: bool = False, -) -> fnv1.RunFunctionRequest: - """Build a RunFunctionRequest with the standard required_resources. - - clusters and replicas populate the "clusters" and "all-replicas" required - resources respectively; both keys are always present (empty when there are - no items). cache, when given, populates the "cache" required resource the - function declares for a deployment that sets modelCacheRef. - cache_resolved_empty marks the "cache" requirement resolved-but-empty (the - key present with no items, i.e. the cache doesn't exist) - distinct from - omitting it, which leaves the requirement unresolved. observed populates - observed.resources. - """ - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), - resources={key: fnv1.Resource(resource=resource.dict_to_struct(r)) for key, r in (observed or {}).items()}, + name: str, + cluster: str, + labels: dict[str, str], + model_cache: str | None, + serving_mode: str | None, + engines: list[dict[str, str]], + args: list[str] | None, + ready: fnv1.Ready, +) -> fnv1.Resource: + """A composed ModelReplica of my-model, each of its engines running one Standalone GPU member.""" + container: dict[str, Any] = {"name": "engine", "image": "vllm/vllm-openai:latest"} + if args is not None: + container["args"] = args + # Every container carries MODELPLANE_SERVED_MODEL_NAME, ahead of any env the + # user wrote, so an arg can reference it. It's how an engine comes up under + # the name Modelplane routes to, instead of Modelplane having to be told + # what it was started with. + container["env"] = [{"name": "MODELPLANE_SERVED_MODEL_NAME", "value": "ml-team/my-model"}] + member = { + # No worker block: it's only set on Worker members, and the XRD + # deliberately has no schema default (defaults apply before CEL + # validation, which forbids worker on a Standalone). + "role": "Standalone", + "nodePoolName": "default", + # The deployment's nodeSelector matched against the cluster's GPU device + # (deviceClassName gpu.nvidia.com), with its CEL selector echoed + # verbatim. + "deviceRequests": [ + { + "name": "gpu", + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [{"cel": 'device.driver == "gpu.nvidia.com"'}], + } + ], + "template": {"spec": {"containers": [container]}}, + } + spec: dict[str, Any] = {"clusterName": cluster} + if model_cache is not None: + spec["modelCacheRef"] = {"name": model_cache} + if serving_mode is not None: + spec["serving"] = {"mode": serving_mode} + spec["engines"] = [{**engine, "copies": 1, "members": [member]} for engine in engines] + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelReplica", + "metadata": {"name": name, "namespace": "ml-team", "labels": labels}, + "spec": spec, + } ), + ready=ready, ) - if clusters: - for c in clusters: - req.required_resources["clusters"].items.append(fnv1.Resource(resource=resource.dict_to_struct(c))) - else: - req.required_resources["clusters"].SetInParent() - if replicas: - for r in replicas: - req.required_resources["all-replicas"].items.append(fnv1.Resource(resource=resource.dict_to_struct(r))) - else: - req.required_resources["all-replicas"].SetInParent() - if cache is not None: - req.required_resources["cache"].items.append(fnv1.Resource(resource=resource.dict_to_struct(cache))) - elif cache_resolved_empty: - req.required_resources["cache"].SetInParent() - return req - -def _want( - resp: fnv1.RunFunctionResponse, - *, - cluster_labels: dict[str, str] | None = None, - cache_name: str | None = None, -) -> fnv1.RunFunctionResponse: - """Attach the requirements selectors a reconcile echoes back. - cluster_labels narrows the "clusters" selector (the intersection of the - deployment's and the referenced cache's clusterSelectors). cache_name adds - the "cache" selector the function declares for a referenced ModelCache. - """ - cluster_sel = fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster") - if cluster_labels: - cluster_sel.match_labels.labels.update(cluster_labels) - resp.requirements.resources["clusters"].CopyFrom(cluster_sel) - resp.requirements.resources["all-replicas"].CopyFrom(_REPLICA_SEL) - if cache_name is not None: - resp.requirements.resources["cache"].CopyFrom( - fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="ModelCache", - match_name=cache_name, - namespace="ml-team", - ) +def _composed_endpoint(*, labels: dict[str, str]) -> fnv1.Resource: + """The ModelEndpoint for my-model's replica on cluster-a, as composed.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": {"name": "my-model-5ab63", "namespace": "ml-team", "labels": labels}, + "spec": { + "origin": "https://cluster.clusters.example.com", + "api": {"schema": "OpenAI", "prefix": "/ml-team/my-model-5ab63/v1"}, + "model": "ml-team/my-model", + }, + } ) - return resp - - -@dataclasses.dataclass -class Case: - """A test case for compose-model-deployment.""" - - name: str - req: fnv1.RunFunctionRequest - want: fnv1.RunFunctionResponse - + ) -def _compose_cases() -> list[Case]: - """The cases for test_compose, and the deployments they share.""" - # A deployment that sets spec.modelCacheRef. - xr_cached = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - spec=v1alpha1.SpecModel( - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), - engines=[_ENGINE], - ) - ), - ), - ).model_dump(exclude_none=True, mode="json") - # A cached deployment that also sets its own clusterSelector, so the - # scheduler intersects it with the cache's footprint. - xr_cached_selector = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - spec=v1alpha1.SpecModel( - clusterSelector=v1alpha1.ClusterSelector(matchLabels={"region": "us-east"}), - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), - engines=[_ENGINE], - ) - ), - ), - ).model_dump(exclude_none=True, mode="json") +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - # A two-replica deployment (no container args) for the co-location case. - xr_two = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=2, - template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE_NO_ARGS])), - ), - ).model_dump(exclude_none=True, mode="json") - # A disaggregated (PrefillDecode) deployment: a Prefill and a Decode engine. - xr_pd = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - spec=v1alpha1.SpecModel( - serving=v1alpha1.Serving(mode="PrefillDecode"), - engines=[ - _ENGINE.model_copy(update={"name": "prefill", "phase": "Prefill"}), - _ENGINE.model_copy(update={"name": "decode", "phase": "Decode"}), - ], - ) +COMPOSE_CASES = [ + # First reconcile: the replica is composed but not yet observed + # Ready, so its endpoint is withheld - routing must not advertise + # a backend whose pods are still warming up (#102). + ComposeCase( + name="freshly scheduled replica composes no endpoint until ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + }, ), - ).model_dump(exclude_none=True, mode="json") - - # A deployment parked at zero replicas. - xr_zero = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=0, - template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE])), - ), - ).model_dump(exclude_none=True, mode="json") - - return [ - Case( - # First reconcile: the replica is composed but not yet observed - # Ready, so its endpoint is withheld - routing must not advertise - # a backend whose pods are still warming up (#102). - name="freshly scheduled replica composes no endpoint until ready", - req=_req(_XR, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, - }, - } - ), - ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) + }, ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], ), - Case( - # A replica that has gone not-Ready (e.g. a crash-loop after - # once serving) has its endpoint withdrawn: the previously - # observed endpoint is absent from desired, so Crossplane - # deletes it and traffic stops routing to the dead backend - # (#102). Omitting it from desired - not composing it - is what - # drives the deletion. - name="not-ready replica withdraws its endpoint", - req=_req( - _XR, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={ - "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=False), - "endpoint-cluster-a-0": { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, - }, + ), + # A replica that has gone not-Ready (e.g. a crash-loop after once + # serving) has its endpoint withdrawn: the previously observed + # endpoint is absent from desired, so Crossplane deletes it and + # traffic stops routing to the dead backend (#102). Omitting it from + # desired - not composing it - is what drives the deletion. + ComposeCase( + name="not-ready replica withdraws its endpoint", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=False), + "endpoint-cluster-a-0": _observed_endpoint(), }, ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, - }, - } - ), - ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - context=structpb.Struct(), - ) + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], ), - Case( - name="no clusters produces warning", - req=_req(_XR, clusters=[]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoClusters", - ), - ], - results=[ - fnv1.Result(severity=fnv1.SEVERITY_WARNING, message="No InferenceClusters found"), - ], - context=structpb.Struct(), - ) + ), + ComposeCase( + name="no clusters produces warning", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), ), + required_resources={ + "clusters": fnv1.Resources(), + "all-replicas": fnv1.Resources(), + }, ), - Case( - name="insufficient capacity produces no replicas", - req=_req(_XR, clusters=[_cluster("cluster-a", nodes=0)]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_FALSE, - ), - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="InsufficientCapacity", - message="0 of 1 replicas scheduled (checked 1 clusters)", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoReplicasScheduled", - ), - ], - context=structpb.Struct(), - ) + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + # Inline rather than _desired_model_deployment, because with no + # cluster the function returns before it writes any status. + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_WARNING, message="No InferenceClusters found"), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoClusters", + ), + ], ), - Case( - # Zero desired parks the deployment before resolve_inputs runs: - # no requirements are declared (the want carries none), nothing - # is composed, and both conditions read True with the - # NoReplicasDesired reason rather than a capacity failure. - name="scaled to zero composes nothing and reports NoReplicasDesired", - req=_req(xr_zero), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_TRUE, - ), + ), + ComposeCase( + name="insufficient capacity produces no replicas", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", - ), - ], - context=structpb.Struct(), ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=0, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + }, ), - Case( - # Scaling an existing deployment to zero: the observed replica - # and endpoint are absent from desired (pruned), and the - # transition is announced while they still exist. - name="scale to zero prunes observed replicas and emits an event", - req=_req( - xr_zero, - observed={ - "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True), - "endpoint-cluster-a-0": { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, - }, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=0, ready_replicas=0, ready=fnv1.READY_FALSE), + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_TRUE, - ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="InsufficientCapacity", + message="0 of 1 replicas scheduled (checked 1 clusters)", ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="NoReplicasDesired", - message="0 replicas desired", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scaled to zero: removing all replicas", - ), - ], - context=structpb.Struct(), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReplicasScheduled", + ), + ], + ), + ), + # Zero desired parks the deployment before resolve_inputs runs: no + # requirements are declared (the want carries none), nothing is + # composed, and both conditions read True with the NoReplicasDesired + # reason rather than a capacity failure. + ComposeCase( + name="scaled to zero composes nothing and reports NoReplicasDesired", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=0, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources(), + "all-replicas": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=0, ready_replicas=0, ready=fnv1.READY_TRUE), ), + context=structpb.Struct(), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + ], ), - Case( - name="ready replica is preserved and keeps its endpoint", - req=_req( - _XR, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={ - "replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True), - "endpoint-cluster-a-0": { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": {"name": "my-model-5ab63", "namespace": "ml-team"}, - }, + ), + # Scaling an existing deployment to zero: the observed replica and + # endpoint are absent from desired (pruned), and the transition is + # announced while they still exist. + ComposeCase( + name="scale to zero prunes observed replicas and emits an event", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=0, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=True), + "endpoint-cluster-a-0": _observed_endpoint(), }, ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 1}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "endpoint-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "origin": "https://cluster.clusters.example.com", - "api": { - "schema": "OpenAI", - "prefix": "/ml-team/my-model-5ab63/v1", - }, - "model": "ml-team/my-model", - }, - } - ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="AllReplicasReady", - message="1 of 1 ready", - ), - ], - context=structpb.Struct(), - ) + required_resources={ + "clusters": fnv1.Resources(), + "all-replicas": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=0, ready_replicas=0, ready=fnv1.READY_TRUE), ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scaled to zero: removing all replicas", + ), + ], + context=structpb.Struct(), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="NoReplicasDesired", + message="0 replicas desired", + ), + ], ), - Case( - name="offline pinned cluster keeps replica but drops endpoint", - req=_req( - _XR, - clusters=[_cluster("cluster-a", ready=False, hostname=None)], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _EXISTING_REPLICA}, - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES, - }, - } - ), - ), - }, - ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - context=structpb.Struct(), - ) + ), + ComposeCase( + name="ready replica is preserved and keeps its endpoint", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=True), + "endpoint-cluster-a-0": _observed_endpoint(), + }, ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + }, ), - Case( - name="deleted pinned cluster triggers replica re-placement", - req=_req( - _XR, - clusters=[_cluster("cluster-b", hostname="cluster-b.clusters.example.com")], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-b-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-f0b76", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-b", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-b", - "engines": _REPLICA_ENGINES, - }, - } - ), - ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=1, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_TRUE, ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-b", - ), - ], - context=structpb.Struct(), - ) + "endpoint-cluster-a-0": _composed_endpoint( + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + } + ), + }, ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="AllReplicasReady", + message="1 of 1 ready", + ), + ], ), - Case( - name="modelCacheRef is propagated onto the composed replica", - req=_req(xr_cached, clusters=[_CLUSTER_A_CACHE], cache=_cache("qwen")), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "modelCacheRef": {"name": "qwen"}, - "engines": _REPLICA_ENGINES, - }, - } - ), - ), + ), + # The replica isn't observed Ready, so it gets no endpoint whatever the + # state of its cluster. + ComposeCase( + name="offline pinned cluster keeps its replica", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=None), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=False, + gateway_hostname=None, + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", ), - cache_name="qwen", + ], + ), + ), + ComposeCase( + name="deleted pinned cluster triggers replica re-placement", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=True), + }, ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-b", + ready=True, + gateway_hostname="cluster-b.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + }, ), - Case( - # The cache stages only to a subset of clusters; the scheduler - # intersects the cache's footprint with the deployment's own - # clusterSelector so replicas never land where the cache isn't. - name="cache clusterSelector is intersected with the deployment's", - req=_req( - xr_cached_selector, - clusters=[_CLUSTER_A_CACHE], - cache=_cache("qwen", match_labels={"tier": "gpu"}), - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "modelCacheRef": {"name": "qwen"}, - "engines": _REPLICA_ENGINES, - }, - } - ), - ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-b-0": _composed_replica( + name="my-model-f0b76", + cluster="cluster-b", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-b", + "modelplane.ai/replica-index": "0", }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-b", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + ), + ), + ComposeCase( + name="modelCacheRef is propagated onto the composed replica", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], ), - cluster_labels={"region": "us-east", "tier": "gpu"}, - cache_name="qwen", ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=True, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + "cache": fnv1.Resources(items=[_cache(cluster_selector=None)]), + }, ), - Case( - # compose-model-cache stages only onto clusters that report cache - # storage, so cluster-a, which reports none, can't host the - # replica's PVC. The replica lands on cluster-b, though cluster-a - # would win the tiebreak by name. - name="a cached replica lands only on a cluster with cache storage", - req=_req( - xr_cached, - clusters=[_CLUSTER_A, _cluster("cluster-b", cache_storage=True)], - cache=_cache("qwen"), - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-b-0": fnv1.Resource(resource=resource.dict_to_struct(_CACHED_REPLICA_B)), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-b", - ), - ], - context=structpb.Struct(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", ), - cache_name="qwen", + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + "cache": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelCache", + match_name="qwen", + namespace="ml-team", + ), + }, ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], ), - Case( - # A replica is running on cluster-a, which has no cache storage, - # say because the bug this guards against put it there. Its PVC - # never appears, so it's dropped and re-placed on cluster-b, the - # way a replica is when the cache's selector stops matching. - name="a running cached replica on a cluster without cache storage is re-placed", - req=_req( - xr_cached, - clusters=[_CLUSTER_A, _cluster("cluster-b", cache_storage=True)], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - cache=_cache("qwen"), - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-b-0": fnv1.Resource(resource=resource.dict_to_struct(_CACHED_REPLICA_B)), + ), + # The cache stages only to a subset of clusters, so the clusters + # requirement matches the labels of both the cache's clusterSelector and + # the deployment's own, and replicas never land where the cache isn't. + ComposeCase( + name="cache clusterSelector is intersected with the deployment's", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=v1alpha1.ClusterSelector(matchLabels={"region": "us-east"}), + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=True, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + "cache": fnv1.Resources(items=[_cache(cluster_selector={"matchLabels": {"tier": "gpu"}})]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-b", - ), - ], - context=structpb.Struct(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", ), - cache_name="qwen", + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_labels=fnv1.MatchLabels(labels={"region": "us-east", "tier": "gpu"}), + ), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + "cache": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelCache", + match_name="qwen", + namespace="ml-team", + ), + }, ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], ), - Case( - # The only candidate has no cache storage, so the cache can't - # stage there and nothing is placed. ReplicasScheduled says why - # rather than blaming capacity. - name="no candidate with cache storage places nothing", - req=_req(xr_cached, clusters=[_CLUSTER_A], cache=_cache("qwen")), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_FALSE, - ), + ), + # compose-model-cache stages only onto clusters that report cache + # storage, so cluster-a, which reports none, can't host the replica's + # PVC. The replica lands on cluster-b, though cluster-a would win the + # tiebreak by name. + ComposeCase( + name="a cached replica lands only on a cluster with cache storage", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ), + _cluster( + name="cluster-b", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=True, + placement_labels=None, + ), + ] + ), + "all-replicas": fnv1.Resources(), + "cache": fnv1.Resources(items=[_cache(cluster_selector=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-b-0": _composed_replica( + name="my-model-f0b76", + cluster="cluster-b", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-b", + "modelplane.ai/replica-index": "0", + }, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ModelCacheResolved", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoCacheStorage", - message="0 of 1 replicas scheduled: no candidate cluster has storage for ModelCache qwen", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoReplicasScheduled", - ), - ], - context=structpb.Struct(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-b", ), - cache_name="qwen", + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + "cache": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelCache", + match_name="qwen", + namespace="ml-team", + ), + }, ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], ), - Case( - # A referenced cache Crossplane hasn't fetched yet leaves the - # footprint unknown. With no replicas to retain, the function - # holds off placing any rather than risk landing them outside the - # footprint: fill is suppressed, so nothing is composed, and - # ModelCacheResolved=False (Unresolved) says why. The wait is - # transient and self-clearing, so it's a condition, not an event. - # The cluster and replica requirements are still declared so the - # cache can resolve alongside them. - name="unresolved cache suppresses new placement", - req=_req(xr_cached, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 0, "ready": 0}}}), - ready=fnv1.READY_FALSE, - ), + ), + # A replica is running on cluster-a, which has no cache storage, say + # because the bug this guards against put it there. Its PVC never + # appears, so it's dropped and re-placed on cluster-b, the way a + # replica is when the cache's selector stops matching. + ComposeCase( + name="a running cached replica on a cluster without cache storage is re-placed", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=True), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ), + _cluster( + name="cluster-b", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=True, + placement_labels=None, + ), + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + "cache": fnv1.Resources(items=[_cache(cluster_selector=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-b-0": _composed_replica( + name="my-model-f0b76", + cluster="cluster-b", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-b", + "modelplane.ai/replica-index": "0", + }, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelCacheUnresolved", - message="Waiting for ModelCache qwen", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="InsufficientCapacity", - message="0 of 1 replicas scheduled (checked 1 clusters)", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="NoReplicasScheduled", - ), - ], - context=structpb.Struct(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-b", ), - cache_name="qwen", + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + "cache": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelCache", + match_name="qwen", + namespace="ml-team", + ), + }, ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], ), - Case( - # The cache a live deployment depends on is deleted (the cache - # requirement resolves but matches nothing - ABSENT). The cache - # only matters when loading weights, which already happened, so - # its disappearance must not tear the deployment down: the - # existing replica is retained (retain ignores fill) even as - # ModelCacheResolved goes False (NotFound) and new placement is - # suppressed. - name="deleted cache retains existing replicas", - req=_req( - xr_cached, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - cache_resolved_empty=True, - ), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 1}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "modelCacheRef": {"name": "qwen"}, - "engines": _REPLICA_ENGINES, - }, - } - ), - ready=fnv1.READY_TRUE, - ), - "endpoint-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelEndpoint", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "origin": "https://cluster.clusters.example.com", - "api": { - "schema": "OpenAI", - "prefix": "/ml-team/my-model-5ab63/v1", - }, - "model": "ml-team/my-model", - }, - } - ), - ), + ), + # The only candidate has no cache storage, so the cache can't stage + # there and nothing is placed. ReplicasScheduled says why rather than + # blaming capacity. + ComposeCase( + name="no candidate with cache storage places nothing", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + "cache": fnv1.Resources(items=[_cache(cluster_selector=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=0, ready_replicas=0, ready=fnv1.READY_FALSE), + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + "cache": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelCache", + match_name="qwen", + namespace="ml-team", + ), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ModelCacheResolved", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoCacheStorage", + message="0 of 1 replicas scheduled: no candidate cluster has storage for ModelCache qwen", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReplicasScheduled", + ), + ], + ), + ), + # A referenced cache Crossplane hasn't fetched yet leaves the + # footprint unknown. With no replicas to retain, the function holds + # off placing any rather than risk landing them outside the + # footprint: fill is suppressed, so nothing is composed, and + # ModelCacheResolved=False (ModelCacheUnresolved) says why. The wait is + # transient and self-clearing, so it's a condition, not an event. The + # cluster and replica requirements are still declared so the cache + # can resolve alongside them. The request has no "cache" requirement + # at all, which is what marks it unresolved. + ComposeCase( + name="unresolved cache suppresses new placement", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=0, ready_replicas=0, ready=fnv1.READY_FALSE), + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + "cache": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelCache", + match_name="qwen", + namespace="ml-team", + ), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelCacheUnresolved", + message="Waiting for ModelCache qwen", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="InsufficientCapacity", + message="0 of 1 replicas scheduled (checked 1 clusters)", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReplicasScheduled", + ), + ], + ), + ), + # The cache a live deployment depends on is deleted: the "cache" + # requirement resolves but matches nothing - ABSENT. The cache only + # matters when loading weights, which already happened, so its + # disappearance must not tear the deployment down: the existing + # replica is retained (retain ignores fill) even as + # ModelCacheResolved goes False (ModelCacheNotFound) and new placement is + # suppressed. + ComposeCase( + name="deleted cache retains existing replicas", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=True), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + "cache": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=1, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache="qwen", + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_TRUE, ), - conditions=[ - fnv1.Condition( - type="ModelCacheResolved", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelCacheNotFound", - message="ModelCache qwen not found; holding replica placement", - ), - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_TRUE, - reason="ReplicasCreated", - message="Scheduled 1 of 1 replicas", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_TRUE, - reason="AllReplicasReady", - message="1 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_WARNING, - message="ModelCache qwen not found; holding replica placement", - ), - ], - context=structpb.Struct(), + "endpoint-cluster-a-0": _composed_endpoint( + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + } + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="ModelCache qwen not found; holding replica placement", ), - cache_name="qwen", + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + "cache": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelCache", + match_name="qwen", + namespace="ml-team", + ), + }, ), + conditions=[ + fnv1.Condition( + type="ModelCacheResolved", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelCacheNotFound", + message="ModelCache qwen not found; holding replica placement", + ), + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="AllReplicasReady", + message="1 of 1 ready", + ), + ], ), - Case( - name="two replicas co-locate on one cluster as distinct resources", - req=_req(xr_two, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 2, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES_NO_ARGS, - }, - } - ), - ), - "replica-cluster-a-1": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-609c5", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "1", - }, - }, - "spec": { - "clusterName": "cluster-a", - "engines": _REPLICA_ENGINES_NO_ARGS, - }, - } - ), - ), + ), + ComposeCase( + name="two replicas co-locate on one cluster as distinct resources", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=2, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=None, + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=2, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=None, + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 2 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 2 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) + "replica-cluster-a-1": _composed_replica( + name="my-model-609c5", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "1", + }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 2 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 2 ready", + ), + ], ), - Case( - # PrefillDecode copies serving and each engine's phase onto the - # replica; the replica backend reads them to front the engines - # with an InferencePool + endpoint picker rather than a Service. - name="PrefillDecode copies serving and engine phases onto the replica", - req=_req(xr_pd, clusters=[_CLUSTER_A]), - want=_want( - fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": {"replicas": {"total": 1, "ready": 0}}}), - ), - resources={ - "replica-cluster-a-0": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelReplica", - "metadata": { - "name": "my-model-5ab63", - "namespace": "ml-team", - "labels": { - "modelplane.ai/deployment": "my-model", - "modelplane.ai/cluster": "cluster-a", - "modelplane.ai/replica-index": "0", - }, - }, - "spec": { - "clusterName": "cluster-a", - "serving": {"mode": "PrefillDecode"}, - "engines": _PD_REPLICA_ENGINES, - }, - } - ), - ), + ), + # PrefillDecode copies serving and each engine's phase onto the + # replica. compose-model-replica reads them to pick disaggregated + # routing, which role-labels the phase engines and puts the pd-sidecar + # on decode. + ComposeCase( + name="PrefillDecode copies serving and engine phases onto the replica", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode="PrefillDecode", + engines=[{"name": "prefill", "phase": "Prefill"}, {"name": "decode", "phase": "Decode"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", }, + model_cache=None, + serving_mode="PrefillDecode", + engines=[ + {"name": "prefill", "phase": "Prefill"}, + {"name": "decode", "phase": "Decode"}, + ], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, ), - conditions=[ - fnv1.Condition( - type="ReplicasScheduled", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Scheduling", - ), - fnv1.Condition( - type="ReplicasReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="ModelStarting", - message="0 of 1 ready", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Scheduled 1 replicas across 1 clusters: cluster-a", - ), - ], - context=structpb.Struct(), - ) + }, ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], ), - ] - - -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + ), + # spec.template.metadata.labels land on the composed ModelReplica and + # ModelEndpoint, alongside the labels Modelplane manages. The + # observed, Ready replica lets the endpoint compose this reconcile. + ComposeCase( + name="template labels are stamped on the replica and endpoint", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels={"tier": "prod", "team": "search"}, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=True), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=1, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "tier": "prod", + "team": "search", + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_TRUE, + ), + "endpoint-cluster-a-0": _composed_endpoint( + labels={ + "tier": "prod", + "team": "search", + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + } + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="AllReplicasReady", + message="1 of 1 ready", + ), + ], + ), + ), + # The XRD's CEL rejects a template label under the modelplane.ai/ + # prefix, but the invariant lives in the function too: managed labels + # are stamped last, so a colliding label can't override them even if + # that CEL rule is relaxed or the function is reused elsewhere. + ComposeCase( + name="a managed label beats a template label of the same key", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels={"modelplane.ai/cluster": "wrong", "tier": "prod"}, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels=None, + ) + ] + ), + "all-replicas": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "tier": "prod", + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + ), + ), + # A cluster's spec.placement.metadata.labels land on the ModelReplica + # and ModelEndpoint composed there. This is the endpoint half of + # residency: a ModelService selects endpoints by label, so without it + # a region-scoped service can't select its own replicas, and nobody + # can label them by hand because Modelplane owns them. The gateway + # half is an InferenceGateway's serviceSelector. + ComposeCase( + name="placement labels are stamped on the replica and endpoint", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels=None, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + resources={ + "replica-cluster-a-0": _observed_replica(ready=True), + }, + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels={"example.org/region": "eu"}, + ) + ] + ), + "all-replicas": fnv1.Resources(items=[_observed_replica(ready=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=1, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "example.org/region": "eu", + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_TRUE, + ), + "endpoint-cluster-a-0": _composed_endpoint( + labels={ + "example.org/region": "eu", + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + } + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_TRUE, + reason="ReplicasCreated", + message="Scheduled 1 of 1 replicas", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="AllReplicasReady", + message="1 of 1 ready", + ), + ], + ), + ), + # The cluster is the authority on where it is, so its placement + # labels are stamped after the deployment's own template labels. + ComposeCase( + name="a cluster's placement label beats a template label of the same key", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_deployment( + replicas=1, + template_labels={"example.org/region": "wrong"}, + cluster_selector=None, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ), + ), + required_resources={ + "clusters": fnv1.Resources( + items=[ + _cluster( + name="cluster-a", + ready=True, + gateway_hostname="cluster.clusters.example.com", + nodes=2, + cache_storage=False, + placement_labels={"example.org/region": "eu"}, + ) + ] + ), + "all-replicas": fnv1.Resources(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_deployment(total_replicas=1, ready_replicas=0, ready=fnv1.READY_UNSPECIFIED), + resources={ + "replica-cluster-a-0": _composed_replica( + name="my-model-5ab63", + cluster="cluster-a", + labels={ + "example.org/region": "eu", + "modelplane.ai/deployment": "my-model", + "modelplane.ai/cluster": "cluster-a", + "modelplane.ai/replica-index": "0", + }, + model_cache=None, + serving_mode=None, + engines=[{"name": "main"}], + args=["--model=Qwen/Qwen3-0.6B"], + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Scheduled 1 replicas across 1 clusters: cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "all-replicas": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="ModelReplica"), + }, + ), + conditions=[ + fnv1.Condition( + type="ReplicasScheduled", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Scheduling", + ), + fnv1.Condition( + type="ReplicasReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="ModelStarting", + message="0 of 1 ready", + ), + ], + ), + ), +] -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) -def test_compose(case: Case) -> None: - """The function fans out ModelReplicas and, once they're Ready, ModelEndpoints.""" +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: ComposeCase) -> None: + """RunFunction fans out ModelReplicas and, once they're Ready, ModelEndpoints.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) assert _to_dict(got) == _to_dict(case.want) -def _composed(resp: fnv1.RunFunctionResponse, kind: str) -> list[dict]: - """Composed desired resources of the given kind, as dicts.""" - result = [] - for r in resp.desired.resources.values(): - d = resource.struct_to_dict(r.resource) - if d.get("kind") == kind: - result.append(d) - return result - - -# spec.template.metadata.labels land on the composed ModelReplicas and -# ModelEndpoints, alongside the labels Modelplane manages. - - -def test_template_labels_stamped_on_replica_and_endpoint() -> None: - """Template labels land on the replica and endpoint beside the managed labels.""" - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - metadata=v1alpha1.Metadata(labels={"tier": "prod", "team": "search"}), - spec=v1alpha1.SpecModel(engines=[_ENGINE]), - ), +RESOLVE_REQUIRED_CASES = [ + ResolveRequiredCase( + name="a requirement that resolved and matched a resource is present", + req=fnv1.RunFunctionRequest( + required_resources={ + "cache": fnv1.Resources( + items=[ + fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelCache", + "metadata": {"name": "qwen"}, + } + ) + ) + ] + ), + }, ), - ).model_dump(exclude_none=True, mode="json") - # An observed, Ready replica lets the endpoint compose this reconcile. - req = _req( - xr, - clusters=[_CLUSTER_A], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - composed = _composed(got, "ModelReplica") + _composed(got, "ModelEndpoint") - assert len(composed) == 2, "expected one ModelReplica and one ModelEndpoint" - for obj in composed: - labels = obj["metadata"]["labels"] - assert labels.get("tier") == "prod" - assert labels.get("team") == "search" - assert labels.get("modelplane.ai/deployment") == "my-model" - assert labels.get("modelplane.ai/cluster") == "cluster-a" - assert labels.get("modelplane.ai/replica-index") == "0" - - -def test_template_labels_managed_labels_win_a_collision() -> None: - """A managed label beats a template label of the same key.""" - # The XRD's CEL rejects a template label under the modelplane.ai/ prefix, - # but the invariant lives in the function too: managed labels are stamped - # last, so a colliding label can't override them even if that CEL rule is - # relaxed or the function is reused elsewhere. - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - metadata=v1alpha1.Metadata(labels={"modelplane.ai/cluster": "wrong", "tier": "prod"}), - spec=v1alpha1.SpecModel(engines=[_ENGINE]), - ), + requirement="cache", + want=( + fn.Resolution.PRESENT, + {"apiVersion": "modelplane.ai/v1alpha1", "kind": "ModelCache", "metadata": {"name": "qwen"}}, ), - ).model_dump(exclude_none=True, mode="json") - got = asyncio.run(fn.FunctionRunner().RunFunction(_req(xr, clusters=[_CLUSTER_A]), None)) - - replica = _composed(got, "ModelReplica")[0] - assert replica["metadata"]["labels"]["modelplane.ai/cluster"] == "cluster-a" - assert replica["metadata"]["labels"]["tier"] == "prod" + ), + # The SDK returns None for this and for a requirement Crossplane hasn't + # fetched alike. Only the requirement's key tells them apart. + ResolveRequiredCase( + name="a requirement that resolved but matched nothing is absent", + req=fnv1.RunFunctionRequest(required_resources={"cache": fnv1.Resources()}), + requirement="cache", + want=(fn.Resolution.ABSENT, None), + ), + ResolveRequiredCase( + name="a requirement Crossplane hasn't fetched is unresolved", + req=fnv1.RunFunctionRequest(), + requirement="cache", + want=(fn.Resolution.UNRESOLVED, None), + ), +] -def test_resolve_required() -> None: +@pytest.mark.parametrize("case", RESOLVE_REQUIRED_CASES, ids=lambda case: case.name) +def test_resolve_required(case: ResolveRequiredCase) -> None: """resolve_required tells a found, a missing, and an unfetched requirement apart.""" - cache = {"apiVersion": "modelplane.ai/v1alpha1", "kind": "ModelCache", "metadata": {"name": "qwen"}} - - # PRESENT: the requirement resolved and matched a resource. - req = fnv1.RunFunctionRequest() - req.required_resources["cache"].items.append(fnv1.Resource(resource=resource.dict_to_struct(cache))) - assert fn.resolve_required(req, "cache") == (fn.Resolution.PRESENT, cache) - - # ABSENT: the requirement resolved but matched nothing (key present, no items). - req = fnv1.RunFunctionRequest() - req.required_resources["cache"].SetInParent() - assert fn.resolve_required(req, "cache") == (fn.Resolution.ABSENT, None) - - # UNRESOLVED: Crossplane has not fetched the requirement (key absent). - req = fnv1.RunFunctionRequest() - assert fn.resolve_required(req, "cache") == (fn.Resolution.UNRESOLVED, None) - - -# The name an engine is started under, and how it gets there. - - -def test_served_model_name_goes_ahead_of_the_users_env() -> None: - """The served model name env var comes before the container's own env.""" - # Env expansion is left to right, so an arg or a later entry referencing - # $(MODELPLANE_SERVED_MODEL_NAME) only resolves if it's first. - template = mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container( - name="engine", - image="vllm/vllm-openai:latest", - env=[mrv1alpha1.EnvItem(name="HF_TOKEN", value="x")], - ) - ] - ) - ) - fn._inject_served_model_name(template, "ml-team/kimi-k2") - assert template.spec is not None - assert [(e.name, e.value) for e in template.spec.containers[0].env or []] == [ - ("MODELPLANE_SERVED_MODEL_NAME", "ml-team/kimi-k2"), - ("HF_TOKEN", "x"), - ] + assert fn.resolve_required(case.req, case.requirement) == case.want + + +INJECT_SERVED_MODEL_NAME_CASES = [ + # Env expansion is left to right, so an arg or a later entry + # referencing $(MODELPLANE_SERVED_MODEL_NAME) only resolves if it's + # first. + InjectServedModelNameCase( + name="the served model name goes ahead of the user's env", + template=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + env=[mrv1alpha1.EnvItem(name="HF_TOKEN", value="x")], + ) + ] + ) + ), + served="ml-team/kimi-k2", + want=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + env=[ + mrv1alpha1.EnvItem(name="MODELPLANE_SERVED_MODEL_NAME", value="ml-team/kimi-k2"), + mrv1alpha1.EnvItem(name="HF_TOKEN", value="x"), + ], + ) + ] + ) + ), + ), + # Modelplane decides this value. Honouring an override would let the + # engine answer to a name nothing routes to, which surfaces as a 404 + # from the engine rather than anything visible in status. + InjectServedModelNameCase( + name="a user override is dropped", + template=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + env=[mrv1alpha1.EnvItem(name="MODELPLANE_SERVED_MODEL_NAME", value="mine")], + ) + ] + ) + ), + served="ml-team/kimi-k2", + want=mrv1alpha1.Template( + spec=mrv1alpha1.Spec( + containers=[ + mrv1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + env=[mrv1alpha1.EnvItem(name="MODELPLANE_SERVED_MODEL_NAME", value="ml-team/kimi-k2")], + ) + ] + ) + ), + ), +] -def test_served_model_name_user_override_is_dropped() -> None: - """A user's own MODELPLANE_SERVED_MODEL_NAME is replaced, not kept.""" - # Modelplane decides this value. Honouring an override would let the engine - # answer to a name nothing routes to, which surfaces as a 404 from the - # engine rather than anything visible in status. - template = mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container( - name="engine", - image="vllm/vllm-openai:latest", - env=[mrv1alpha1.EnvItem(name="MODELPLANE_SERVED_MODEL_NAME", value="mine")], - ) - ] - ) - ) - fn._inject_served_model_name(template, "ml-team/kimi-k2") - assert template.spec is not None - assert [(e.name, e.value) for e in template.spec.containers[0].env or []] == [ - ("MODELPLANE_SERVED_MODEL_NAME", "ml-team/kimi-k2"), - ] +@pytest.mark.parametrize("case", INJECT_SERVED_MODEL_NAME_CASES, ids=lambda case: case.name) +def test_inject_served_model_name(case: InjectServedModelNameCase) -> None: + """_inject_served_model_name puts the served model name first in each container's env.""" + # _inject_served_model_name edits the template in place, so give it a copy + # and leave the table as written. + got = case.template.model_copy(deep=True) + fn._inject_served_model_name(got, case.served) + assert got.model_dump() == case.want.model_dump() -def test_served_model_name_is_namespaced() -> None: - """served_model_name prefixes the deployment's name with its namespace.""" +SERVED_MODEL_NAME_CASES = [ # So two deployments in different namespaces can't collide, and a # ModelService can rewrite one name for a whole deployment. - assert fn.served_model_name("ml-team", "kimi-k2") == "ml-team/kimi-k2" - - -# A cluster's spec.placement.metadata.labels land on the ModelReplicas and -# ModelEndpoints composed there. -# -# This is the endpoint half of residency: a ModelService selects endpoints by -# label, so without it a region-scoped service can't select its own replicas, -# and nobody can label them by hand because Modelplane owns them. The gateway -# half is an InferenceGateway's serviceSelector. - - -def test_placement_labels_stamped_on_replica_and_endpoint() -> None: - """A cluster's placement labels land on the replica and endpoint composed there.""" - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel(spec=v1alpha1.SpecModel(engines=[_ENGINE])), - ), - ).model_dump(exclude_none=True, mode="json") - req = _req( - xr, - clusters=[_cluster("cluster-a", placement_labels={"example.org/region": "eu"})], - replicas=[_EXISTING_REPLICA], - observed={"replica-cluster-a-0": _replica_status(_EXISTING_REPLICA, ready=True)}, - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - composed = _composed(got, "ModelReplica") + _composed(got, "ModelEndpoint") - assert len(composed) == 2, "expected one ModelReplica and one ModelEndpoint" - for obj in composed: - assert obj["metadata"]["labels"].get("example.org/region") == "eu" + ServedModelNameCase( + name="the served model name is namespaced", + namespace="ml-team", + deployment="kimi-k2", + want="ml-team/kimi-k2", + ), +] -def test_placement_labels_cluster_label_beats_a_template_label() -> None: - """A cluster's placement label beats a template label of the same key.""" - # The cluster is the authority on where it is, so its placement labels are - # stamped after the deployment's own template labels. - xr = v1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), - spec=v1alpha1.SpecModel1( - replicas=1, - template=v1alpha1.TemplateModel( - metadata=v1alpha1.Metadata(labels={"example.org/region": "wrong"}), - spec=v1alpha1.SpecModel(engines=[_ENGINE]), - ), - ), - ).model_dump(exclude_none=True, mode="json") - got = asyncio.run( - fn.FunctionRunner().RunFunction( - _req(xr, clusters=[_cluster("cluster-a", placement_labels={"example.org/region": "eu"})]), - None, - ) - ) - replica = _composed(got, "ModelReplica")[0] - assert replica["metadata"]["labels"]["example.org/region"] == "eu" +@pytest.mark.parametrize("case", SERVED_MODEL_NAME_CASES, ids=lambda case: case.name) +def test_served_model_name(case: ServedModelNameCase) -> None: + """served_model_name prefixes the deployment's name with its namespace.""" + assert fn.served_model_name(case.namespace, case.deployment) == case.want diff --git a/functions/compose-model-deployment/tests/test_quantity.py b/functions/compose-model-deployment/tests/test_quantity.py index 5c0429415..957b68106 100644 --- a/functions/compose-model-deployment/tests/test_quantity.py +++ b/functions/compose-model-deployment/tests/test_quantity.py @@ -32,8 +32,9 @@ These expected values come from running the inputs through the real Kubernetes code, not from assertion by hand. When you change the quantity module or want to add a case, derive its expected value with the parity oracle in ./oracle (see -oracle/README.md) rather than reasoning about it - upstream has surprises (e.g. -binary-suffix overflow saturates to int64-max, so 8Ei == 10Ei). +the package comment in oracle/main.go) rather than reasoning about it - +upstream has surprises (e.g. binary-suffix overflow saturates to int64-max, so +8Ei == 10Ei). """ import dataclasses @@ -42,65 +43,65 @@ from function import cel, quantity -def _eval(expr: str) -> bool: - """Compile and evaluate a deviceless boolean CEL expression.""" - return cel.Program(expr).matches({}) - - @dataclasses.dataclass -class Case: +class QuantityCase: + """A test case for a quantity CEL expression.""" + name: str expr: str want: bool @dataclasses.dataclass -class ParseErrCase: +class ParseRejectsCase: + """A test case for a string quantity.parse rejects.""" + name: str - input: str + s: str + want: str QUANTITY_CASES = [ # parse + isQuantity. - Case(name="parse", expr='quantity("12Mi").compareTo(quantity("12Mi")) == 0', want=True), - Case(name="isQuantity int string", expr='isQuantity("20")', want=True), - Case(name="isQuantity megabytes", expr='isQuantity("20M")', want=True), - Case(name="isQuantity mebibytes", expr='isQuantity("20Mi")', want=True), - Case(name="isQuantity invalid suffix", expr='isQuantity("20Mo")', want=False), - Case(name="isQuantity passing regex bad suffix", expr='isQuantity("10Mm")', want=False), + QuantityCase(name="parse", expr='quantity("12Mi").compareTo(quantity("12Mi")) == 0', want=True), + QuantityCase(name="isQuantity int string", expr='isQuantity("20")', want=True), + QuantityCase(name="isQuantity megabytes", expr='isQuantity("20M")', want=True), + QuantityCase(name="isQuantity mebibytes", expr='isQuantity("20Mi")', want=True), + QuantityCase(name="isQuantity invalid suffix", expr='isQuantity("20Mo")', want=False), + QuantityCase(name="isQuantity passing regex bad suffix", expr='isQuantity("10Mm")', want=False), # resource.Quantity accepts decimal exponents and nano/micro suffixes. - Case(name="isQuantity exponent lowercase", expr='isQuantity("256e3")', want=True), - Case(name="isQuantity exponent uppercase", expr='isQuantity("1E3")', want=True), - Case(name="exponent value", expr='quantity("256e3").compareTo(quantity("256000")) == 0', want=True), - Case(name="isQuantity nano", expr='isQuantity("100n")', want=True), - Case(name="isQuantity micro", expr='isQuantity("100u")', want=True), - Case(name="isQuantity trailing dot", expr='isQuantity("5.")', want=True), + QuantityCase(name="isQuantity exponent lowercase", expr='isQuantity("256e3")', want=True), + QuantityCase(name="isQuantity exponent uppercase", expr='isQuantity("1E3")', want=True), + QuantityCase(name="exponent value", expr='quantity("256e3").compareTo(quantity("256000")) == 0', want=True), + QuantityCase(name="isQuantity nano", expr='isQuantity("100n")', want=True), + QuantityCase(name="isQuantity micro", expr='isQuantity("100u")', want=True), + QuantityCase(name="isQuantity trailing dot", expr='isQuantity("5.")', want=True), # The quantity() constructor does NOT trim whitespace. - Case(name="isQuantity leading whitespace false", expr='isQuantity(" 5Gi")', want=False), - Case(name="isQuantity trailing whitespace false", expr='isQuantity("5Gi ")', want=False), + QuantityCase(name="isQuantity leading whitespace false", expr='isQuantity(" 5Gi")', want=False), + QuantityCase(name="isQuantity trailing whitespace false", expr='isQuantity("5Gi ")', want=False), # Values equal at nano resolution compare equal (resource.Quantity.Cmp # rounds to nano). - Case( + QuantityCase( name="nano rounding equality", expr='quantity("0.0000000004").compareTo(quantity("0.000000001")) == 0', want=True, ), # doc-comment isQuantity examples. - Case(name="isQuantity 1.3G", expr='isQuantity("1.3G")', want=True), - Case(name="isQuantity 1.3Gi", expr='isQuantity("1.3Gi")', want=True), - Case(name="isQuantity comma", expr='isQuantity("1,3G")', want=False), - Case(name="isQuantity 10000k", expr='isQuantity("10000k")', want=True), - Case(name="isQuantity capital K", expr='isQuantity("200K")', want=False), - Case(name="isQuantity Three", expr='isQuantity("Three")', want=False), - Case(name="isQuantity bare suffix", expr='isQuantity("Mi")', want=False), + QuantityCase(name="isQuantity 1.3G", expr='isQuantity("1.3G")', want=True), + QuantityCase(name="isQuantity 1.3Gi", expr='isQuantity("1.3Gi")', want=True), + QuantityCase(name="isQuantity comma", expr='isQuantity("1,3G")', want=False), + QuantityCase(name="isQuantity 10000k", expr='isQuantity("10000k")', want=True), + QuantityCase(name="isQuantity capital K", expr='isQuantity("200K")', want=False), + QuantityCase(name="isQuantity Three", expr='isQuantity("Three")', want=False), + QuantityCase(name="isQuantity bare suffix", expr='isQuantity("Mi")', want=False), # equality. - Case(name="equality reflexivity", expr='quantity("200M") == quantity("200M")', want=True), - Case( + QuantityCase(name="equality reflexivity", expr='quantity("200M") == quantity("200M")', want=True), + QuantityCase( name="equality symmetry", expr='quantity("200M") == quantity("0.2G") && quantity("0.2G") == quantity("200M")', want=True, ), - Case( + QuantityCase( name="equality transitivity", expr=( 'quantity("2M") == quantity("0.002G") && quantity("2000k") == quantity("2M") && ' @@ -108,29 +109,29 @@ class ParseErrCase: ), want=True, ), - Case(name="inequality", expr='quantity("200M") == quantity("0.3G")', want=False), + QuantityCase(name="inequality", expr='quantity("200M") == quantity("0.3G")', want=False), # isLessThan / isGreaterThan. - Case(name="less", expr='quantity("50M").isLessThan(quantity("50Mi"))', want=True), - Case(name="less obvious", expr='quantity("50M").isLessThan(quantity("100M"))', want=True), - Case(name="less false", expr='quantity("100M").isLessThan(quantity("50M"))', want=False), - Case(name="greater", expr='quantity("50Mi").isGreaterThan(quantity("50M"))', want=True), - Case(name="greater obvious", expr='quantity("150Mi").isGreaterThan(quantity("100Mi"))', want=True), - Case(name="greater false", expr='quantity("50M").isGreaterThan(quantity("100M"))', want=False), + QuantityCase(name="less", expr='quantity("50M").isLessThan(quantity("50Mi"))', want=True), + QuantityCase(name="less obvious", expr='quantity("50M").isLessThan(quantity("100M"))', want=True), + QuantityCase(name="less false", expr='quantity("100M").isLessThan(quantity("50M"))', want=False), + QuantityCase(name="greater", expr='quantity("50Mi").isGreaterThan(quantity("50M"))', want=True), + QuantityCase(name="greater obvious", expr='quantity("150Mi").isGreaterThan(quantity("100Mi"))', want=True), + QuantityCase(name="greater false", expr='quantity("50M").isGreaterThan(quantity("100M"))', want=False), # compareTo. - Case(name="compare equal", expr='quantity("200M").compareTo(quantity("0.2G")) == 0', want=True), - Case(name="compare less", expr='quantity("50M").compareTo(quantity("50Mi")) == -1', want=True), - Case(name="compare greater", expr='quantity("50Mi").compareTo(quantity("50M")) == 1', want=True), + QuantityCase(name="compare equal", expr='quantity("200M").compareTo(quantity("0.2G")) == 0', want=True), + QuantityCase(name="compare less", expr='quantity("50M").compareTo(quantity("50Mi")) == -1', want=True), + QuantityCase(name="compare greater", expr='quantity("50Mi").compareTo(quantity("50M")) == 1', want=True), # add / sub (quantity and int overloads). - Case(name="add quantity", expr='quantity("50k").add(quantity("20")) == quantity("50.02k")', want=True), - Case(name="add int not less", expr='quantity("50k").add(20).isLessThan(quantity("50020"))', want=False), - Case(name="sub quantity", expr='quantity("50k").sub(quantity("20")) == quantity("49.98k")', want=True), - Case(name="sub int", expr='quantity("50k").sub(20) == quantity("49980")', want=True), - Case( + QuantityCase(name="add quantity", expr='quantity("50k").add(quantity("20")) == quantity("50.02k")', want=True), + QuantityCase(name="add int not less", expr='quantity("50k").add(20).isLessThan(quantity("50020"))', want=False), + QuantityCase(name="sub quantity", expr='quantity("50k").sub(quantity("20")) == quantity("49.98k")', want=True), + QuantityCase(name="sub int", expr='quantity("50k").sub(20) == quantity("49980")', want=True), + QuantityCase( name="arith chain 1", expr='quantity("50k").add(20).sub(quantity("100k")).asInteger() == -49980', want=True, ), - Case( + QuantityCase( name="arith chain 2", expr='quantity("50k").add(20).sub(quantity("100k")).sub(-50000).asInteger() == 20', want=True, @@ -140,54 +141,56 @@ class ParseErrCase: # surface. celpy can't tell the two call styles apart, so we accept # both, but the test asserts the upstream-correct global form (see # cel.py's documented divergences). - Case(name="sign positive", expr='sign(quantity("50k")) == 1', want=True), - Case(name="sign negative", expr='sign(quantity("-50k")) == -1', want=True), - Case(name="sign zero", expr='sign(quantity("0")) == 0', want=True), + QuantityCase(name="sign positive", expr='sign(quantity("50k")) == 1', want=True), + QuantityCase(name="sign negative", expr='sign(quantity("-50k")) == -1', want=True), + QuantityCase(name="sign zero", expr='sign(quantity("0")) == 0', want=True), # Binary-suffix overflow saturates to int64-max, keeping sign, so # 8Ei/10Ei/100Ei all compare equal to int64-max (resource.Quantity # stores BinarySI in an int64). Confirmed against resource.Quantity. - Case( + QuantityCase( name="Ei saturates to int64 max", expr='quantity("8Ei").compareTo(quantity("9223372036854775807")) == 0', want=True, ), - Case(name="8Ei equals 10Ei", expr='quantity("8Ei").compareTo(quantity("10Ei")) == 0', want=True), - Case(name="10Ei equals 100Ei", expr='quantity("10Ei").compareTo(quantity("100Ei")) == 0', want=True), - Case( + QuantityCase(name="8Ei equals 10Ei", expr='quantity("8Ei").compareTo(quantity("10Ei")) == 0', want=True), + QuantityCase(name="10Ei equals 100Ei", expr='quantity("10Ei").compareTo(quantity("100Ei")) == 0', want=True), + QuantityCase( name="negative Ei saturates", expr='quantity("-10Ei").compareTo(quantity("-9223372036854775807")) == 0', want=True, ), # 7Ei is below int64-max, so it does NOT saturate and stays less. - Case(name="7Ei below saturation", expr='quantity("7Ei").isLessThan(quantity("8Ei"))', want=True), + QuantityCase(name="7Ei below saturation", expr='quantity("7Ei").isLessThan(quantity("8Ei"))', want=True), # Large DECIMAL-path values do not saturate (only the binary path # does) and must not raise on nano-rounding. isQuantity must be true # and the value must round-trip. - Case(name="isQuantity 256E", expr='isQuantity("256E")', want=True), - Case(name="isQuantity 10E", expr='isQuantity("10E")', want=True), - Case(name="256E value", expr='quantity("256E").compareTo(quantity("256000000000000000000")) == 0', want=True), - Case(name="256E greater than 1Ei", expr='quantity("256E").isGreaterThan(quantity("1Ei"))', want=True), + QuantityCase(name="isQuantity 256E", expr='isQuantity("256E")', want=True), + QuantityCase(name="isQuantity 10E", expr='isQuantity("10E")', want=True), + QuantityCase( + name="256E value", expr='quantity("256E").compareTo(quantity("256000000000000000000")) == 0', want=True + ), + QuantityCase(name="256E greater than 1Ei", expr='quantity("256E").isGreaterThan(quantity("1Ei"))', want=True), # asInteger / isInteger. - Case(name="as integer", expr='quantity("50k").asInteger() == 50000', want=True), - Case(name="is integer true small", expr='quantity("50").isInteger()', want=True), - Case(name="is integer true big magnitude", expr='quantity("50000000G").isInteger()', want=True), - Case( + QuantityCase(name="as integer", expr='quantity("50k").asInteger() == 50000', want=True), + QuantityCase(name="is integer true small", expr='quantity("50").isInteger()', want=True), + QuantityCase(name="is integer true big magnitude", expr='quantity("50000000G").isInteger()', want=True), + QuantityCase( name="is integer false overflow", expr='quantity("9999999999999999999999999999999999999G").isInteger()', want=False, ), # asInteger overflow is a runtime error upstream -> non-match here. - Case( + QuantityCase( name="as integer overflow is non-match", expr='quantity("9999999999999999999999999999999999999G").asInteger() > 0', want=False, ), # asApproximateFloat. - Case(name="as approximate float", expr='quantity("50.703k").asApproximateFloat() == 50703.0', want=True), + QuantityCase(name="as approximate float", expr='quantity("50.703k").asApproximateFloat() == 50703.0', want=True), # An invalid suffix is a runtime error upstream -> non-match here. # (Uses a member method upstream accepts, isGreaterThan, so the # non-match is the parse failure, not a rejected call form.) - Case( + QuantityCase( name="invalid suffix is non-match", expr='quantity("10Mo").isGreaterThan(quantity("1"))', want=False, @@ -196,28 +199,28 @@ class ParseErrCase: @pytest.mark.parametrize("case", QUANTITY_CASES, ids=lambda case: case.name) -def test_quantity(case: Case) -> None: +def test_quantity(case: QuantityCase) -> None: """A quantity CEL expression evaluates as it does upstream.""" - assert _eval(case.expr) == case.want + got = cel.Program(case.expr).matches({}) + assert got == case.want -# The bare-suffix row ("Mi") is a DELIBERATE divergence, not parity: upstream -# parses most bare suffixes as 0 but inconsistently errors on a few (see -# parse()'s docstring). We reject every bare suffix; no device capacity is -# ever a bare suffix. PARSE_REJECTS_CASES = [ - ParseErrCase(name="invalid suffix Mo", input="10Mo"), - ParseErrCase(name="passing regex bad suffix Mm", input="10Mm"), - ParseErrCase(name="capital K", input="200K"), - ParseErrCase(name="comma", input="1,3G"), - ParseErrCase(name="word", input="Three"), - ParseErrCase(name="bare suffix (deliberate divergence)", input="Mi"), - ParseErrCase(name="empty", input=""), + ParseRejectsCase(name="invalid suffix Mo", s="10Mo", want="invalid quantity: '10Mo'"), + ParseRejectsCase(name="passing regex bad suffix Mm", s="10Mm", want="invalid quantity: '10Mm'"), + ParseRejectsCase(name="capital K", s="200K", want="invalid quantity: '200K'"), + ParseRejectsCase(name="comma", s="1,3G", want="invalid quantity: '1,3G'"), + ParseRejectsCase(name="word", s="Three", want="invalid quantity: 'Three'"), + # A DELIBERATE divergence, not parity: upstream parses most bare suffixes + # as 0 but inconsistently errors on a few (see parse()'s docstring). We + # reject every bare suffix; no device capacity is ever a bare suffix. + ParseRejectsCase(name="bare suffix (deliberate divergence)", s="Mi", want="invalid quantity: 'Mi'"), + ParseRejectsCase(name="empty", s="", want="invalid quantity: ''"), ] @pytest.mark.parametrize("case", PARSE_REJECTS_CASES, ids=lambda case: case.name) -def test_parse_rejects(case: ParseErrCase) -> None: +def test_parse_rejects(case: ParseRejectsCase) -> None: """parse() rejects what resource.Quantity rejects, which drives the non-matches above.""" - with pytest.raises(ValueError, match="invalid quantity"): - quantity.parse(case.input) + with pytest.raises(ValueError, match=case.want): + quantity.parse(case.s) diff --git a/functions/compose-model-deployment/tests/test_scheduling.py b/functions/compose-model-deployment/tests/test_scheduling.py index a584c9c9f..bd9b960fb 100644 --- a/functions/compose-model-deployment/tests/test_scheduling.py +++ b/functions/compose-model-deployment/tests/test_scheduling.py @@ -25,6 +25,7 @@ import dataclasses import datetime +from typing import Literal import pytest from function import cel, scheduling @@ -33,17 +34,16 @@ from models.ai.modelplane.modelreplica import v1alpha1 as mrv1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -# A fixed transition time keeps observed conditions deterministic. -_TRANSITION_TIME = datetime.datetime(2025, 1, 1, tzinfo=datetime.UTC) - -# A GPU memory selector reused across cases. +# The selectors the cases' device requests use. They're named because which one +# a case uses is often what it tests, such as _MEM_200 against _MEM_LT_200, and +# an 80-character literal would bury that. A request's selectors reach the +# Candidate unchanged, so the expected DeviceRequests use them too. _MEM_141 = 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("141Gi")) >= 0' _MEM_200 = 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("200Gi")) >= 0' _MEM_LT_200 = 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("200Gi")) < 0' _IB = 'device.attributes["nic.nvidia.com"].linkType == "infiniband"' -# Default engine name used by the single-engine helpers below. -_ENGINE = "main" +_Role = Literal["Standalone", "Leader", "Worker"] @dataclasses.dataclass @@ -54,161 +54,76 @@ class Case: deployment: mdv1alpha1.ModelDeployment clusters: list[icv1alpha1.InferenceCluster] all_replicas: list[mrv1alpha1.ModelReplica] + fill: bool want: list[scheduling.Candidate] -def _request(name: str = "gpu", count: int = 1, cel_exprs: list[str] | None = None) -> mdv1alpha1.Device: - """A nodeSelector device request.""" - return mdv1alpha1.Device( - name=name, - count=count, - selectors=[mdv1alpha1.Selector(cel=c) for c in (cel_exprs or [_MEM_141])], - ) - - -def _template() -> mdv1alpha1.Template: - return mdv1alpha1.Template( - spec=mdv1alpha1.Spec( - containers=[mdv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")], +def _member(*, role: _Role, worker_nodes: int | None, devices: list[mdv1alpha1.Device] | None) -> mdv1alpha1.Member: + """A ModelDeployment member running vLLM, which claims nothing if devices is None.""" + return mdv1alpha1.Member( + role=role, + worker=mdv1alpha1.Worker(nodes=worker_nodes) if worker_nodes is not None else None, + nodeSelector=mdv1alpha1.NodeSelector(devices=devices) if devices is not None else None, + template=mdv1alpha1.Template( + spec=mdv1alpha1.Spec(containers=[mdv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")]), ), ) -def _node_selector(requests: list[mdv1alpha1.Device] | None) -> mdv1alpha1.NodeSelector: - return mdv1alpha1.NodeSelector(devices=requests if requests is not None else [_request()]) - - -def _engine( - name: str = _ENGINE, - *, - copies: int = 1, - pipeline: int = 1, - requests: list[mdv1alpha1.Device] | None = None, -) -> mdv1alpha1.Engine: - """An engine of homogeneous members. - - pipeline == 1 is a single Standalone member; pipeline > 1 is a Leader plus a - Worker spanning (pipeline - 1) nodes, so the engine spans `pipeline` nodes. - Engine copies multiply that, so node cost is pipeline * copies. Every member - carries the same nodeSelector; heterogeneous-member cases build their - members directly. - """ - if pipeline == 1: - members = [mdv1alpha1.Member(role="Standalone", nodeSelector=_node_selector(requests), template=_template())] - else: - members = [ - mdv1alpha1.Member(role="Leader", nodeSelector=_node_selector(requests), template=_template()), - mdv1alpha1.Member( - role="Worker", - worker=mdv1alpha1.Worker(nodes=pipeline - 1), - nodeSelector=_node_selector(requests), - template=_template(), - ), - ] - return mdv1alpha1.Engine(name=name, copies=copies, members=members) - - def _deployment( - name: str = "my-model", - replicas: int = 1, - pipeline: int = 1, - count: int = 1, - requests: list[mdv1alpha1.Device] | None = None, - engines: list[mdv1alpha1.Engine] | None = None, - tolerations: list[mdv1alpha1.Toleration] | None = None, + *, + replicas: int, + members: list[mdv1alpha1.Member], + tolerations: list[mdv1alpha1.Toleration] | None, ) -> mdv1alpha1.ModelDeployment: - """Construct a ModelDeployment. - - The single-engine helpers map a node shape onto one engine: pipeline sets the - engine's node span (a Standalone, or a Leader plus a Worker) and count sets - the engine's copies, so node cost is pipeline * count. Multi-engine cases - pass `engines` directly. - """ - if engines is None: - engines = [_engine(copies=count, pipeline=pipeline, requests=requests)] + """The ModelDeployment my-model, whose one engine, main, has the given members.""" return mdv1alpha1.ModelDeployment( - metadata=metav1.ObjectMeta(name=name, namespace="ml-team"), + metadata=metav1.ObjectMeta(name="my-model", namespace="ml-team"), spec=mdv1alpha1.SpecModel1( replicas=replicas, template=mdv1alpha1.TemplateModel( - spec=mdv1alpha1.SpecModel(engines=engines, tolerations=tolerations), + spec=mdv1alpha1.SpecModel( + engines=[mdv1alpha1.Engine(name="main", copies=1, members=members)], + tolerations=tolerations, + ), ), ), ) -def _gpu_device( - name: str = "gpu", - *, - claim: str = "DRA", - driver: str = "gpu.nvidia.com", - device_class: str = "gpu.nvidia.com", - count: int = 1, - memory: str = "141Gi", -) -> dict: - """A GPU device dict for a pool, with memory capacity.""" - d = { - "name": name, - "claim": claim, - "driver": driver, - "count": count, - "capacity": {"memory": {"value": memory}}, - } - if claim == "DRA": - d["deviceClassName"] = device_class - return d - - -def _nic_device(*, link_type: str = "infiniband", count: int = 1) -> dict: - """A synthetic NIC device dict for a pool.""" - return { - "name": "nic", - "claim": "Synthetic", - "driver": "nic.nvidia.com", - "count": count, - "attributes": {"linkType": {"string": link_type}}, - } +def _gpu_device(*, name: str, claim: Literal["DRA", "Synthetic"], count: int, memory: str) -> icv1alpha1.Device: + """A gpu.nvidia.com GPU in a pool. Only a DRA device has a device class to claim it by.""" + return icv1alpha1.Device( + name=name, + claim=claim, + driver="gpu.nvidia.com", + deviceClassName="gpu.nvidia.com" if claim == "DRA" else None, + count=count, + capacity={"memory": icv1alpha1.Capacity(value=memory)}, + ) -def _pool(name: str, *, nodes: int = 2, devices: list[dict] | None = None) -> dict: - """A pool with devices, for nodeSelector tests.""" - return { - "name": name, - "nodes": nodes, - "devices": devices if devices is not None else [_gpu_device()], - } +def _nic_device(*, link_type: str) -> icv1alpha1.Device: + """A Synthetic NIC in a pool, which a nodeSelector can match but nothing claims.""" + return icv1alpha1.Device( + name="nic", + claim="Synthetic", + driver="nic.nvidia.com", + count=1, + attributes={"linkType": icv1alpha1.Attributes(string=link_type)}, + ) def _cluster( - name: str, *, - placement_labels: dict[str, str] | None = None, - ready: bool = True, - gateway_hostname: str = "cluster-a.clusters.example.com", - pools: list[dict] | None = None, - taints: list[icv1alpha1.Taint] | None = None, + name: str, + gateway_hostname: str | None, + ready: bool, + pools: list[icv1alpha1.GpuPool], + taints: list[icv1alpha1.Taint] | None, + placement_labels: dict[str, str] | None, ) -> icv1alpha1.InferenceCluster: - """Construct an InferenceCluster with the given readiness and pools. - - A "ready" cluster has a Ready=True condition and a gateway hostname. An - address alone is not enough: an InferenceGateway addresses a cluster by name. - Setting ready=False or gateway_hostname="" produces a degraded cluster - the scheduler will retain but not pick anew. - """ - if pools is None: - pools = [{"name": "default", "nodes": 2, "devices": [_gpu_device()]}] - - status = "True" if ready else "False" - reason = "Available" if ready else "Unavailable" - conditions = [ - icv1alpha1.Condition( - type="Ready", - status=status, - reason=reason, - lastTransitionTime=_TRANSITION_TIME, - ) - ] - + """An InferenceCluster with the given readiness, gateway hostname, GPU pools, taints and placement labels.""" return icv1alpha1.InferenceCluster( metadata=metav1.ObjectMeta(name=name), spec=icv1alpha1.Spec( @@ -219,1563 +134,5266 @@ def _cluster( taints=taints, placement=( icv1alpha1.Placement(metadata=icv1alpha1.Metadata(labels=placement_labels)) - if placement_labels + if placement_labels is not None else None ), ), status=icv1alpha1.Status( - conditions=conditions, - gateway=( - icv1alpha1.Gateway(address="10.0.0.1", hostname=gateway_hostname) - if gateway_hostname - else icv1alpha1.Gateway(address="10.0.0.1") - ), + conditions=[ + icv1alpha1.Condition( + type="Ready", + status="True" if ready else "False", + reason="Available" if ready else "Unavailable", + lastTransitionTime=datetime.datetime(2025, 1, 1, tzinfo=datetime.UTC), + ), + ], + # The scheduler needs a hostname, not just an address, because an + # InferenceGateway addresses a cluster by name. + gateway=icv1alpha1.Gateway(address="10.0.0.1", hostname=gateway_hostname), providerConfigRef=icv1alpha1.ProviderConfigRef(name=name), - gpuPools=[icv1alpha1.GpuPool(**p) for p in pools], + gpuPools=pools, ), ) -def _replica_device_requests() -> list[mrv1alpha1.DeviceRequest]: - return [ - mrv1alpha1.DeviceRequest( - name="gpu", - deviceClassName="gpu.nvidia.com", - count=1, - selectors=[mrv1alpha1.Selector(cel=_MEM_141)], - ), - ] - - -def _replica_engine( - name: str = _ENGINE, +def _replica_member( *, - pool: str = "default", - copies: int = 1, - pipeline: int = 1, -) -> mrv1alpha1.Engine: - """One engine of an observed ModelReplica, with per-member pool pins and resolved requests.""" - template = mrv1alpha1.Template( - spec=mrv1alpha1.Spec(containers=[mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")]), + role: _Role, + worker_nodes: int | None, + pool: str, + device_requests: list[mrv1alpha1.DeviceRequest] | None, +) -> mrv1alpha1.Member: + """A ModelReplica member running vLLM, pinned to pool, with its device requests resolved.""" + return mrv1alpha1.Member( + role=role, + worker=mrv1alpha1.Worker(nodes=worker_nodes) if worker_nodes is not None else None, + nodePoolName=pool, + deviceRequests=device_requests, + template=mrv1alpha1.Template( + spec=mrv1alpha1.Spec(containers=[mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")]), + ), ) - if pipeline == 1: - members = [ - mrv1alpha1.Member( - role="Standalone", nodePoolName=pool, deviceRequests=_replica_device_requests(), template=template - ) - ] - else: - members = [ - mrv1alpha1.Member( - role="Leader", nodePoolName=pool, deviceRequests=_replica_device_requests(), template=template - ), - mrv1alpha1.Member( - role="Worker", - worker=mrv1alpha1.Worker(nodes=pipeline - 1), - nodePoolName=pool, - deviceRequests=_replica_device_requests(), - template=template, - ), - ] - return mrv1alpha1.Engine(name=name, copies=copies, members=members) def _replica( - deployment_name: str, - cluster_name: str, - *, - pool: str = "default", - index: int = 0, - pipeline: int = 1, - count: int = 1, - engines: list[mrv1alpha1.Engine] | None = None, + *, name: str, deployment: str, cluster: str, index: int, members: list[mrv1alpha1.Member] ) -> mrv1alpha1.ModelReplica: - """Construct an observed ModelReplica pinned to a (cluster, index). - - Mirrors _deployment's single-engine mapping: pipeline sets the engine's node - span and count its copies, so node cost is pipeline * count. - """ - if engines is None: - engines = [_replica_engine(pool=pool, copies=count, pipeline=pipeline)] + """An observed ModelReplica of deployment, pinned to (cluster, index), whose one engine, main, has members.""" return mrv1alpha1.ModelReplica( metadata=metav1.ObjectMeta( - name=f"{deployment_name}-{cluster_name}-{index}", + name=name, namespace="ml-team", labels={ - "modelplane.ai/deployment": deployment_name, - "modelplane.ai/cluster": cluster_name, + "modelplane.ai/deployment": deployment, + "modelplane.ai/cluster": cluster, "modelplane.ai/replica-index": str(index), }, ), - spec=mrv1alpha1.SpecModel(clusterName=cluster_name, engines=engines), - ) - - -def _replica_with_pool( - deployment_name: str, - cluster_name: str, - *, - pool: str, - index: int = 0, - pipeline: int = 1, - count: int = 1, -) -> mrv1alpha1.ModelReplica: - """An observed ModelReplica pinned to a cluster AND a specific node pool.""" - return _replica(deployment_name, cluster_name, pool=pool, index=index, pipeline=pipeline, count=count) - - -def _collision_replica( - object_name: str, - cluster_name: str, - *, - pool: str, - index: int = 0, -) -> mrv1alpha1.ModelReplica: - """A my-model replica with an explicit object name, to force an identity clash. - - Carries the my-model deployment label and a chosen (cluster, index) so it - collides with a normal my-model replica, while its distinct metadata.name is - the tiebreak the scheduler sorts on. - """ - r = _replica_with_pool("my-model", cluster_name, pool=pool, index=index) - assert r.metadata is not None - r.metadata.name = object_name - return r - - -# Convenience: the resolved DeviceRequest for a default GPU request matching a -# default pool, used in expected candidates for nodeSelector cases. -def _resolved(name: str = "gpu", count: int = 1, cel_exprs: list[str] | None = None) -> scheduling.DeviceRequest: - return scheduling.DeviceRequest( - name=name, - device_class_name="gpu.nvidia.com", - count=count, - cel_selectors=cel_exprs or [_MEM_141], + spec=mrv1alpha1.SpecModel( + clusterName=cluster, engines=[mrv1alpha1.Engine(name="main", copies=1, members=members)] + ), ) -def _placement( +def _candidate( *, - name: str = _ENGINE, - pool: str = "default", - device_requests: list[scheduling.DeviceRequest] | None = None, - pipeline: int = 1, -) -> scheduling.EnginePlacement: - """An expected EnginePlacement: one member placement per member. - - Mirrors _engine's member shape: pipeline == 1 is a single Standalone, - pipeline > 1 a Leader plus a Worker, all on the same pool with the same - resolved requests. - """ - dr = device_requests if device_requests is not None else [_resolved()] - if pipeline == 1: - members = [scheduling.MemberPlacement(role="Standalone", pool=pool, device_requests=dr)] - else: - members = [ - scheduling.MemberPlacement(role="Leader", pool=pool, device_requests=dr), - scheduling.MemberPlacement(role="Worker", pool=pool, device_requests=dr), - ] - return scheduling.EnginePlacement(name=name, members=members) - - -# Convenience: build an expected Candidate defaulting to index 0, so the many -# single-replica-per-cluster cases stay terse. A placed or retained replica -# resolves to one engine on the default pool with the default GPU request; a -# degraded/unplaced cluster carries no gateway. Cases that need a specific pool, -# request, or engine layout pass `engines` explicitly. -def _cand( name: str, - *, - index: int = 0, - pool: str = "default", - device_requests: list[scheduling.DeviceRequest] | None = None, - pipeline: int = 1, - engines: list[scheduling.EnginePlacement] | None = None, - gateway_hostname: str = "", - placement_labels: dict[str, str] | None = None, + index: int, + gateway_hostname: str, + placement_labels: dict[str, str], + members: list[scheduling.MemberPlacement], ) -> scheduling.Candidate: - if engines is None: - engines = [_placement(pool=pool, device_requests=device_requests, pipeline=pipeline)] + """An expected Candidate for (name, index), whose one engine, main, places members.""" return scheduling.Candidate( name=name, index=index, - engines=engines, gateway_hostname=gateway_hostname, - placement_labels=placement_labels or {}, + placement_labels=placement_labels, + engines=[scheduling.EnginePlacement(name="main", members=members)], ) -# Deployments use the default single-GPU nodeSelector request (any pool's GPU -# device satisfies it), so these focus on placement rather than pool matching; -# NODE_SELECTOR_CASES covers request-to-device matching. SCHEDULE_CASES = [ + # Placement: retain, spread, scale and capacity. Every member requests one + # GPU matching _MEM_141, which every pool's 141Gi GPU satisfies, so these + # cases focus on placement rather than pool matching. Case( name="no clusters returns no candidates", - deployment=_deployment(), + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[], all_replicas=[], + fill=True, want=[], ), Case( name="single ready cluster is picked", - deployment=_deployment(), - clusters=[_cluster("cluster-a")], - all_replicas=[], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default")], - ), - Case( - name="not-ready cluster is not picked for a new replica", - deployment=_deployment(), - clusters=[_cluster("cluster-a", ready=False)], - all_replicas=[], - want=[], - ), - Case( - name="cluster without gateway address is not picked", - deployment=_deployment(), - clusters=[_cluster("cluster-a", gateway_hostname="")], - all_replicas=[], - want=[], - ), - Case( - name="multi-node deployment needs enough nodes", - deployment=_deployment(pipeline=4), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], - all_replicas=[], - want=[], - ), - Case( - name="existing replica is retained on its pinned cluster", - deployment=_deployment(), - clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - # cluster-a wins even though cluster-b is also viable. The pin - # still matches, so it's retained with its resolved pool/requests. - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="degraded pinned cluster is retained with empty gateway", - deployment=_deployment(), - clusters=[_cluster("cluster-a", ready=False, gateway_hostname="")], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="")], - ), - Case( - name="deleted pinned cluster triggers re-placement", - deployment=_deployment(), - clusters=[_cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com")], - all_replicas=[_replica("my-model", "cluster-a")], - want=[_cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default")], - ), - Case( - name="scale up places new replicas on additional clusters", - deployment=_deployment(replicas=2), + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ], - all_replicas=[_replica("my-model", "cluster-a")], - want=[ - _cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com"), - _cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default"), + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) ], - ), - Case( - name="scale up with no extra capacity returns only retained", - deployment=_deployment(replicas=2), - # Single-node pool, already filled by the retained replica, so no - # second replica can be placed - not even on the same cluster. - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default")], - ), - Case( - name="two replicas pack onto one cluster when it is the only option", - deployment=_deployment(replicas=2), - # One cluster, a 2-node pool, two 1-node replicas. With nowhere - # to spread, both pack onto cluster-a at indices 0 and 1. - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], all_replicas=[], + fill=True, want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), Case( - name="two replicas spread across two clusters before packing", - deployment=_deployment(replicas=2), - # Both clusters can hold two replicas, but we prefer one each. + name="not-ready cluster is not picked for a new replica", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=2)]), _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=2)], - ), + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=False, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) ], all_replicas=[], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], + fill=True, + want=[], ), Case( - name="three replicas spread first then pack the remainder", - deployment=_deployment(replicas=3), - # Two clusters, plenty of room. Spread gives a, b one each, then - # the third lands back on cluster-a (lowest load, name tiebreak). + name="cluster without a gateway hostname is not picked", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], - ), + name="cluster-a", + gateway_hostname=None, + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) ], all_replicas=[], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], + fill=True, + want=[], ), Case( - name="capacity forces packing past the spread preference", - deployment=_deployment(replicas=3), - # cluster-b holds one replica; cluster-a has room for the rest. - # Spread puts one on each, then the third can't fit on b (full), - # so it packs onto a. + name="multi-node deployment needs enough nodes", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + _member( + role="Worker", + worker_nodes=3, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=1)], - ), + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) ], all_replicas=[], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], + fill=True, + want=[], ), + # cluster-a wins even though cluster-b is also viable. The pin + # still matches, so it's retained with its resolved pool/requests. Case( - name="new replica spreads onto an empty cluster before doubling up", - deployment=_deployment(replicas=2), - # cluster-a already hosts a replica; cluster-b is empty. The new - # replica prefers empty cluster-b over packing onto a. + name="existing replica is retained on its pinned cluster", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), _cluster( - "cluster-b", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, ), ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), - ], - ), - Case( - name="new replica takes the lowest free index on a packed cluster", - deployment=_deployment(replicas=3), - # Only cluster-a exists, already hosting indices 0 and 2 (1 was - # deleted). The new replica fills the gap at index 1. - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=2), + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) ], + fill=True, want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=2, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), Case( - name="scale down packs off by dropping the highest index first", - deployment=_deployment(replicas=2), - # cluster-a hosts indices 0 and 1; cluster-b hosts index 0. Three - # replicas, want two. Highest index (a/1) is dropped, keeping the - # spread across a/0 and b/0. + name="degraded pinned cluster is retained with empty gateway", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), _cluster( - "cluster-b", - gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], - ), + name="cluster-a", + gateway_hostname=None, + ready=False, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) ], all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=1), - _replica_with_pool("my-model", "cluster-b", pool="default", index=0), - ], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) ], - ), - Case( - name="retained replica is charged at its own node cost, not the new shape", - # The deployment's workers grew to pipeline=4 (4 nodes/replica), - # but the existing replica was created at pipeline=2 and is - # retained (no nodeSelector change rolls it). It still consumes - # only its original 2 nodes. The pool has 6, so a second replica - # at the new 4-node cost must still fit (6 - 2 = 4). Regression: - # charging the retained replica at the new shape (4) would leave - # 2 free and wrongly refuse the placement. - deployment=_deployment(replicas=2, pipeline=4), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=6)])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default", pipeline=2)], - # The retained replica is re-stamped to the deployment's current - # pipeline=4 shape but still charged its observed 2 nodes in the - # ledger. + fill=True, want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pipeline=4), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pipeline=4), + _candidate( + name="cluster-a", + index=0, + gateway_hostname="", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), Case( - name="scale down drops from the most-loaded cluster to preserve spread", - deployment=_deployment(replicas=2), - # cluster-a hosts two replicas, cluster-b one. Scaling 3->2 must - # drop a's extra (a/1), NOT b's sole replica - otherwise we'd - # leave a packed and b empty, the opposite of spread. b's index - # is 3 (higher than a/1) to prove we drop by cluster load, not by - # a global index comparison. + name="deleted pinned cluster triggers re-placement", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a", pools=[_pool("default", nodes=4)]), _cluster( - "cluster-b", + name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("default", nodes=4)], - ), - ], - all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=1), - _replica_with_pool("my-model", "cluster-b", pool="default", index=3), - ], - want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", index=3, gateway_hostname="cluster-b.clusters.example.com", pool="default"), + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) ], - ), - Case( - name="co-located replicas are both retained across a reconcile", - deployment=_deployment(replicas=2), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="default", index=0), - _replica_with_pool("my-model", "cluster-a", pool="default", index=1), + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) ], + fill=True, want=[ - _cand(name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-a", index=1, gateway_hostname="cluster-a.clusters.example.com", pool="default"), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), Case( - name="scale down across clusters drops higher cluster name at equal index", - deployment=_deployment(replicas=1), + name="scale up places new replicas on additional clusters", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # Single-node pool, already filled by the retained replica, so no + # second replica can be placed - not even on the same cluster. + Case( + name="scale up with no extra capacity returns only retained", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=1, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) ], all_replicas=[ - _replica("my-model", "cluster-b"), - _replica("my-model", "cluster-a"), + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # One cluster, a 2-node pool, two 1-node replicas. With nowhere + # to spread, both pack onto cluster-a at indices 0 and 1. + Case( + name="two replicas pack onto one cluster when it is the only option", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # Both clusters can hold two replicas, but we prefer one each. + Case( + name="two replicas spread across two clusters before packing", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # Two clusters, plenty of room. Spread gives a, b one each, then + # the third lands back on cluster-a (lowest load, name tiebreak). + Case( + name="three replicas spread first then pack the remainder", + deployment=_deployment( + replicas=3, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # cluster-b holds one replica; cluster-a has room for the rest. + # Spread puts one on each, then the third can't fit on b (full), + # so it packs onto a. + Case( + name="capacity forces packing past the spread preference", + deployment=_deployment( + replicas=3, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=1, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # cluster-a already hosts a replica; cluster-b is empty. The new + # replica prefers empty cluster-b over packing onto a. + Case( + name="new replica spreads onto an empty cluster before doubling up", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), ], - # Both at index 0, so the (index, name) tiebreak keeps cluster-a. - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # Only cluster-a exists, already hosting indices 0 and 2 (1 was + # deleted). The new replica fills the gap at index 1. + Case( + name="new replica takes the lowest free index on a packed cluster", + deployment=_deployment( + replicas=3, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-a-2", + deployment="my-model", + cluster="cluster-a", + index=2, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=2, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # cluster-a hosts indices 0 and 1; cluster-b hosts index 0. Three + # replicas, want two. Highest index (a/1) is dropped, keeping the + # spread across a/0 and b/0. + Case( + name="scale down packs off by dropping the highest index first", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-a-1", + deployment="my-model", + cluster="cluster-a", + index=1, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-b-0", + deployment="my-model", + cluster="cluster-b", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # The deployment's Worker grew to 3 nodes (4 nodes/replica), but + # the existing replica was created with a 1-node Worker (2 + # nodes/replica) and is retained (no nodeSelector change rolls it). + # It's re-stamped to the deployment's current 4-node shape, but still + # consumes only its original 2 nodes in the ledger. The pool has 6, so a + # second replica at the new 4-node cost must still fit (6 - 2 = 4). + # Regression: charging the retained replica at the new shape (4) + # would leave 2 free and wrongly refuse the placement. + Case( + name="retained replica is charged at its own node cost, not the new shape", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + _member( + role="Worker", + worker_nodes=3, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=6, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Leader", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ), + _replica_member( + role="Worker", + worker_nodes=1, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ), + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + scheduling.MemberPlacement( + role="Worker", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + scheduling.MemberPlacement( + role="Worker", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # cluster-a hosts two replicas, cluster-b one. Scaling 3->2 must + # drop a's extra (a/1), NOT b's sole replica - otherwise we'd + # leave a packed and b empty, the opposite of spread. b's index + # is 3 (higher than a/1) to prove we drop by cluster load, not by + # a global index comparison. + Case( + name="scale down drops from the most-loaded cluster to preserve spread", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-a-1", + deployment="my-model", + cluster="cluster-a", + index=1, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-b-3", + deployment="my-model", + cluster="cluster-b", + index=3, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=3, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="co-located replicas are both retained across a reconcile", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-a-1", + deployment="my-model", + cluster="cluster-a", + index=1, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # Both at index 0, so the (index, name) tiebreak keeps cluster-a. + Case( + name="scale down across clusters drops higher cluster name at equal index", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[ + _replica( + name="my-model-cluster-b-0", + deployment="my-model", + cluster="cluster-b", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="new placement is alphabetical for determinism", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-c", + gateway_hostname="cluster-c.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # other-model occupies the single node on cluster-a. + Case( + name="other deployment's replicas consume node capacity", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=1, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="other-model-cluster-a-0", + deployment="other-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[], + ), + # Retained on its pin: the single node it already occupies isn't + # charged against itself, so it stays rather than being evicted. + Case( + name="our own observed replicas don't double-count against us", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=1, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # other-model is pinned to pool "gone", which the cluster no + # longer publishes. Its pods are pinned to a node label no node + # carries, so they're unschedulable and occupy nothing. The one + # published node on "frontier" is therefore free for our replica. + # Charging the unattributable replica would wrongly report the + # cluster full. + Case( + name="another deployment pinned to a deleted pool consumes no capacity", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=1, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="other-model-cluster-a-0", + deployment="other-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="gone", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # Two of our replicas collide on (cluster-a, index 0) with + # different pinned pools. Retain keeps the first by replica name + # (my-model-cluster-a-0 on "a" sorts before the "-dup" replica on + # "b"), independent of input order, so the schedule is a function + # of state not of delivery order. Both pools match, so either + # would be a valid placement - only determinism is under test. + Case( + name="colliding (cluster, index) retains deterministically by replica name", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="a", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ), + icv1alpha1.GpuPool( + name="b", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ), + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0-dup", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="b", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="a", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="a", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # fill=False retains existing replicas but places no new ones. A caller + # passes fill=False when it can't yet trust the candidate set (for the + # ModelDeployment, when a referenced ModelCache is unresolved). Retain runs + # unconditionally; only the placement of new replicas is held. + Case( + name="no replicas yet: nothing is placed", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=False, + want=[], + ), + Case( + name="existing replica is retained despite fill=False", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=False, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="scale-up shortfall is not filled, only the retained replica remains", + deployment=_deployment( + replicas=3, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=False, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # nodeSelector device-request matching and pool pinning. + Case( + name="matching request picks the cluster and records the pool", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="non-matching request filters the cluster out", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_200)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], + ), + # Request 8 GPUs, pool device has only 4. + Case( + name="device count not covered filters out", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=8, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=4, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], + ), + # A pool device published with count 0 must read as "none + # available", not default to 1. Regression: `d.count or 1` + # treated 0 as 1 and placed a replica whose ResourceClaim no + # device could satisfy. The status schema permits 0 even though + # an InferenceClass device count is floored at 1. + Case( + name="published device count of zero satisfies no request", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=0, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], + ), + # An autoscaled-to-zero pool has a matching GPU device but no + # nodes, so it can host no replica. + Case( + name="published pool node count of zero hosts nothing", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=0, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], + ), + # Only the claim: DRA gpu request is resolved; the synthetic nic + # matched for scheduling but isn't claimed. + Case( + name="synthetic NIC device matches but is not in resolved requests", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="multi-device: missing NIC filters the pool out", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], + ), + # Two distinct requests, each matching the same single GPU + # device. DRA allocates distinct devices per request, so a + # count:1 device can satisfy only one. The pool must not match. + Case( + name="two requests cannot both claim one single-count device", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu-a", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="gpu-b", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], + ), + # Two count:5 requests need 10 GPUs total; the device has 8. + # Capacity is consumed across requests, so the pool must not + # match (regression: an earlier version checked each request + # against the full device count independently). + Case( + name="two requests against one device must fit within its count", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu-a", count=5, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="gpu-b", count=5, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=8, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], + ), + # 8-GPU device, two count:4 requests = 8 total. Both resolve. + Case( + name="two requests sharing a device fit when count covers both", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu-a", count=4, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="gpu-b", count=4, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=8, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu-a", + device_class_name="gpu.nvidia.com", + count=4, + cel_selectors=[_MEM_141], + ), + scheduling.DeviceRequest( + name="gpu-b", + device_class_name="gpu.nvidia.com", + count=4, + cel_selectors=[_MEM_141], + ), + ], + ), + ], + ), + ], + ), + # Both pools carry a claimable GPU; the synthetic NIC's link type + # is the discriminator. Only the infiniband pool satisfies the + # nic selector, so it's picked though it's listed second. + Case( + name="only the pool whose NIC matches the selector is picked", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="dev", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="gpudirect-tcpx"), + ], + ), + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ), + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + # The sole request matches a synthetic NIC. The replica's serving + # workload would have no ResourceClaim to bind GPUs through, so + # the pool is not a viable host and nothing is scheduled. + Case( + name="synthetic-only selector leaves nothing to claim, pool ineligible", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[], + fill=True, + want=[], ), Case( - name="new placement is alphabetical for determinism", - deployment=_deployment(replicas=2), + name="retained replica keeps its pinned pool", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-c", gateway_hostname="cluster-c.clusters.example.com"), - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) ], - all_replicas=[], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="frontier", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, want=[ - _cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", pool="default"), - _cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default"), + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), + # A claimable GPU keeps both pools viable hosts; the synthetic + # NIC's link type is the drifting discriminator. Case( - name="other deployment's replicas consume node capacity", - deployment=_deployment(pipeline=1), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], - # other-model occupies the single node on cluster-a. - all_replicas=[_replica("other-model", "cluster-a")], - want=[], - ), - Case( - name="our own observed replicas don't double-count against us", - deployment=_deployment(pipeline=1), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=1)])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - # Retained on its pin: the single node it already occupies isn't - # charged against itself, so it stays rather than being evicted. - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="replica labeled for our deployment but pinned to unknown cluster is ignored", - deployment=_deployment(), - clusters=[_cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com")], - all_replicas=[_replica("my-model", "cluster-a")], - want=[_cand(name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", pool="default")], - ), - Case( - name="another deployment pinned to a deleted pool consumes no capacity", - # other-model is pinned to pool "gone", which the cluster no - # longer publishes. Its pods are pinned to a node label no node - # carries, so they're unschedulable and occupy nothing. The one - # published node on "frontier" is therefore free for our replica. - # Charging the unattributable replica would wrongly report the - # cluster full. - deployment=_deployment(requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], - all_replicas=[_replica_with_pool("other-model", "cluster-a", pool="gone")], - want=[ - _cand( + name="selector drift re-places replica onto a now-matching pool", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="a", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="gpudirect-tcpx"), + ], + ), + icv1alpha1.GpuPool( + name="b", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ), + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="a", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], ) ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="b", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], ), Case( - name="colliding (cluster, index) retains deterministically by replica name", - # Two of our replicas collide on (cluster-a, index 0) with - # different pinned pools. Retain keeps the first by replica name - # (my-model-cluster-a-0 on "a" sorts before the "-dup" replica on - # "b"), independent of input order, so the schedule is a function - # of state not of delivery order. Both pools match, so either - # would be a valid placement - only determinism is under test. - deployment=_deployment(requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), + name="pinned pool that still matches stays pinned (attribute drift is sticky)", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)]), + ], + ) + ], + tolerations=None, + ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool("a", devices=[_gpu_device()]), - _pool("b", devices=[_gpu_device()]), + icv1alpha1.GpuPool( + name="a", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ), + icv1alpha1.GpuPool( + name="b", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ), ], + taints=None, + placement_labels=None, ) ], all_replicas=[ - _collision_replica("my-model-cluster-a-0-dup", "cluster-a", pool="b", index=0), - _replica_with_pool("my-model", "cluster-a", pool="a", index=0), + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="a", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) ], + fill=True, want=[ - _cand( + _candidate( name="cluster-a", + index=0, gateway_hostname="cluster-a.clusters.example.com", - pool="a", - device_requests=[_resolved()], - ) + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="a", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), -] - - -@pytest.mark.parametrize("case", SCHEDULE_CASES, ids=lambda case: case.name) -def test_schedule(case: Case) -> None: - """The scheduler retains existing pins and places new replicas.""" - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) - assert got == case.want - - -# A caller passes fill=False when it can't yet trust the candidate set (for the -# ModelDeployment, when a referenced ModelCache is unresolved). Retain runs -# unconditionally; only the placement of new replicas is held. -FILL_FALSE_CASES = [ Case( - name="no replicas yet: nothing is placed", - deployment=_deployment(), - clusters=[_cluster("cluster-a")], - all_replicas=[], + name="no matching pool anywhere drops the replica entirely", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)]), + ], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="a", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="gpudirect-tcpx"), + ], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="a", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, want=[], ), Case( - name="existing replica is retained despite fill=False", - deployment=_deployment(), - clusters=[_cluster("cluster-a")], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="scale-up shortfall is not filled, only the retained replica remains", - deployment=_deployment(replicas=3), + name="replica pinned to an unpublished pool is re-placed onto a matching one", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)]), + mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)]), + ], + ) + ], + tolerations=None, + ), clusters=[ - _cluster("cluster-a"), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ) + ], + taints=None, + placement_labels=None, + ) ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="default")], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), -] - - -@pytest.mark.parametrize("case", FILL_FALSE_CASES, ids=lambda case: case.name) -def test_fill_false_is_retain_only(case: Case) -> None: - """fill=False retains existing replicas but places no new ones.""" - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas, fill=False) - assert got == case.want - - -# nodeSelector device-request matching and pool pinning. -NODE_SELECTOR_CASES = [ - Case( - name="matching request picks the cluster and records the pool", - deployment=_deployment(requests=[_request(cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], - all_replicas=[], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, want=[ - _cand( + _candidate( name="cluster-a", + index=0, gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], - ) + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), + # a/0 is pinned to a pool that still matches (retained). a/1 is + # pinned to a pool no longer published, so it's dropped and will + # be re-placed. The pool has just 2 nodes; both are notionally in + # use by a/0 and a/1. The refill must see a/1's node freeing up + # (it's being deleted) and re-place onto frontier at index 1. + # Regression: the ledger must not charge dropped replicas. Case( - name="non-matching request filters the cluster out", - deployment=_deployment(requests=[_request(cel_exprs=[_MEM_200])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], - all_replicas=[], - want=[], - ), - Case( - name="device count not covered filters out", - # Request 8 GPUs, pool device has only 4. - deployment=_deployment(requests=[_request(count=8, cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=4)])])], - all_replicas=[], - want=[], - ), - Case( - name="published device count of zero satisfies no request", - # A pool device published with count 0 must read as "none - # available", not default to 1. Regression: `d.count or 1` - # treated 0 as 1 and placed a replica whose ResourceClaim no - # device could satisfy. The status schema permits 0 even though - # an InferenceClass device count is floored at 1. - deployment=_deployment(requests=[_request(count=1, cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=0)])])], - all_replicas=[], - want=[], + name="dropping a non-matching replica frees its node for the refill", + deployment=_deployment( + replicas=2, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="frontier", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + _replica( + name="my-model-cluster-a-1", + deployment="my-model", + cluster="cluster-a", + index=1, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="gone", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ), + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + _candidate( + name="cluster-a", + index=1, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], ), + # Request 8 GPUs. Pool 'a' has 4/node (doesn't fit); pool 'b' + # has 8 and does. The replica must pin to 'b'. Case( - name="published pool node count of zero hosts nothing", - # An autoscaled-to-zero pool has a matching GPU device but no - # nodes, so it can host no replica. - deployment=_deployment(requests=[_request(count=1, cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=0)])], + name="device count is checked against the pinned pool, not a cluster-wide sum", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=8, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="a", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=4, memory="141Gi")] + ), + icv1alpha1.GpuPool( + name="b", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=8, memory="141Gi")] + ), + ], + taints=None, + placement_labels=None, + ) + ], all_replicas=[], - want=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="b", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=8, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], ), + # Per-member placement: single-pool engines, rejection when no pool fits, + # and claimless ride-along members. + # + # The leader's request matches both pools; the worker's only + # matches big. The whole-engine pass must put both members on + # big - the one pool that satisfies them all. Case( - name="synthetic NIC device matches but is not in resolved requests", + name="a single pool satisfying every member hosts the whole engine", deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_200)])], + ), + ], + tolerations=None, ), clusters=[ _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device()])], + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="small", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ), + icv1alpha1.GpuPool( + name="big", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="200Gi")] + ), + ], + taints=None, + placement_labels=None, ) ], all_replicas=[], - # Only the claim: DRA gpu request is resolved; the synthetic nic - # matched for scheduling but isn't claimed. + fill=True, want=[ - _cand( + _candidate( name="cluster-a", + index=0, gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved(name="gpu")], - ) + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="big", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + scheduling.MemberPlacement( + role="Worker", + pool="big", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_200], + ) + ], + ), + ], + ), ], ), + # The leader only fits big (>= 200Gi); the worker only fits + # small (< 200Gi). No single pool satisfies both. The scheduler + # never splits an engine across pools - it can't tell whether + # big and small share a fabric - so the engine is rejected and + # the replica goes unplaced (#149). Case( - name="multi-device: missing NIC filters the pool out", + name="members no single pool satisfies are not scheduled", deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_200)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_LT_200)])], + ), + ], + tolerations=None, ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device()])])], + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="small", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ), + icv1alpha1.GpuPool( + name="big", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="200Gi")] + ), + ], + taints=None, + placement_labels=None, + ) + ], all_replicas=[], + fill=True, want=[], ), + # Both members match only big (>= 141Gi); small (40Gi) matches + # neither. big has one free node but the gang needs two. The + # engine doesn't fit any single pool, so it's rejected; with + # cluster-a the only cluster the replica goes unplaced. Case( - name="two requests cannot both claim one single-count device", - # Two distinct requests, each matching the same single GPU - # device. DRA allocates distinct devices per request, so a - # count:1 device can satisfy only one. The pool must not match. + name="a gang too big for its only matching pool is rejected", deployment=_deployment( - requests=[ - _request(name="gpu-a", cel_exprs=[_MEM_141]), - _request(name="gpu-b", cel_exprs=[_MEM_141]), - ] + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=1)])])], + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="small", nodes=8, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="40Gi")] + ), + icv1alpha1.GpuPool( + name="big", nodes=1, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ), + ], + taints=None, + placement_labels=None, + ) + ], all_replicas=[], + fill=True, want=[], ), + # Same gang. cluster-a's matching pool has only one free node + # (too few for the two-member gang), so the scheduler rejects + # cluster-a and places the whole gang on cluster-b, whose pool + # has room for both members. Case( - name="two requests against one device must fit within its count", - # Two count:5 requests need 10 GPUs total; the device has 8. - # Capacity is consumed across requests, so the pool must not - # match (regression: an earlier version checked each request - # against the full device count independently). + name="a gang too big for one cluster's pool lands whole on another", deployment=_deployment( - requests=[ - _request(name="gpu-a", count=5, cel_exprs=[_MEM_141]), - _request(name="gpu-b", count=5, cel_exprs=[_MEM_141]), - ] + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=8)])])], + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="small", nodes=8, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="40Gi")] + ), + icv1alpha1.GpuPool( + name="big", nodes=1, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ), + ], + taints=None, + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="big", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], all_replicas=[], - want=[], + fill=True, + want=[ + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="big", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + scheduling.MemberPlacement( + role="Worker", + pool="big", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], ), + # On pool 'a' the leader's request matches only a Synthetic + # device (nothing to claim) while the worker claims, so the + # whole engine *could* land there - but pool 'b' satisfies the + # leader claimably. The engine must go to 'b'; placing on 'a' + # would run the leader without the GPU it asked for. Case( - name="two requests sharing a device fit when count covers both", - # 8-GPU device, two count:4 requests = 8 total. Both resolve. + name="a member claimable elsewhere is not stranded on a synthetic match", deployment=_deployment( - requests=[ - _request(name="gpu-a", count=4, cel_exprs=[_MEM_141]), - _request(name="gpu-b", count=4, cel_exprs=[_MEM_141]), - ] + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_200)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, ), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", devices=[_gpu_device(count=8)])])], + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="a", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _gpu_device(name="syn", claim="Synthetic", count=1, memory="200Gi"), + ], + ), + icv1alpha1.GpuPool( + name="b", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="200Gi")] + ), + ], + taints=None, + placement_labels=None, + ) + ], all_replicas=[], + fill=True, want=[ - _cand( + _candidate( name="cluster-a", + index=0, gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[ - _resolved(name="gpu-a", count=4), - _resolved(name="gpu-b", count=4), + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="b", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_200], + ) + ], + ), + scheduling.MemberPlacement( + role="Worker", + pool="b", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), ], - ) + ), ], ), + # The leader's request matches only the pool's synthetic NIC on + # every pool - deliberate (a selector that pins without + # claiming). It places claimless alongside the claiming worker. Case( - name="first matching pool wins (deterministic)", - # Both pools carry a claimable GPU; the synthetic NIC's link type - # is the discriminator. Only the infiniband pool satisfies the - # nic selector, so it wins regardless of pool order. + name="a member synthetic-only everywhere places claimless with its gang", deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="nic", count=1, selectors=[mdv1alpha1.Selector(cel=_IB)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool("dev", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")]), - _pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[ + _gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi"), + _nic_device(link_type="infiniband"), + ], + ) ], + taints=None, + placement_labels=None, ) ], all_replicas=[], + fill=True, want=[ - _cand( + _candidate( name="cluster-a", + index=0, gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved(name="gpu")], - ) + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="frontier", + device_requests=[], + ), + scheduling.MemberPlacement( + role="Worker", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), Case( - name="synthetic-only selector leaves nothing to claim, pool ineligible", - # The sole request matches a synthetic NIC. The replica's serving - # workload would have no ResourceClaim to bind GPUs through, so - # the pool is not a viable host and nothing is scheduled. - deployment=_deployment(requests=[_request(name="nic", cel_exprs=[_IB])]), + name="a member that matches nowhere fails the whole replica", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_200)])], + ), + ], + tolerations=None, + ), clusters=[ _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")])], + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, ) ], all_replicas=[], + fill=True, want=[], ), + # The leader carries no nodeSelector: it claims nothing, follows + # the worker's pool, and costs no nodes - the 1-node pool fits + # the whole gang because only the worker occupies a node. Case( - name="retained replica keeps its pinned pool", - deployment=_deployment(requests=[_request(cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier")])], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="frontier")], - want=[ - _cand( + name="a claimless leader rides along on its gang's pool at zero cost", + deployment=_deployment( + replicas=1, + members=[ + _member(role="Leader", worker_nodes=None, devices=None), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, + ), + clusters=[ + _cluster( name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=1, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, ) ], + all_replicas=[], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="frontier", + device_requests=[], + ), + scheduling.MemberPlacement( + role="Worker", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], ), Case( - name="selector drift re-places replica onto a now-matching pool", - # A claimable GPU keeps both pools viable hosts; the synthetic - # NIC's link type is the drifting discriminator. + name="a retained replica's claimless member keeps its pin", deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] + replicas=1, + members=[ + _member(role="Leader", worker_nodes=None, devices=None), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool("a", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")]), - _pool("b", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), + icv1alpha1.GpuPool( + name="frontier", + nodes=1, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member(role="Leader", worker_nodes=None, pool="frontier", device_requests=None), + _replica_member( + role="Worker", + worker_nodes=1, + pool="frontier", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ), ], ) ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], + fill=True, want=[ - _cand( + _candidate( name="cluster-a", + index=0, gateway_hostname="cluster-a.clusters.example.com", - pool="b", - device_requests=[_resolved(name="gpu")], - ) + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="frontier", + device_requests=[], + ), + scheduling.MemberPlacement( + role="Worker", + pool="frontier", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), + # other-model's gang occupies only its worker's node: its + # claimless leader shares that node. The 2-node pool has 1 node + # free, so our 1-node deployment fits. Charging the claimless + # leader a node would wrongly report insufficient capacity. Case( - name="pinned pool that still matches stays pinned (attribute drift is sticky)", + name="another deployment's claimless member consumes no capacity", deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool("a", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), - _pool("b", devices=[_gpu_device(), _nic_device(link_type="infiniband")]), + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="other-model-cluster-a-0", + deployment="other-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member(role="Leader", worker_nodes=None, pool="default", device_requests=None), + _replica_member( + role="Worker", + worker_nodes=1, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ), ], ) ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], + fill=True, want=[ - _cand( + _candidate( name="cluster-a", + index=0, gateway_hostname="cluster-a.clusters.example.com", - pool="a", - device_requests=[_resolved(name="gpu")], - ) + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), + # The deployment grew a Worker (Standalone -> Leader+Worker). + # The observed single-member replica no longer lines up, so it + # is re-placed with the new shape. Case( - name="no matching pool anywhere drops the replica entirely", + name="a member shape change re-places the replica", deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] + replicas=1, + members=[ + _member( + role="Leader", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + _member( + role="Worker", + worker_nodes=1, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ), + ], + tolerations=None, ), clusters=[ _cluster( - "cluster-a", - pools=[_pool("a", devices=[_gpu_device(), _nic_device(link_type="gpudirect-tcpx")])], + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=4, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, ) ], - all_replicas=[_replica_with_pool("my-model", "cluster-a", pool="a")], - want=[], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Leader", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + scheduling.MemberPlacement( + role="Worker", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], ), + # Taints on InferenceClusters gate placement; a matching toleration on the + # ModelDeployment overrides them. NoSchedule keeps new replicas off a + # cluster but leaves existing ones; NoExecute additionally drains the + # existing ones, which fill reschedules onto a tolerated cluster. Case( - name="replica with no pool pin is re-placed when a selector now applies", + name="a NoSchedule taint keeps a new replica off the cluster", deployment=_deployment( - requests=[ - _request(name="gpu", cel_exprs=[_MEM_141]), - _request(name="nic", cel_exprs=[_IB]), - ] + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, ), clusters=[ _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device(link_type="infiniband")])], - ) + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=[icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule")], + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), ], - all_replicas=[_replica("my-model", "cluster-a")], + all_replicas=[], + fill=True, want=[ - _cand( + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], + ), + Case( + name="a NoSchedule taint leaves an existing replica in place", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved(name="gpu")], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=[icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule")], + placement_labels=None, + ) + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], ) ], + fill=True, + want=[ + _candidate( + name="cluster-a", + index=0, + gateway_hostname="cluster-a.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], ), Case( - name="dropping a non-matching replica frees its node for the refill", - # a/0 is pinned to a pool that still matches (retained). a/1 is - # pinned to a pool no longer published, so it's dropped and will - # be re-placed. The pool has just 2 nodes; both are notionally in - # use by a/0 and a/1. The refill must see a/1's node freeing up - # (it's being deleted) and re-place onto frontier at index 1. - # Regression: the ledger must not charge dropped replicas. - deployment=_deployment(replicas=2, requests=[_request(name="gpu", cel_exprs=[_MEM_141])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=2)])], - all_replicas=[ - _replica_with_pool("my-model", "cluster-a", pool="frontier", index=0), - _replica_with_pool("my-model", "cluster-a", pool="gone", index=1), + name="a toleration allows placement on a tainted cluster", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=[mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists")], + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=[icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule")], + placement_labels=None, + ) ], + all_replicas=[], + fill=True, want=[ - _cand( + _candidate( name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], - ), - _cand( - name="cluster-a", - index=1, - gateway_hostname="cluster-a.clusters.example.com", - pool="frontier", - device_requests=[_resolved()], + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], ), ], ), Case( - name="device count is checked against the pinned pool, not a cluster-wide sum", - # Request 8 GPUs. Pool 'a' has 4/node (doesn't fit); pool 'b' - # has 8 and does. The replica must pin to 'b'. - deployment=_deployment(requests=[_request(count=8, cel_exprs=[_MEM_141])]), + name="a NoExecute taint drains a replica and reschedules it", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=[icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute")], + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, pools=[ - _pool("a", devices=[_gpu_device(count=4)]), - _pool("b", devices=[_gpu_device(count=8)]), + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) ], ) ], - all_replicas=[], + fill=True, want=[ - _cand( - name="cluster-a", - gateway_hostname="cluster-a.clusters.example.com", - pool="b", - device_requests=[_resolved(count=8)], - ) + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), ], ), -] - - -@pytest.mark.parametrize("case", NODE_SELECTOR_CASES, ids=lambda case: case.name) -def test_node_selector(case: Case) -> None: - """The scheduler places replicas only on pools whose devices satisfy the nodeSelector.""" - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) - assert got == case.want - - -def test_node_selector_invalid_cel_raises() -> None: - """A malformed expression raises CELCompileError, which the caller handles.""" - deployment = _deployment(requests=[_request(cel_exprs=["this is ) not valid ("])]) - with pytest.raises(cel.CELCompileError, match=r"this is \) not valid \("): - scheduling.schedule(deployment, [_cluster("cluster-a", pools=[_pool("frontier")])], []) - - -def _gang( - leader_requests: list[mdv1alpha1.Device] | None, - worker_requests: list[mdv1alpha1.Device] | None, - *, - worker_nodes: int = 1, -) -> mdv1alpha1.Engine: - """A Leader/Worker engine with per-member (possibly heterogeneous) selectors. - - Either member's requests may be None, meaning that member carries no - nodeSelector and claims nothing. - """ - leader = mdv1alpha1.Member(role="Leader", template=_template()) - if leader_requests is not None: - leader.nodeSelector = _node_selector(leader_requests) - worker = mdv1alpha1.Member(role="Worker", worker=mdv1alpha1.Worker(nodes=worker_nodes), template=_template()) - if worker_requests is not None: - worker.nodeSelector = _node_selector(worker_requests) - return mdv1alpha1.Engine(name=_ENGINE, members=[leader, worker]) - - -# Per-member placement: single-pool engines, rejection when no pool fits, and -# claimless ride-along members. -MEMBERS_CASES = [ Case( - name="a single pool satisfying every member hosts the whole engine", - # The leader's request matches both pools; the worker's only - # matches big. The whole-engine pass must put both members on - # big - the one pool that satisfies them all. - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_200])], + name="a NoExecute toleration retains a replica in place", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], ) - ] + ], + tolerations=[mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists")], ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool("small", devices=[_gpu_device(memory="141Gi")]), - _pool("big", devices=[_gpu_device(memory="200Gi")]), + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) ], + taints=[icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute")], + placement_labels=None, ) ], - all_replicas=[], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, want=[ - scheduling.Candidate( + _candidate( name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="big", device_requests=[_resolved()]), - scheduling.MemberPlacement( - role="Worker", - pool="big", - device_requests=[_resolved(cel_exprs=[_MEM_200])], - ), + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) ], - ) + ), ], - ) + ), ], ), + # Draining with no tolerated cluster to reschedule onto yields fewer than + # spec.replicas; the deploy function surfaces the shortfall. Case( - name="members no single pool satisfies are not scheduled", - # The leader only fits big (>= 200Gi); the worker only fits - # small (< 200Gi). No single pool satisfies both. The scheduler - # never splits an engine across pools - it can't tell whether - # big and small share a fabric - so the engine is rejected and - # the replica goes unplaced (#149). - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_200])], - [_request(cel_exprs=[_MEM_LT_200])], + name="a NoExecute drain leaves the count unmet when there's nowhere to go", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], ) - ] + ], + tolerations=None, ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool("small", devices=[_gpu_device(memory="141Gi")]), - _pool("big", devices=[_gpu_device(memory="200Gi")]), + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) ], + taints=[icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute")], + placement_labels=None, ) ], - all_replicas=[], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) + ], + ) + ], + fill=True, want=[], ), + # Matching the key but not the effect doesn't tolerate: an operator who + # tolerates only NoSchedule is still drained by a NoExecute taint. Case( - name="a gang too big for its only matching pool is rejected", - # Both members match only big (>= 141Gi); small (40Gi) matches - # neither. big has one free node but the gang needs two. The - # engine doesn't fit any single pool, so it's rejected; with big - # the only cluster the replica goes unplaced. - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_141])], + name="a NoSchedule toleration does not cover a NoExecute taint", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], ) - ] + ], + tolerations=[ + mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists", effect="NoSchedule") + ], ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=[icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute")], + placement_labels=None, + ), + _cluster( + name="cluster-b", + gateway_hostname="cluster-b.clusters.example.com", + ready=True, pools=[ - _pool("small", nodes=8, devices=[_gpu_device(memory="40Gi")]), - _pool("big", nodes=1, devices=[_gpu_device(memory="141Gi")]), + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, + ), + ], + all_replicas=[ + _replica( + name="my-model-cluster-a-0", + deployment="my-model", + cluster="cluster-a", + index=0, + members=[ + _replica_member( + role="Standalone", + worker_nodes=None, + pool="default", + device_requests=[ + mrv1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[mrv1alpha1.Selector(cel=_MEM_141)], + ) + ], + ) ], ) ], - all_replicas=[], - want=[], + fill=True, + want=[ + _candidate( + name="cluster-b", + index=0, + gateway_hostname="cluster-b.clusters.example.com", + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) + ], + ), + ], + ), + ], ), + # Tolerating one of a cluster's taints isn't enough; any untolerated taint + # keeps new replicas off. Case( - name="a gang too big for one cluster's pool lands whole on another", - # Same gang. cluster-a's matching pool has only one free node - # (too few for the two-member gang), so the scheduler rejects - # cluster-a and places the whole gang on cluster-b, whose pool - # has room for both members. - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_141])], + name="an untolerated second taint still repels", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], ) - ] + ], + tolerations=[mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists")], ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool("small", nodes=8, devices=[_gpu_device(memory="40Gi")]), - _pool("big", nodes=1, devices=[_gpu_device(memory="141Gi")]), + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) ], + taints=[ + icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule"), + icv1alpha1.Taint(key="modelplane.ai/reserved", effect="NoSchedule"), + ], + placement_labels=None, ), _cluster( - "cluster-b", + name="cluster-b", gateway_hostname="cluster-b.clusters.example.com", - pools=[_pool("big", nodes=2, devices=[_gpu_device(memory="141Gi")])], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) + ], + taints=None, + placement_labels=None, ), ], all_replicas=[], + fill=True, want=[ - scheduling.Candidate( + _candidate( name="cluster-b", index=0, gateway_hostname="cluster-b.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="big", device_requests=[_resolved()]), - scheduling.MemberPlacement(role="Worker", pool="big", device_requests=[_resolved()]), + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) ], - ) + ), ], - ) + ), ], ), + # Equal tolerates only when key and value both match. Case( - name="a member claimable elsewhere is not stranded on a synthetic match", - # On pool-a the leader's request matches only a Synthetic - # device (nothing to claim) while the worker claims, so the - # whole engine *could* land there - but pool-b satisfies the - # leader claimably. The engine must go to pool-b; placing on - # pool-a would run the leader without the GPU it asked for. - deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_200])], - [_request(cel_exprs=[_MEM_141])], + name="an Equal toleration tolerates a taint with the same value", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], ) - ] + ], + tolerations=[mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="on")], ), clusters=[ _cluster( - "cluster-a", + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, pools=[ - _pool( - "a", - devices=[ - _gpu_device(memory="141Gi"), - _gpu_device(name="syn", claim="Synthetic", memory="200Gi"), - ], - ), - _pool("b", devices=[_gpu_device(memory="200Gi")]), + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] + ) ], + taints=[icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule")], + placement_labels=None, ) ], all_replicas=[], + fill=True, want=[ - scheduling.Candidate( + _candidate( name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement( - role="Leader", - pool="b", - device_requests=[_resolved(cel_exprs=[_MEM_200])], - ), - scheduling.MemberPlacement( - role="Worker", - pool="b", - device_requests=[_resolved(cel_exprs=[_MEM_141])], - ), + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) ], - ) + ), ], - ) + ), ], ), Case( - name="a member synthetic-only everywhere places claimless with its gang", - # The leader's request matches only the pool's synthetic NIC on - # every pool - deliberate (a selector that pins without - # claiming). It places claimless alongside the claiming worker. - deployment=_deployment( - engines=[ - _gang( - [_request(name="nic", cel_exprs=[_IB])], - [_request(cel_exprs=[_MEM_141])], + name="an Equal toleration does not tolerate a taint with another value", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], ) - ] + ], + tolerations=[mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="off")], ), clusters=[ _cluster( - "cluster-a", - pools=[_pool("frontier", devices=[_gpu_device(), _nic_device()])], - ) - ], - all_replicas=[], - want=[ - scheduling.Candidate( name="cluster-a", - index=0, gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), - scheduling.MemberPlacement(role="Worker", pool="frontier", device_requests=[_resolved()]), - ], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] ) ], + taints=[icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule")], + placement_labels=None, ) ], + all_replicas=[], + fill=True, + want=[], ), + # An Exists toleration with no key tolerates any taint on the cluster. Case( - name="a member that matches nowhere fails the whole replica", + name="a keyless Exists toleration tolerates every taint", deployment=_deployment( - engines=[ - _gang( - [_request(cel_exprs=[_MEM_141])], - [_request(cel_exprs=[_MEM_200])], + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], ) - ] + ], + tolerations=[mdv1alpha1.Toleration(operator="Exists")], ), - clusters=[_cluster("cluster-a", pools=[_pool("default", devices=[_gpu_device(memory="141Gi")])])], - all_replicas=[], - want=[], - ), - Case( - name="a claimless leader rides along on its gang's pool at zero cost", - # The leader carries no nodeSelector: it claims nothing, follows - # the worker's pool, and costs no nodes - the 1-node pool fits - # the whole gang because only the worker occupies a node. - deployment=_deployment(engines=[_gang(None, [_request(cel_exprs=[_MEM_141])])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], - all_replicas=[], - want=[ - scheduling.Candidate( + clusters=[ + _cluster( name="cluster-a", - index=0, gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), - scheduling.MemberPlacement(role="Worker", pool="frontier", device_requests=[_resolved()]), - ], + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] ) ], - ) - ], - ), - Case( - name="a retained replica's claimless member keeps its pin", - deployment=_deployment(engines=[_gang(None, [_request(cel_exprs=[_MEM_141])])]), - clusters=[_cluster("cluster-a", pools=[_pool("frontier", nodes=1)])], - all_replicas=[ - _replica( - "my-model", - "cluster-a", - engines=[ - mrv1alpha1.Engine( - name=_ENGINE, - members=[ - mrv1alpha1.Member( - role="Leader", - nodePoolName="frontier", - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - mrv1alpha1.Member( - role="Worker", - worker=mrv1alpha1.Worker(nodes=1), - nodePoolName="frontier", - deviceRequests=_replica_device_requests(), - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - ], - ) + taints=[ + icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule"), + icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute"), ], + placement_labels=None, ) ], + all_replicas=[], + fill=True, want=[ - scheduling.Candidate( + _candidate( name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="frontier", device_requests=[]), - scheduling.MemberPlacement(role="Worker", pool="frontier", device_requests=[_resolved()]), + placement_labels={}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) ], - ) + ), ], - ) + ), ], ), + # A cluster's placement labels reach the Candidate, and so the ModelReplica + # and ModelEndpoint composed from it. This is how a self-hosted endpoint + # gets its region: a ModelService selects endpoints by label, so without it + # a region-scoped service can't select its own replicas, and it can't label + # them by hand because Modelplane owns them. Case( - name="another deployment's claimless member consumes no capacity", - # other-model's gang occupies only its worker's node: its - # claimless leader shares that node. The 2-node pool has 1 node - # free, so our 1-node deployment fits. Charging the claimless - # leader a node would wrongly report insufficient capacity. - deployment=_deployment(), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=2)])], - all_replicas=[ - _replica( - "other-model", - "cluster-a", - engines=[ - mrv1alpha1.Engine( - name="main", - members=[ - mrv1alpha1.Member( - role="Leader", - nodePoolName="default", - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - mrv1alpha1.Member( - role="Worker", - worker=mrv1alpha1.Worker(nodes=1), - nodePoolName="default", - deviceRequests=_replica_device_requests(), - template=mrv1alpha1.Template( - spec=mrv1alpha1.Spec( - containers=[ - mrv1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - ] - ) - ), - ), - ], + name="a cluster's placement labels reach the candidate", + deployment=_deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[mdv1alpha1.Device(name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel=_MEM_141)])], + ) + ], + tolerations=None, + ), + clusters=[ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="default", nodes=2, devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")] ) ], + taints=None, + placement_labels={"example.org/region": "eu"}, ) ], - want=[_cand(name="cluster-a", gateway_hostname="cluster-a.clusters.example.com")], - ), - Case( - name="a member shape change re-places the replica", - # The deployment grew a Worker (Standalone -> Leader+Worker). - # The observed single-member replica no longer lines up, so it - # is re-placed with the new shape. - deployment=_deployment(engines=[_gang([_request()], [_request()])]), - clusters=[_cluster("cluster-a", pools=[_pool("default", nodes=4)])], - all_replicas=[_replica("my-model", "cluster-a")], + all_replicas=[], + fill=True, want=[ - scheduling.Candidate( + _candidate( name="cluster-a", index=0, gateway_hostname="cluster-a.clusters.example.com", - engines=[ - scheduling.EnginePlacement( - name=_ENGINE, - members=[ - scheduling.MemberPlacement(role="Leader", pool="default", device_requests=[_resolved()]), - scheduling.MemberPlacement(role="Worker", pool="default", device_requests=[_resolved()]), + placement_labels={"example.org/region": "eu"}, + members=[ + scheduling.MemberPlacement( + role="Standalone", + pool="default", + device_requests=[ + scheduling.DeviceRequest( + name="gpu", + device_class_name="gpu.nvidia.com", + count=1, + cel_selectors=[_MEM_141], + ) ], - ) + ), ], - ) + ), ], ), ] -@pytest.mark.parametrize("case", MEMBERS_CASES, ids=lambda case: case.name) -def test_members(case: Case) -> None: - """The scheduler places every member of an engine on one pool that fits them all.""" - got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas) +@pytest.mark.parametrize("case", SCHEDULE_CASES, ids=lambda case: case.name) +def test_schedule(case: Case) -> None: + """schedule() retains existing replicas and places new ones on clusters that can host them.""" + got = scheduling.schedule(case.deployment, case.clusters, case.all_replicas, fill=case.fill) assert got == case.want -# Taints on InferenceClusters gate placement; a matching toleration on the -# ModelDeployment overrides them. NoSchedule keeps new replicas off a cluster but -# leaves existing ones; NoExecute additionally drains the existing ones, which -# fill reschedules onto a tolerated cluster. -_MAINT = icv1alpha1.Taint(key="modelplane.ai/maintenance", value="on", effect="NoSchedule") -_DECOMM = icv1alpha1.Taint(key="modelplane.ai/decommission", effect="NoExecute") - - -def _names(got: list[scheduling.Candidate]) -> list[tuple[str, int]]: - return [(c.name, c.index) for c in got] - - -def test_taints_noschedule_keeps_new_replicas_off() -> None: - """A NoSchedule taint keeps a new replica off the cluster.""" - clusters = [ - _cluster("cluster-a", taints=[_MAINT]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1), clusters, []) - assert _names(got) == [("cluster-b", 0)] - - -def test_taints_noschedule_leaves_existing_replica_in_place() -> None: - """A NoSchedule taint leaves an existing replica where it is.""" - existing = _replica("my-model", "cluster-a") - got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a", taints=[_MAINT])], [existing]) - assert _names(got) == [("cluster-a", 0)] - - -def test_taints_toleration_allows_placement_on_tainted() -> None: - """A matching toleration lets a new replica onto a tainted cluster.""" - tol = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists") - got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), [_cluster("cluster-a", taints=[_MAINT])], []) - assert _names(got) == [("cluster-a", 0)] - - -def test_taints_noexecute_drains_and_reschedules() -> None: - """A NoExecute taint drains an existing replica onto another cluster.""" - existing = _replica("my-model", "cluster-a") - clusters = [ - _cluster("cluster-a", taints=[_DECOMM]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1), clusters, [existing]) - assert _names(got) == [("cluster-b", 0)] - - -def test_taints_noexecute_toleration_retains_in_place() -> None: - """A NoExecute toleration keeps an existing replica on a draining cluster.""" - tol = mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists") - existing = _replica("my-model", "cluster-a") - got = scheduling.schedule( - _deployment(replicas=1, tolerations=[tol]), [_cluster("cluster-a", taints=[_DECOMM])], [existing] - ) - assert _names(got) == [("cluster-a", 0)] - - -def test_taints_noexecute_drain_leaves_count_unmet_when_nowhere_to_go() -> None: - """Draining with no tolerated cluster to go to yields fewer than spec.replicas.""" - # The deploy function surfaces the shortfall. - existing = _replica("my-model", "cluster-a") - got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a", taints=[_DECOMM])], [existing]) - assert got == [] - - -def test_taints_noschedule_toleration_does_not_cover_a_noexecute_taint() -> None: - """A toleration that matches the key but not the effect doesn't tolerate.""" - # An operator who tolerates only NoSchedule is still drained by a NoExecute - # taint. - tol = mdv1alpha1.Toleration(key="modelplane.ai/decommission", operator="Exists", effect="NoSchedule") - existing = _replica("my-model", "cluster-a") - clusters = [ - _cluster("cluster-a", taints=[_DECOMM]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, [existing]) - assert _names(got) == [("cluster-b", 0)] - - -def test_taints_untolerated_second_taint_still_repels() -> None: - """Any untolerated taint keeps new replicas off, even if another is tolerated.""" - other = icv1alpha1.Taint(key="modelplane.ai/reserved", effect="NoSchedule") - tol = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Exists") - clusters = [ - _cluster("cluster-a", taints=[_MAINT, other]), - _cluster("cluster-b", gateway_hostname="cluster-b.clusters.example.com"), - ] - got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, []) - assert _names(got) == [("cluster-b", 0)] - - -def test_taints_equal_toleration_matches_on_value() -> None: - """An Equal toleration tolerates only when key and value both match.""" - match = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="on") - placed = scheduling.schedule( - _deployment(replicas=1, tolerations=[match]), [_cluster("cluster-a", taints=[_MAINT])], [] - ) - assert _names(placed) == [("cluster-a", 0)] - - mismatch = mdv1alpha1.Toleration(key="modelplane.ai/maintenance", operator="Equal", value="off") - repelled = scheduling.schedule( - _deployment(replicas=1, tolerations=[mismatch]), [_cluster("cluster-a", taints=[_MAINT])], [] - ) - assert repelled == [] - - -def test_taints_keyless_exists_tolerates_every_taint() -> None: - """An Exists toleration with no key tolerates any taint on the cluster.""" - tol = mdv1alpha1.Toleration(operator="Exists") - clusters = [_cluster("cluster-a", taints=[_MAINT, _DECOMM])] - got = scheduling.schedule(_deployment(replicas=1, tolerations=[tol]), clusters, []) - assert _names(got) == [("cluster-a", 0)] - - -# A cluster's placement labels reach the Candidate, and so the ModelReplica and -# ModelEndpoint composed from it. -# -# This is how a self-hosted endpoint gets its region: a ModelService selects -# endpoints by label, so without it a region-scoped service can't select its own -# replicas, and it can't label them by hand because Modelplane owns them. - - -def test_placement_labels_reach_the_candidate() -> None: - """A cluster's placement labels reach the Candidate.""" - got = scheduling.schedule( - _deployment(replicas=1), - [_cluster("cluster-a", placement_labels={"example.org/region": "eu"})], - [], - ) - assert [c.placement_labels for c in got] == [{"example.org/region": "eu"}] - - -def test_placement_labels_a_cluster_declaring_none_yields_none() -> None: - """A cluster that declares no placement labels yields a Candidate with none.""" - got = scheduling.schedule(_deployment(replicas=1), [_cluster("cluster-a")], []) - assert [c.placement_labels for c in got] == [{}] +def test_schedule_invalid_cel_raises() -> None: + """A malformed expression raises CELCompileError, which the caller handles.""" + with pytest.raises(cel.CELCompileError, match=r"this is \) not valid \("): + scheduling.schedule( + _deployment( + replicas=1, + members=[ + _member( + role="Standalone", + worker_nodes=None, + devices=[ + mdv1alpha1.Device( + name="gpu", count=1, selectors=[mdv1alpha1.Selector(cel="this is ) not valid (")] + ) + ], + ) + ], + tolerations=None, + ), + [ + _cluster( + name="cluster-a", + gateway_hostname="cluster-a.clusters.example.com", + ready=True, + pools=[ + icv1alpha1.GpuPool( + name="frontier", + nodes=2, + devices=[_gpu_device(name="gpu", claim="DRA", count=1, memory="141Gi")], + ) + ], + taints=None, + placement_labels=None, + ) + ], + [], + ) diff --git a/functions/compose-model-deployment/tests/test_semver.py b/functions/compose-model-deployment/tests/test_semver.py index 433a3f058..1a40e67ba 100644 --- a/functions/compose-model-deployment/tests/test_semver.py +++ b/functions/compose-model-deployment/tests/test_semver.py @@ -34,98 +34,99 @@ from function import cel, semver -def _eval(expr: str) -> bool: - """Compile and evaluate a deviceless boolean CEL expression.""" - return cel.Program(expr).matches({}) - - @dataclasses.dataclass -class Case: +class SemverCase: + """A test case for a semver CEL expression.""" + name: str expr: str want: bool @dataclasses.dataclass -class ParseErrCase: +class ParseRejectsCase: + """A test case for a version string semver.parse rejects.""" + name: str - input: str + s: str + want: str SEMVER_CASES = [ - # parse + doc-comment examples. - Case(name="parse", expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', want=True), - Case(name="parse with prerelease", expr='semver("0.1.0-alpha.1").major() == 0', want=True), + # parse + doc-comment examples. Upstream's parse row returns the version + # itself, so wrapped to assert a bool it's the same expression as its + # "compare equal" row below. Both are kept to mirror upstream's table. + SemverCase(name="parse", expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', want=True), + SemverCase(name="parse with prerelease", expr='semver("0.1.0-alpha.1").major() == 0', want=True), # isSemver strict. - Case(name="isSemver full", expr='isSemver("1.2.3-beta.1+build.1")', want=True), - Case(name="isSemver simple", expr='isSemver("1.0.0")', want=True), - Case(name="isSemver hello", expr='isSemver("hello")', want=False), - Case(name="isSemver empty false", expr='isSemver("")', want=False), - Case(name="isSemver v prefix false", expr='isSemver("v1.0.0")', want=False), - Case(name="isSemver v1.0 false", expr='isSemver("v1.0")', want=False), - Case(name="isSemver leading whitespace false", expr='isSemver(" 1.0.0")', want=False), - Case(name="isSemver inner whitespace false", expr='isSemver("1. 0.0")', want=False), - Case(name="isSemver trailing whitespace false", expr='isSemver("1.0.0 ")', want=False), - Case(name="isSemver leading zeros false", expr='isSemver("01.01.01")', want=False), - Case(name="isSemver major only false", expr='isSemver("1")', want=False), - Case(name="isSemver major minor only false", expr='isSemver("1.1")', want=False), - Case(name="isSemver 200K", expr='isSemver("200K")', want=False), - Case(name="isSemver Mi", expr='isSemver("Mi")', want=False), + SemverCase(name="isSemver full", expr='isSemver("1.2.3-beta.1+build.1")', want=True), + SemverCase(name="isSemver simple", expr='isSemver("1.0.0")', want=True), + SemverCase(name="isSemver hello", expr='isSemver("hello")', want=False), + SemverCase(name="isSemver empty false", expr='isSemver("")', want=False), + SemverCase(name="isSemver v prefix false", expr='isSemver("v1.0.0")', want=False), + SemverCase(name="isSemver v1.0 false", expr='isSemver("v1.0")', want=False), + SemverCase(name="isSemver leading whitespace false", expr='isSemver(" 1.0.0")', want=False), + SemverCase(name="isSemver inner whitespace false", expr='isSemver("1. 0.0")', want=False), + SemverCase(name="isSemver trailing whitespace false", expr='isSemver("1.0.0 ")', want=False), + SemverCase(name="isSemver leading zeros false", expr='isSemver("01.01.01")', want=False), + SemverCase(name="isSemver major only false", expr='isSemver("1")', want=False), + SemverCase(name="isSemver major minor only false", expr='isSemver("1.1")', want=False), + SemverCase(name="isSemver 200K", expr='isSemver("200K")', want=False), + SemverCase(name="isSemver Mi", expr='isSemver("Mi")', want=False), # isSemver normalize overload. Normalization does NOT trim whitespace. - Case(name="isSemver empty normalize false", expr='isSemver("", true)', want=False), - Case(name="isSemver leading whitespace normalize false", expr='isSemver(" 1.0.0", true)', want=False), - Case(name="isSemver inner whitespace normalize false", expr='isSemver("1. 0.0", true)', want=False), - Case(name="isSemver trailing whitespace normalize false", expr='isSemver("1.0.0 ", true)', want=False), - Case(name="isSemver v prefix normalize true", expr='isSemver("v1.0.0", true)', want=True), - Case(name="isSemver leading zeros normalize true", expr='isSemver("01.01.01", true)', want=True), - Case(name="isSemver major only normalize true", expr='isSemver("1", true)', want=True), - Case(name="isSemver major minor only normalize true", expr='isSemver("1.1", true)', want=True), + SemverCase(name="isSemver empty normalize false", expr='isSemver("", true)', want=False), + SemverCase(name="isSemver leading whitespace normalize false", expr='isSemver(" 1.0.0", true)', want=False), + SemverCase(name="isSemver inner whitespace normalize false", expr='isSemver("1. 0.0", true)', want=False), + SemverCase(name="isSemver trailing whitespace normalize false", expr='isSemver("1.0.0 ", true)', want=False), + SemverCase(name="isSemver v prefix normalize true", expr='isSemver("v1.0.0", true)', want=True), + SemverCase(name="isSemver leading zeros normalize true", expr='isSemver("01.01.01", true)', want=True), + SemverCase(name="isSemver major only normalize true", expr='isSemver("1", true)', want=True), + SemverCase(name="isSemver major minor only normalize true", expr='isSemver("1.1", true)', want=True), # normalize equality and semver(...) examples. - Case(name="equality normalize", expr='semver("v01.01", true) == semver("1.1.0")', want=True), - Case(name="semver v prefix normalize major", expr='semver("v1.0.0", true).major() == 1', want=True), - Case(name="semver short normalize patch", expr='semver("1.0", true).patch() == 0', want=True), - Case(name="semver leading zeros normalize", expr='semver("01.01.01", true).minor() == 1', want=True), + SemverCase(name="equality normalize", expr='semver("v01.01", true) == semver("1.1.0")', want=True), + SemverCase(name="semver v prefix normalize major", expr='semver("v1.0.0", true).major() == 1', want=True), + SemverCase(name="semver short normalize patch", expr='semver("1.0", true).patch() == 0', want=True), + SemverCase(name="semver leading zeros normalize", expr='semver("01.01.01", true).minor() == 1', want=True), # equality / comparison. - Case(name="equality reflexivity", expr='semver("1.2.3") == semver("1.2.3")', want=True), - Case(name="inequality", expr='semver("1.2.3") == semver("1.0.0")', want=False), - Case(name="less", expr='semver("1.0.0").isLessThan(semver("1.2.3"))', want=True), - Case(name="less false", expr='semver("1.0.0").isLessThan(semver("1.0.0"))', want=False), - Case(name="greater", expr='semver("1.2.3").isGreaterThan(semver("1.0.0"))', want=True), - Case(name="greater false", expr='semver("1.0.0").isGreaterThan(semver("1.0.0"))', want=False), - Case(name="compare equal", expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', want=True), - Case(name="compare less", expr='semver("1.2.3").compareTo(semver("2.0.0")) == -1', want=True), - Case(name="compare greater", expr='semver("1.2.3").compareTo(semver("0.1.2")) == 1', want=True), + SemverCase(name="equality reflexivity", expr='semver("1.2.3") == semver("1.2.3")', want=True), + SemverCase(name="inequality", expr='semver("1.2.3") == semver("1.0.0")', want=False), + SemverCase(name="less", expr='semver("1.0.0").isLessThan(semver("1.2.3"))', want=True), + SemverCase(name="less false", expr='semver("1.0.0").isLessThan(semver("1.0.0"))', want=False), + SemverCase(name="greater", expr='semver("1.2.3").isGreaterThan(semver("1.0.0"))', want=True), + SemverCase(name="greater false", expr='semver("1.0.0").isGreaterThan(semver("1.0.0"))', want=False), + SemverCase(name="compare equal", expr='semver("1.2.3").compareTo(semver("1.2.3")) == 0', want=True), + SemverCase(name="compare less", expr='semver("1.2.3").compareTo(semver("2.0.0")) == -1', want=True), + SemverCase(name="compare greater", expr='semver("1.2.3").compareTo(semver("0.1.2")) == 1', want=True), # major / minor / patch. - Case(name="major", expr='semver("1.2.3").major() == 1', want=True), - Case(name="minor", expr='semver("1.2.3").minor() == 2', want=True), - Case(name="patch", expr='semver("1.2.3").patch() == 3', want=True), + SemverCase(name="major", expr='semver("1.2.3").major() == 1', want=True), + SemverCase(name="minor", expr='semver("1.2.3").minor() == 2', want=True), + SemverCase(name="patch", expr='semver("1.2.3").patch() == 3', want=True), # A bad version is a runtime error upstream -> non-match here. - Case(name="bad version is non-match", expr='semver("v1.0").major() == 1', want=False), + SemverCase(name="bad version is non-match", expr='semver("v1.0").major() == 1', want=False), ] @pytest.mark.parametrize("case", SEMVER_CASES, ids=lambda case: case.name) -def test_semver(case: Case) -> None: +def test_semver(case: SemverCase) -> None: """A semver CEL expression evaluates as it does upstream.""" - assert _eval(case.expr) == case.want + got = cel.Program(case.expr).matches({}) + assert got == case.want PARSE_REJECTS_CASES = [ - ParseErrCase(name="v prefix", input="v1.0"), - ParseErrCase(name="major only", input="1"), - ParseErrCase(name="major minor only", input="1.1"), - ParseErrCase(name="leading zeros", input="01.01.01"), - ParseErrCase(name="leading whitespace", input=" 1.0.0"), - ParseErrCase(name="trailing whitespace", input="1.0.0 "), - ParseErrCase(name="empty", input=""), - ParseErrCase(name="word", input="hello"), + ParseRejectsCase(name="v prefix", s="v1.0", want=r"no Major\.Minor\.Patch elements found"), + ParseRejectsCase(name="major only", s="1", want=r"no Major\.Minor\.Patch elements found"), + ParseRejectsCase(name="major minor only", s="1.1", want=r"no Major\.Minor\.Patch elements found"), + ParseRejectsCase(name="leading zeros", s="01.01.01", want="major number must not contain leading zeroes: '01'"), + ParseRejectsCase(name="leading whitespace", s=" 1.0.0", want=r"invalid character\(s\) in major number: ' 1'"), + ParseRejectsCase(name="trailing whitespace", s="1.0.0 ", want=r"invalid character\(s\) in patch number: '0 '"), + ParseRejectsCase(name="empty", s="", want="version string empty"), + ParseRejectsCase(name="word", s="hello", want=r"no Major\.Minor\.Patch elements found"), ] @pytest.mark.parametrize("case", PARSE_REJECTS_CASES, ids=lambda case: case.name) -def test_parse_rejects(case: ParseErrCase) -> None: +def test_parse_rejects(case: ParseRejectsCase) -> None: """parse() (strict) rejects what blang/semver Parse rejects.""" - with pytest.raises( - ValueError, match=r"version string empty|no Major\.Minor\.Patch|invalid character|leading zeroes" - ): - semver.parse(case.input) + with pytest.raises(ValueError, match=case.want): + semver.parse(case.s) diff --git a/functions/compose-model-endpoint/tests/test_fn.py b/functions/compose-model-endpoint/tests/test_fn.py index a149608a9..35bcbd8ff 100644 --- a/functions/compose-model-endpoint/tests/test_fn.py +++ b/functions/compose-model-endpoint/tests/test_fn.py @@ -15,7 +15,6 @@ """Tests for the compose-model-endpoint function.""" import asyncio -import base64 import dataclasses import json @@ -27,9 +26,7 @@ from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.modelendpoint import v1alpha1 - -_NS = "ml-team" -_NAME = "together-kimi-k2" +from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @dataclasses.dataclass @@ -41,190 +38,199 @@ class Case: want: fnv1.RunFunctionResponse -def _xr(**spec) -> dict: # noqa: ANN003 - """The ModelEndpoint XR, built from the generated model so a field the XRD - doesn't define can't creep into a test.""" - xr = v1alpha1.ModelEndpoint( - apiVersion="modelplane.ai/v1alpha1", - kind="ModelEndpoint", - metadata={"name": _NAME, "namespace": _NS}, - spec=v1alpha1.Spec(origin="https://api.together.xyz", **spec), - ) - return xr.model_dump(exclude_none=True, mode="json", by_alias=True) - - -def _api_key(secret: str, key: str = "apiKey") -> v1alpha1.Credential: - """An API key credential read from the named Secret.""" - return v1alpha1.Credential( - method="APIKey", apiKey=v1alpha1.ApiKey(secretRef=v1alpha1.SecretRef(name=secret, key=key)) +def _model_endpoint(*, credential_key: str | None) -> fnv1.Resource: + """The together-kimi-k2 XR, with an API key under credential_key of together-api-key, or no credential if None.""" + credential = None + if credential_key is not None: + credential = v1alpha1.Credential( + method="APIKey", + apiKey=v1alpha1.ApiKey(secretRef=v1alpha1.SecretRef(name="together-api-key", key=credential_key)), + ) + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ModelEndpoint( + apiVersion="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + metadata=metav1.ObjectMeta(name="together-kimi-k2", namespace="ml-team"), + spec=v1alpha1.Spec(origin="https://api.together.xyz", credential=credential), + ).model_dump(exclude_none=True, mode="json", by_alias=True) + ) ) -def _secret(name: str, data: dict[str, str]) -> dict: - """A Secret as the API server stores it, values base64 encoded.""" - return { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": name, "namespace": _NS}, - "data": {k: base64.b64encode(v.encode()).decode() for k, v in data.items()}, - } - - -def _credential_requirement(name: str) -> fnv1.Requirements: - return fnv1.Requirements( - resources={"credential": fnv1.ResourceSelector(api_version="v1", kind="Secret", match_name=name, namespace=_NS)} +def _credential_secret(*, key: str) -> fnv1.Resource: + """The together-api-key Secret, as the credential requirement returns it, holding sk-abc under key.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "together-api-key", "namespace": "ml-team"}, + # Base64 encoded, as the API server stores it. c2stYWJj is + # "sk-abc". + "data": {key: "c2stYWJj"}, + } + ) ) -def _response( - *, - reason: str, - status: fnv1.Status, - message: str | None = None, - requirements: fnv1.Requirements | None = None, -) -> fnv1.RunFunctionResponse: - """The whole response. This function composes no resources, so desired - carries only the composite's readiness, which mirrors EndpointReady, and - asserting the whole thing proves it stays that way.""" - ready = fnv1.READY_TRUE if status == fnv1.STATUS_CONDITION_TRUE else fnv1.READY_FALSE - rsp = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=ready)), - context=structpb.Struct(), - conditions=[ - fnv1.Condition(type=fn.CONDITION_TYPE_ENDPOINT_READY, status=status, reason=reason, message=message) - ], - ) - if requirements is not None: - rsp.requirements.CopyFrom(requirements) - if message is not None: - rsp.results.append(fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message=message)) - return rsp +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) +# This function composes no resources, so desired carries only the composite's +# readiness, which mirrors EndpointReady. Comparing the whole response proves it +# stays that way. The desired XR is written inline although every case has it: +# it's a bare fnv1.Resource carrying only readiness, so a helper would only +# rename its constructor. COMPOSE_CASES = [ Case( name="no credential: usable as soon as it exists", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_xr()))), - ), - want=_response( - reason=fn.CONDITION_REASON_ENDPOINT_USABLE, - status=fnv1.STATUS_CONDITION_TRUE, + req=fnv1.RunFunctionRequest(observed=fnv1.State(composite=_model_endpoint(credential_key=None))), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_TRUE)), + context=structpb.Struct(), + conditions=[ + fnv1.Condition(type="EndpointReady", status=fnv1.STATUS_CONDITION_TRUE, reason="EndpointUsable"), + ], ), ), Case( - name="a credential that resolves", + name="a credential that resolves: usable", req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) - ), - required_resources={ - "credential": fnv1.Resources( - items=[ - fnv1.Resource( - resource=resource.dict_to_struct(_secret("together-api-key", {"apiKey": "sk-abc"})) - ) - ] - ) - }, + observed=fnv1.State(composite=_model_endpoint(credential_key="apiKey")), + required_resources={"credential": fnv1.Resources(items=[_credential_secret(key="apiKey")])}, ), - want=_response( - reason=fn.CONDITION_REASON_ENDPOINT_USABLE, - status=fnv1.STATUS_CONDITION_TRUE, - requirements=_credential_requirement("together-api-key"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_TRUE)), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "credential": fnv1.ResourceSelector( + api_version="v1", kind="Secret", match_name="together-api-key", namespace="ml-team" + ) + } + ), + conditions=[ + fnv1.Condition(type="EndpointReady", status=fnv1.STATUS_CONDITION_TRUE, reason="EndpointUsable"), + ], ), ), Case( - name="a credential Secret that does not exist", + name="a credential Secret that does not exist: credential missing", req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) - ), - required_resources={"credential": fnv1.Resources(items=[])}, + observed=fnv1.State(composite=_model_endpoint(credential_key="apiKey")), + required_resources={"credential": fnv1.Resources()}, ), - want=_response( - reason=fn.CONDITION_REASON_CREDENTIAL_MISSING, - status=fnv1.STATUS_CONDITION_FALSE, - message="Secret together-api-key does not exist", - requirements=_credential_requirement("together-api-key"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Secret together-api-key does not exist")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "credential": fnv1.ResourceSelector( + api_version="v1", kind="Secret", match_name="together-api-key", namespace="ml-team" + ) + } + ), + conditions=[ + fnv1.Condition( + type="EndpointReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="CredentialMissing", + message="Secret together-api-key does not exist", + ), + ], ), ), + # A Secret that exists but lacks the key is the likelier mistake, and would + # otherwise surface as a 401 from the provider. Case( - # A Secret that exists but lacks the key is the likelier mistake, - # and would otherwise surface as a 401 from the provider. - name="a credential Secret missing the key", + name="a credential Secret missing the key: credential missing", req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) - ), - required_resources={ - "credential": fnv1.Resources( - items=[ - fnv1.Resource( - resource=resource.dict_to_struct(_secret("together-api-key", {"token": "sk-abc"})) - ) - ] - ) - }, + observed=fnv1.State(composite=_model_endpoint(credential_key="apiKey")), + required_resources={"credential": fnv1.Resources(items=[_credential_secret(key="token")])}, ), - want=_response( - reason=fn.CONDITION_REASON_CREDENTIAL_MISSING, - status=fnv1.STATUS_CONDITION_FALSE, - message="Secret together-api-key has no key apiKey", - requirements=_credential_requirement("together-api-key"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Secret together-api-key has no key apiKey")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "credential": fnv1.ResourceSelector( + api_version="v1", kind="Secret", match_name="together-api-key", namespace="ml-team" + ) + } + ), + conditions=[ + fnv1.Condition( + type="EndpointReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="CredentialMissing", + message="Secret together-api-key has no key apiKey", + ), + ], ), ), Case( - name="a credential under a non-default key", + name="a credential under a non-default key: usable", req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - _xr(credential=_api_key("together-api-key", key="TOGETHER_API_KEY")) + observed=fnv1.State(composite=_model_endpoint(credential_key="TOGETHER_API_KEY")), + required_resources={"credential": fnv1.Resources(items=[_credential_secret(key="TOGETHER_API_KEY")])}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_TRUE)), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "credential": fnv1.ResourceSelector( + api_version="v1", kind="Secret", match_name="together-api-key", namespace="ml-team" ) - ) + } ), - required_resources={ - "credential": fnv1.Resources( - items=[ - fnv1.Resource( - resource=resource.dict_to_struct( - _secret("together-api-key", {"TOGETHER_API_KEY": "sk-abc"}) - ) - ) - ] - ) - }, - ), - want=_response( - reason=fn.CONDITION_REASON_ENDPOINT_USABLE, - status=fnv1.STATUS_CONDITION_TRUE, - requirements=_credential_requirement("together-api-key"), + conditions=[ + fnv1.Condition(type="EndpointReady", status=fnv1.STATUS_CONDITION_TRUE, reason="EndpointUsable"), + ], ), ), Case( - name="an unresolved credential requirement", + name="an unresolved credential requirement: wait for it to resolve", req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(credential=_api_key("together-api-key")))) - ), + observed=fnv1.State(composite=_model_endpoint(credential_key="apiKey")), ), - want=_response( - reason=fn.CONDITION_REASON_WAITING_FOR_CREDENTIAL, - status=fnv1.STATUS_CONDITION_FALSE, - message="Waiting for Secret together-api-key to resolve", - requirements=_credential_requirement("together-api-key"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for Secret together-api-key to resolve") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "credential": fnv1.ResourceSelector( + api_version="v1", kind="Secret", match_name="together-api-key", namespace="ml-team" + ) + } + ), + conditions=[ + fnv1.Condition( + type="EndpointReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCredential", + message="Waiting for Secret together-api-key to resolve", + ), + ], ), ), ] -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - @pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction reports whether the endpoint's credential makes it usable.""" diff --git a/functions/compose-model-replica/tests/test_backends.py b/functions/compose-model-replica/tests/test_backends.py index da5194def..b60a29d4b 100644 --- a/functions/compose-model-replica/tests/test_backends.py +++ b/functions/compose-model-replica/tests/test_backends.py @@ -12,1401 +12,6974 @@ # See the License for the specific language governing permissions and # limitations under the License. -"""Tests for compose-model-replica backends. - -A backend builds the workload (Deployment, LeaderWorkerSet, or PodCliqueSet) and the -ResourceClaimTemplates for one worker engine; the InferencePool, endpoint picker, -and HTTPRoute that front a replica's engines are built by routing.apply. Manifests are -asserted with a `Case` table: each case builds an engine's backend and compares -the composed manifests to a full `want`. Backend selection and serving are -dispatch/behaviour tests below the table. +"""Tests for compose-model-replica's backends and routing. + +A backend builds the workload (Deployment, LeaderWorkerSet, or PodCliqueSet) and +the ResourceClaimTemplates for one worker engine. routing.apply fronts a +replica's engines with an InferencePool, endpoint picker and HTTPRoute. """ import dataclasses -from typing import Any +import json import pytest -from crossplane.function import resource from function import routing from function.backends import base, grove, llmd, native -from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 from models.ai.modelplane.modelreplica import v1alpha1 from models.io.crossplane.m.kubernetes.object import v1alpha1 as k8sobjv1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -_SERVING = "modelplane.ai/serving" -_WORKLOAD = "modelplane.ai/workload" -_CLIQUE_ROLE = "modelplane.ai/clique-role" -_QUEUE_LABEL = "kai.scheduler/queue" -_QUEUE = "modelplane" -_SCHEDULER = "kai-scheduler" -# A GPU device request (claim: DRA), as compose-model-deployment stamps it. -_GPU_CEL = 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' +@dataclasses.dataclass +class BuildCase: + """A test case for a backend's build.""" + name: str + backend: base.Backend + replica: v1alpha1.ModelReplica + provider_config: str + serving_label: str + stack: str + want: dict[str, dict] -def _gpu_request(count: int) -> v1alpha1.DeviceRequest: - return v1alpha1.DeviceRequest( - name="gpu", - deviceClassName="gpu.nvidia.com", - count=count, - selectors=[v1alpha1.Selector(cel=_GPU_CEL)], - ) +@dataclasses.dataclass +class SelectBackendCase: + """A test case for base.select_backend.""" -def _standalone_engine( - name: str = "main", - *, - copies: int = 1, - args: list[str] | None = None, - command: list[str] | None = None, - device_requests: list[v1alpha1.DeviceRequest] | None = None, -) -> v1alpha1.Engine: - """A single Standalone-member engine.""" - container = v1alpha1.Container( - name="engine", - image="vllm/vllm-openai:latest", - args=args if args is not None else ["--model=Qwen/Qwen3-0.6B"], - ) - if command is not None: - container.command = command - return v1alpha1.Engine( - name=name, - copies=copies, - members=[ - v1alpha1.Member( - role="Standalone", - nodePoolName="frontier", - deviceRequests=device_requests if device_requests is not None else [_gpu_request(1)], - template=v1alpha1.Template(spec=v1alpha1.Spec(containers=[container])), - ), - ], - ) + name: str + engine: v1alpha1.Engine + stack: str + want: str -def _gang_engine( - name: str = "main", - *, - copies: int = 1, - nodes: int = 1, - leader_args: list[str] | None = None, - leader_command: list[str] | None = None, - worker_args: list[str] | None = None, - worker_command: list[str] | None = None, - leader_device_requests: list[v1alpha1.DeviceRequest] | None = None, - leader_pool: str = "frontier", -) -> v1alpha1.Engine: - """A Leader + Worker engine. - - The members carry their own pool pins and device requests, defaulting to a - homogeneous gang on one pool. leader_device_requests=[] makes the leader - claimless (a coordinator-only leader); leader_pool moves it to another - pool. - """ - - def member( - role: str, - nodes: int | None, - args: list[str] | None, - command: list[str] | None, - device_requests: list[v1alpha1.DeviceRequest], - pool: str, - ) -> v1alpha1.Member: - container = v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest") - if args is not None: - container.args = args - if command is not None: - container.command = command - kwargs: dict[str, Any] = { - "role": role, - "nodePoolName": pool, - "template": v1alpha1.Template(spec=v1alpha1.Spec(containers=[container])), - } - if device_requests: - kwargs["deviceRequests"] = device_requests - if nodes is not None: - kwargs["worker"] = v1alpha1.Worker(nodes=nodes) - return v1alpha1.Member(**kwargs) - - leader_requests = leader_device_requests if leader_device_requests is not None else [_gpu_request(8)] - return v1alpha1.Engine( - name=name, - copies=copies, - members=[ - member("Leader", None, leader_args, leader_command, leader_requests, leader_pool), - member("Worker", nodes, worker_args, worker_command, [_gpu_request(8)], "frontier"), - ], - ) +@dataclasses.dataclass +class CacheMountsCase: + """A test case for base.cache_mounts.""" + name: str + replica: v1alpha1.ModelReplica + want: tuple[list[dict], list[dict]] + + +@dataclasses.dataclass +class CacheEnvCase: + """A test case for base.cache_env.""" + + name: str + replica: v1alpha1.ModelReplica + want: list[dict] + + +@dataclasses.dataclass +class ApplyCase: + """A test case for routing.apply.""" + + name: str + composed: dict[str, dict] + replica: v1alpha1.ModelReplica + provider_config: str + want: dict[str, dict] -def _replica( - name: str = "r", *, namespace: str = "ml-team", engines: list[v1alpha1.Engine] | None = None -) -> v1alpha1.ModelReplica: - if engines is None: - engines = [_standalone_engine()] - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name=name, namespace=namespace), - spec=v1alpha1.SpecModel(clusterName="cluster-a", engines=engines), - ) +@dataclasses.dataclass +class KvBlockSizeCase: + """A test case for routing._kv_block_size.""" + + name: str + engine_args: list[str] + want: int + + +@dataclasses.dataclass +class DisaggregatedEppConfigYamlCase: + """A test case for routing._disaggregated_epp_config_yaml.""" + + name: str + block_size: int + want: str -# The composed workload name for the default replica "r" / engine "main": -# engine-qualified so a multi-engine replica's workloads don't collide. -_WORKLOAD_NAME = resource.child_name("r", "main") -# The Grove PodCliqueSet name for the same replica/engine, budgeted tighter -# than _WORKLOAD_NAME per base.grove_pcs_name. -_GROVE_PCS_NAME = base.grove_pcs_name(_replica(), _gang_engine()) +@dataclasses.dataclass +class RemoteNamespaceCase: + """A test case for base.remote_namespace.""" + + name: str + replica: v1alpha1.ModelReplica + want: str -def _claim_template(count: int, *, replica: str = "r", engine: str = "main", role: str = "standalone") -> dict: - """The ResourceClaimTemplate manifest a member's device requests produce.""" +def _route() -> dict: + """The Object composing the replica's HTTPRoute to its InferencePool.""" return { - "apiVersion": "resource.k8s.io/v1", - "kind": "ResourceClaimTemplate", - "metadata": {"name": resource.child_name(replica, engine, role, "devices"), "namespace": "mp-ml-team-51733"}, "spec": { - "spec": { - "devices": { - "requests": [ - { - "name": "gpu", - "exactly": { - "deviceClassName": "gpu.nvidia.com", - "count": count, - "selectors": [{"cel": {"expression": _GPU_CEL}}], - }, - } - ] + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "HTTPRoute", + "metadata": {"name": "r", "namespace": "mp-ml-team-51733"}, + "spec": { + "parentRefs": [{"name": "cluster-gateway", "namespace": "modelplane-system"}], + "rules": [ + { + "matches": [{"path": {"type": "PathPrefix", "value": "/ml-team/r/"}}], + # No request timeout, so long token streams + # aren't severed. + "timeouts": {"request": "0s"}, + "filters": [ + { + "type": "URLRewrite", + "urlRewrite": { + "path": {"type": "ReplacePrefixMatch", "replacePrefixMatch": "/"} + }, + } + ], + "backendRefs": [ + { + "group": "inference.networking.k8s.io", + "kind": "InferencePool", + "name": "r-pool", + } + ], + } + ], + }, } - } - }, + }, + } } -_CLUSTER = icv1alpha1.InferenceCluster( - metadata=metav1.ObjectMeta(name="cluster-a"), - spec=icv1alpha1.Spec( - cluster=icv1alpha1.Cluster( - source="Existing", existing=icv1alpha1.Existing(secretRef=icv1alpha1.SecretRef(name="k")) - ) - ), - status=icv1alpha1.Status(providerConfigRef=icv1alpha1.ProviderConfigRef(name="cluster-a-pc")), -) +def _inference_pool() -> dict: + """The Object composing the InferencePool that fronts the replica's serving pods.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "inference.networking.k8s.io/v1", + "kind": "InferencePool", + "metadata": {"name": "r-pool", "namespace": "mp-ml-team-51733"}, + "spec": { + "selector": {"matchLabels": {"modelplane.ai/serving": "r"}}, + "targetPorts": [{"number": 8000}], + "endpointPickerRef": { + "name": "r-epp", + "port": {"number": 9002}, + "failureMode": "FailOpen", + }, + }, + } + }, + } + } + + +def _epp(*, config_checksum: str) -> dict: + """The Object composing the endpoint picker's Deployment, rolled by its config's checksum.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": "r-epp", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": 1, + "selector": {"matchLabels": {"app": "r-epp"}}, + "template": { + "metadata": { + "labels": {"app": "r-epp"}, + "annotations": {"modelplane.ai/epp-config-checksum": config_checksum}, + }, + "spec": { + "serviceAccountName": "r-epp", + "containers": [ + { + "name": "epp", + "image": "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0", + "args": [ + "--pool-name=r-pool", + "--pool-namespace=mp-ml-team-51733", + "--pool-group=inference.networking.k8s.io", + "--config-file=/config/epp-config.yaml", + "--grpc-port=9002", + ], + "ports": [ + {"name": "grpc", "containerPort": 9002}, + {"name": "grpc-health", "containerPort": 9003}, + ], + "volumeMounts": [{"name": "config", "mountPath": "/config"}], + } + ], + "volumes": [{"name": "config", "configMap": {"name": "r-epp"}}], + }, + }, + }, + } + }, + } + } -_PC = "cluster-a-pc" +def _epp_config(*, config: str) -> dict: + """The Object composing the ConfigMap that holds the endpoint picker's config.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "r-epp", "namespace": "mp-ml-team-51733"}, + "data": {"epp-config.yaml": config}, + } + }, + } + } -_NATIVE_WANT = { - "model-serving-main": { - "apiVersion": "apps/v1", - "kind": "Deployment", - "metadata": {"name": _WORKLOAD_NAME, "namespace": "mp-ml-team-51733"}, + +def _epp_role() -> dict: + """The Object composing the endpoint picker's Role.""" + return { "spec": { - "replicas": 1, - "selector": {"matchLabels": {_WORKLOAD: _WORKLOAD_NAME}}, - "template": { - "metadata": {"labels": {_SERVING: "r", _WORKLOAD: _WORKLOAD_NAME}}, - "spec": { - "containers": [ + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "Role", + "metadata": {"name": "r-epp", "namespace": "mp-ml-team-51733"}, + "rules": [ + {"apiGroups": [""], "resources": ["pods"], "verbs": ["get", "watch", "list"]}, { - "name": "engine", - "image": "vllm/vllm-openai:latest", - "args": ["--model=Qwen/Qwen3-0.6B"], - "ports": [{"containerPort": 8000}], - "resources": {"claims": [{"name": "devices"}]}, - "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], - "readinessProbe": { - "httpGet": {"path": "/health", "port": 8000}, - "initialDelaySeconds": 30, - "periodSeconds": 10, - "timeoutSeconds": 5, - }, - } - ], - "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], - "nodeSelector": {"modelplane.ai/pool": "frontier"}, - "resourceClaims": [ + "apiGroups": ["inference.networking.k8s.io"], + "resources": ["inferencepools"], + "verbs": ["get", "watch", "list"], + }, { - "name": "devices", - "resourceClaimTemplateName": resource.child_name("r", "main", "standalone", "devices"), - } + "apiGroups": ["inference.networking.x-k8s.io"], + "resources": ["inferenceobjectives"], + "verbs": ["get", "watch", "list"], + }, ], - "tolerations": [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}], - }, + } }, - }, - }, - "resource-claim-main-standalone": _claim_template(1), -} + } + } -def _claims(role: str) -> list[dict]: - """The pod-level claim referencing a member's ResourceClaimTemplate.""" - return [ - { - "name": "devices", - "resourceClaimTemplateName": resource.child_name("r", "main", role, "devices"), +def _epp_role_binding() -> dict: + """The Object composing the endpoint picker's RoleBinding.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "RoleBinding", + "metadata": {"name": "r-epp", "namespace": "mp-ml-team-51733"}, + "subjects": [{"kind": "ServiceAccount", "name": "r-epp", "namespace": "mp-ml-team-51733"}], + "roleRef": {"apiGroup": "rbac.authorization.k8s.io", "kind": "Role", "name": "r-epp"}, + } + }, } - ] - + } -def _clique(manifest: dict, name: str) -> dict: - """The named clique from a PodCliqueSet manifest.""" - return next(c for c in manifest["spec"]["template"]["cliques"] if c["name"] == name) +def _epp_service_account() -> dict: + """The Object composing the endpoint picker's ServiceAccount.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ServiceAccount", + "metadata": {"name": "r-epp", "namespace": "mp-ml-team-51733"}, + } + }, + } + } -def _pcs(leader_container: dict, worker_container: dict, *, worker_replicas: int = 1, copies: int = 1) -> dict: - node_selector = {"modelplane.ai/pool": "frontier"} - tolerations = [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}] - def pod_spec(container: dict, role: str) -> dict: - return { - "containers": [container], - "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], - "schedulerName": _SCHEDULER, - "nodeSelector": node_selector, - "resourceClaims": _claims(role), - "tolerations": tolerations, +def _epp_service() -> dict: + """The Object composing the endpoint picker's Service.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Service", + "metadata": {"name": "r-epp", "namespace": "mp-ml-team-51733"}, + "spec": { + "selector": {"app": "r-epp"}, + "ports": [{"name": "grpc-ext-proc", "port": 9002, "targetPort": 9002, "appProtocol": "http2"}], + }, + } + }, } + } + +def _main_deployment( + *, + name: str, + claim_template_name: str, + replicas: int, + pod_metadata: dict, + containers: list[dict], + volumes: list[dict], +) -> dict: + """The Object composing the main engine's Deployment.""" return { - "apiVersion": "grove.io/v1alpha1", - "kind": "PodCliqueSet", - "metadata": {"name": _GROVE_PCS_NAME, "namespace": "mp-ml-team-51733"}, "spec": { - "replicas": 1, - "template": { - "cliqueStartupType": "CliqueStartupTypeExplicit", - "terminationDelay": "4h", - "headlessServiceConfig": {"publishNotReadyAddresses": True}, - "cliques": [ - { - "name": "leader", - "labels": {_SERVING: "r", _QUEUE_LABEL: _QUEUE, _CLIQUE_ROLE: "leader"}, - "spec": { - "roleName": "leader", - "replicas": 1, - "minAvailable": 1, - "podSpec": pod_spec(leader_container, "leader"), + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": name, "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": replicas, + "selector": {"matchLabels": {"modelplane.ai/workload": name}}, + "template": { + "metadata": pod_metadata, + "spec": { + "containers": containers, + "volumes": volumes, + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": claim_template_name, + } + ], + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ], + }, }, }, - { - "name": "worker", - "labels": {_QUEUE_LABEL: _QUEUE}, - "spec": { - "roleName": "worker", - "replicas": worker_replicas, - "minAvailable": worker_replicas, - "podSpec": pod_spec(worker_container, "worker"), + } + }, + } + } + + +def _prefill_deployment(*, labels: dict, containers: list[dict], volumes: list[dict]) -> dict: + """The Object composing the prefill engine's Deployment.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": "r-prefill-d90b0", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": 1, + "selector": {"matchLabels": {"modelplane.ai/workload": "r-prefill-d90b0"}}, + "template": { + "metadata": {"labels": labels}, + "spec": { + "containers": containers, + "volumes": volumes, + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-prefill-standalone-devices-f62af", + } + ], + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ], + }, }, }, - ], - "podCliqueScalingGroups": [ - { - "name": "gang", - "cliqueNames": ["leader", "worker"], - "replicas": copies, - # 1 regardless of copies, so a wedged gang doesn't take - # the healthy ones down with it. - "minAvailable": 1, - } - ], + } }, - }, + } } -def _engine( - *, serving: bool, args: list[str] | None = None, command: list[str] | None = None, env: list[dict] | None = None -) -> dict[str, Any]: - c: dict[str, Any] = { - "name": "engine", - "image": "vllm/vllm-openai:latest", - "resources": {"claims": [{"name": "devices"}]}, - "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], - } - if command is not None: - c["command"] = command - if args is not None: - c["args"] = args - # A container carries an env only when the test gives it one. The Grove - # backend always injects a leader-address alias (see grove.py); callers - # composing a Grove _GROVE_WANT container pass it explicitly. - if env is not None: - c["env"] = env - if serving: - c["ports"] = [{"containerPort": 8000}] - c["readinessProbe"] = { - "httpGet": {"path": "/health", "port": 8000}, - "initialDelaySeconds": 30, - "periodSeconds": 10, - "timeoutSeconds": 5, +def _decode_deployment(*, labels: dict, containers: list[dict], volumes: list[dict]) -> dict: + """The Object composing the decode engine's Deployment.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": "r-decode-4b27b", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": 1, + "selector": {"matchLabels": {"modelplane.ai/workload": "r-decode-4b27b"}}, + "template": { + "metadata": {"labels": labels}, + "spec": { + "containers": containers, + "volumes": volumes, + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-decode-standalone-devices-63392", + } + ], + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ], + }, + }, + }, + } + }, } - return c + } -# A multi-node engine with verbatim leader/worker commands - no flag injection, -# no bootstrap. The follower addresses the leader through -# $(MODELPLANE_LEADER_ADDRESS). -_LEADER_CMD = [ - "/bin/sh", - "-c", - "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B " - "--tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", -] -_WORKER_CMD = ["/bin/sh", "-c", "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block"] -_GROVE_WANT = { - "model-serving-main": _pcs( - _engine(serving=True, command=_LEADER_CMD, env=[base.grove_leader_address_env()]), - _engine(serving=False, command=_WORKER_CMD, env=[base.grove_leader_address_env()]), - ), - "resource-claim-main-leader": _claim_template(8, role="leader"), - "resource-claim-main-worker": _claim_template(8, role="worker"), -} +def _main_leader_worker_set( + *, + replicas: int, + size: int, + leader_containers: list[dict], + leader_volumes: list[dict], + worker_containers: list[dict], + worker_volumes: list[dict], +) -> dict: + """The Object composing the main engine's LeaderWorkerSet.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "leaderworkerset.x-k8s.io/v1", + "kind": "LeaderWorkerSet", + "metadata": {"name": "r-main-bb4e3", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": replicas, + "leaderWorkerTemplate": { + "size": size, + "leaderTemplate": { + "metadata": { + "labels": {"modelplane.ai/serving": "r", "modelplane.ai/lws-role": "leader"} + }, + "spec": { + "containers": leader_containers, + "volumes": leader_volumes, + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ], + }, + }, + "workerTemplate": { + "spec": { + "containers": worker_containers, + "volumes": worker_volumes, + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ], + } + }, + }, + }, + } + }, + } + } -@dataclasses.dataclass -class Case: - name: str - backend: base.Backend - engine: v1alpha1.Engine - want: dict - stack: str = "Standard" +def _main_pod_clique_set(*, cliques: list[dict]) -> dict: + """The Object composing the main engine's PodCliqueSet, from its leader and worker cliques.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status) && has(object.status.observedGeneration) && object.status.observedGeneration == object.metadata.generation && object.spec.replicas > 0 && has(object.status.availableReplicas) && object.status.availableReplicas >= object.spec.replicas", + }, + "forProvider": { + "manifest": { + "apiVersion": "grove.io/v1alpha1", + "kind": "PodCliqueSet", + "metadata": {"name": "r-main-bb4e3", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": 1, + "template": { + "cliqueStartupType": "CliqueStartupTypeExplicit", + "terminationDelay": "4h", + "headlessServiceConfig": {"publishNotReadyAddresses": True}, + "cliques": cliques, + "podCliqueScalingGroups": [ + { + "name": "gang", + "cliqueNames": ["leader", "worker"], + "replicas": 1, + # 1 whatever the copies, so a wedged gang + # doesn't take the healthy ones down with it. + "minAvailable": 1, + } + ], + }, + }, + } + }, + } + } -MANIFESTS_CASES = [ - Case( - name="native Standalone engine composes a Deployment", - backend=native.NativeBackend(), - engine=_standalone_engine(), - want=_NATIVE_WANT, - ), - Case( - name="Grove Leader/Worker engine composes a PodCliqueSet, commands verbatim", - backend=grove.GroveBackend(), - engine=_gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD), - want=_GROVE_WANT, - stack="Dynamo", - ), -] +def _main_standalone_claim_template(*, name: str, requests: list[dict]) -> dict: + """The Object composing the ResourceClaimTemplate for the main engine's Standalone member.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": name, "namespace": "mp-ml-team-51733"}, + "spec": {"spec": {"devices": {"requests": requests}}}, + } + }, + } + } -@pytest.mark.parametrize("case", MANIFESTS_CASES, ids=lambda case: case.name) -def test_manifests(case: Case) -> None: - """A backend composes an engine's manifests.""" - replica = _replica(engines=[case.engine]) - out = case.backend.build(replica, case.engine, _PC, base.serving_label(replica), case.stack) - got = {key: obj.spec.forProvider.manifest for key, obj in out.items()} - assert got == case.want - - -def test_leader_address_env_injected_but_not_rank() -> None: - """The Grove backend injects a leader address alias, but no rank.""" - # The Grove backend injects MODELPLANE_LEADER_ADDRESS (aliasing Grove's - # own GROVE_PCSG_* vars) but not MODELPLANE_RANK: Grove exposes no - # group-wide pod index yet (grove#755, open), so a gang engine's - # command computes its own rank from GROVE_PCLQ_POD_INDEX directly - # (see grove.py and the multinode example). - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - # Spelled out rather than compared against grove_leader_address_env(), - # which would pass whatever that function returned. The PCSG vars are - # what make the address vary per gang; the PCS-scoped ones are - # identical across gangs and would silently point every copy at gang - # 0's leader. - want = { - "name": "MODELPLANE_LEADER_ADDRESS", - "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", +def _main_leader_claim_template() -> dict: + """The Object composing the ResourceClaimTemplate for the main engine's leader, for 8 GPUs.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": "r-main-leader-devices-f58c6", "namespace": "mp-ml-team-51733"}, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ] + } + } + }, + } + }, + } } - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - assert container["env"] == [want] - - -def test_user_env_passed_through() -> None: - """A Grove member's own env follows the leader address alias.""" - # A member's own env passes through verbatim, after the leader-address - # alias (see test_leader_address_env_injected_but_not_rank). - engine = _gang_engine( - leader_command=_LEADER_CMD, - worker_command=_WORKER_CMD, - ) - spec = engine.members[0].template.spec - assert spec is not None - spec.containers[0].env = [v1alpha1.EnvItem(name="HF_TOKEN", value="x")] - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader")["spec"]["podSpec"] - env = leader["containers"][0]["env"] - assert env == [base.grove_leader_address_env(), {"name": "HF_TOKEN", "value": "x"}] - - -def test_fieldref_env_passes_through() -> None: - """A Grove member's pod-field env survives into the composed manifest.""" - # A pod-field env (e.g. VLLM_HOST_IP from status.podIP, which multi-NIC - # RDMA nodes need so the engine binds the right interface — #141) survives - # model_dump into the composed manifest. - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - spec = engine.members[0].template.spec - assert spec is not None - spec.containers[0].env = [ - v1alpha1.EnvItem( - name="VLLM_HOST_IP", - valueFrom=v1alpha1.ValueFrom(fieldRef=v1alpha1.FieldRef(fieldPath="status.podIP")), - ) - ] - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader")["spec"]["podSpec"] - env = leader["containers"][0]["env"] - assert env == [ - base.grove_leader_address_env(), - {"name": "VLLM_HOST_IP", "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}}, - ] - - -def test_member_metadata_propagates_to_native_pod_template() -> None: - """A Standalone member's template metadata lands on the Deployment's pod template.""" - # A Standalone member's template.metadata labels and annotations land - # on the Deployment's pod template, merged with the managed labels - # (#378). - engine = _standalone_engine() - engine.members[0].template.metadata = v1alpha1.Metadata( - labels={"example.com/role": "standalone"}, - annotations={"example.com/config": "standalone"}, - ) - replica = _replica(engines=[engine]) - out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - meta = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["metadata"] - assert meta["labels"] == {"example.com/role": "standalone", _SERVING: "r", _WORKLOAD: _WORKLOAD_NAME} - assert meta["annotations"] == {"example.com/config": "standalone"} - - -def test_member_metadata_propagates_to_cliques_independently() -> None: - """Each Grove member's template metadata lands on its own clique only.""" - # Leader metadata lands on the leader clique and worker metadata on the - # worker clique; neither leaks into the other. Grove propagates a - # clique's labels and annotations to its pods (#378). - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - engine.members[0].template.metadata = v1alpha1.Metadata( - labels={"example.com/role": "leader"}, annotations={"example.com/config": "leader"} - ) - engine.members[1].template.metadata = v1alpha1.Metadata( - labels={"example.com/role": "worker"}, annotations={"example.com/config": "worker"} - ) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader") - assert leader["labels"] == { - "example.com/role": "leader", - _SERVING: "r", - _QUEUE_LABEL: _QUEUE, - _CLIQUE_ROLE: "leader", + + +def _main_worker_claim_template() -> dict: + """The Object composing the ResourceClaimTemplate for the main engine's worker, for 8 GPUs.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": "r-main-worker-devices-99b8a", "namespace": "mp-ml-team-51733"}, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ] + } + } + }, + } + }, + } } - assert leader["annotations"] == {"example.com/config": "leader"} - worker = _clique(manifest, "worker") - assert worker["labels"] == {"example.com/role": "worker", _QUEUE_LABEL: _QUEUE} - assert worker["annotations"] == {"example.com/config": "worker"} - - -def test_worker_without_metadata_composes_only_managed_labels() -> None: - """A Grove worker with no template metadata carries only the queue label.""" - # A worker member with no template.metadata composes a worker clique - # carrying only the managed queue label and no annotations key. - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - worker = _clique(manifest, "worker") - assert worker["labels"] == {_QUEUE_LABEL: _QUEUE} - assert "annotations" not in worker - - -def _names(out: dict[str, k8sobjv1alpha1.Object]) -> set[str]: - """The names of the manifests a backend composed.""" - return {o.spec.forProvider.manifest["metadata"]["name"] for o in out.values()} - - -def test_co_located_replicas_get_distinct_names() -> None: - """Two replicas on one cluster compose distinct resource names.""" - # Two replicas of one deployment on the same cluster must produce - # distinct resource names on the remote cluster. - a = _replica("dep-clusterA") - b = _replica("dep-clusterB") - out_a = native.NativeBackend().build(a, a.spec.engines[0], _PC, base.serving_label(a), "Standard") - out_b = native.NativeBackend().build(b, b.spec.engines[0], _PC, base.serving_label(b), "Standard") - assert _names(out_a) & _names(out_b) == set() - - -def test_multi_engine_qualifies_workload_names() -> None: - """Each engine of a multi-engine replica composes distinctly named resources.""" - # A replica with two engines names each engine's workload distinctly so - # they don't collide on the remote cluster. - engines = [_standalone_engine("prefill"), _standalone_engine("decode")] - replica = _replica(engines=engines) - names = set() - for g in engines: - out = native.NativeBackend().build(replica, g, _PC, base.serving_label(replica), "Standard") - names |= _names(out) - assert len(names) == 4 # 2 deployments + 2 claim templates - - -@pytest.mark.parametrize( - ("backend", "engine", "stack", "want_cel"), - [ - pytest.param(native.NativeBackend(), _standalone_engine(), "Standard", base.AVAILABLE_CEL, id="native"), - pytest.param( - grove.GroveBackend(), - _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD), - "Dynamo", - base.GROVE_AVAILABLE_CEL, - id="grove", - ), - ], -) -def test_workload_readiness_policies(backend: base.Backend, engine: v1alpha1.Engine, stack: str, want_cel: str) -> None: - """A workload's readiness derives from its status, and a claim template's from its creation.""" - # A Deployment reports readiness from its Available condition; a - # PodCliqueSet publishes no such condition, so it's derived from its - # replica counters instead (base.GROVE_AVAILABLE_CEL). Either way the - # claim templates are ready on create. - replica = _replica(engines=[engine]) - out = backend.build(replica, engine, _PC, base.serving_label(replica), stack) - serving = out["model-serving-main"].spec.readiness - assert serving is not None - assert serving.policy == "DeriveFromCelQuery" - assert serving.celQuery == want_cel - for key, obj in out.items(): - if key.startswith("resource-claim"): - readiness = obj.spec.readiness - assert readiness is not None - assert readiness.policy == "SuccessfulCreate" - - -def test_multiple_device_requests_single_container_claim() -> None: - """Several device requests compose one container claim and one template carrying them all.""" - # resources.claims is a list-map keyed on name alone, so N device - # requests must NOT produce N container claims all named "devices". The - # container references the whole pod claim once; the template carries all - # requests. - engine = _standalone_engine( - device_requests=[ - v1alpha1.DeviceRequest(name="gpu", deviceClassName="gpu.nvidia.com", count=8), - v1alpha1.DeviceRequest(name="nic", deviceClassName="nic.nvidia.com", count=8), - ], - ) - replica = _replica(engines=[engine]) - out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - pod = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"] - claims = pod["containers"][0]["resources"]["claims"] - assert claims == [{"name": "devices"}] - assert pod["resourceClaims"][0]["name"] == "devices" - template = out["resource-claim-main-standalone"].spec.forProvider.manifest - template_requests = template["spec"]["spec"]["devices"]["requests"] - assert [r["name"] for r in template_requests] == ["gpu", "nic"] - claim_readiness = out["resource-claim-main-standalone"].spec.readiness - assert claim_readiness is not None - assert claim_readiness.policy == "SuccessfulCreate" - - -def test_claimless_leader_gets_no_claim() -> None: - """A Grove leader with no device requests composes no claim, but still pins and tolerates.""" - # A coordinator-only leader (e.g. a vLLM DP head running - # --data-parallel-size-local=0) carries no deviceRequests. Its pod must - # get no resourceClaims, its container no resources.claims, and no - # leader ResourceClaimTemplate must be composed - only the worker's. - # It still pins to its pool and tolerates the GPU taint. - engine = _gang_engine( - leader_command=_LEADER_CMD, - worker_command=_WORKER_CMD, - leader_device_requests=[], - ) - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - - assert "resource-claim-main-leader" not in out - assert "resource-claim-main-worker" in out - - manifest = out["model-serving-main"].spec.forProvider.manifest - leader = _clique(manifest, "leader")["spec"]["podSpec"] - assert "resourceClaims" not in leader - assert "resources" not in leader["containers"][0] - assert leader["nodeSelector"] == {"modelplane.ai/pool": "frontier"} - assert leader["tolerations"] == [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}] - - worker = _clique(manifest, "worker")["spec"]["podSpec"] - assert worker["resourceClaims"] == _claims("worker") - assert worker["containers"][0]["resources"] == {"claims": [{"name": "devices"}]} - - -def test_members_pin_to_their_own_pools() -> None: - """Each Grove member's pods pin to that member's own pool.""" - # The scheduler may split a gang across pools when no single pool - # satisfies every member. Each member's pods must pin to that member's - # pool, not a shared engine-wide one. - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD, leader_pool="head") - replica = _replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - assert _clique(manifest, "leader")["spec"]["podSpec"]["nodeSelector"] == {"modelplane.ai/pool": "head"} - assert _clique(manifest, "worker")["spec"]["podSpec"]["nodeSelector"] == {"modelplane.ai/pool": "frontier"} - - -# The LeaderWorkerSet backend for a Leader/Worker gang engine. - -_LWS_ROLE = "modelplane.ai/lws-role" - - -def _llmd_lws(engine: v1alpha1.Engine, replica: v1alpha1.ModelReplica) -> dict: - """The LeaderWorkerSet manifest the llm-d backend composes for engine.""" - out = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - return out["model-serving-main"].spec.forProvider.manifest - - -def test_llmd_leader_worker_set_shape() -> None: - """The llm-d backend composes a LeaderWorkerSet of copies gangs, each the leader plus its workers.""" - engine = _gang_engine(nodes=3, copies=2) - replica = _replica(engines=[engine]) - manifest = _llmd_lws(engine, replica) - assert manifest["apiVersion"] == "leaderworkerset.x-k8s.io/v1" - assert manifest["kind"] == "LeaderWorkerSet" - assert manifest["metadata"] == {"name": _WORKLOAD_NAME, "namespace": "mp-ml-team-51733"} - assert manifest["spec"]["replicas"] == 2 - # Gang size is the leader plus the worker's node count. - assert manifest["spec"]["leaderWorkerTemplate"]["size"] == 4 - - -def test_llmd_only_leader_carries_serving_label() -> None: - """Only the LeaderWorkerSet's leader carries the serving label.""" - engine = _gang_engine() - replica = _replica(engines=[engine]) - lwt = _llmd_lws(engine, replica)["spec"]["leaderWorkerTemplate"] - leader_labels = lwt["leaderTemplate"]["metadata"]["labels"] - assert leader_labels[_SERVING] == "r" - assert leader_labels[_LWS_ROLE] == "leader" - # The worker followers never serve, so they carry no metadata at all. - assert "metadata" not in lwt["workerTemplate"] - - -def test_llmd_leader_address_and_rank_env_injected() -> None: - """Every LeaderWorkerSet container leads with the leader address and rank aliases.""" - # Every gang container leads with the backend-neutral coordination vars - # aliasing LWS_LEADER_ADDRESS / LWS_WORKER_INDEX. - engine = _gang_engine() - replica = _replica(engines=[engine]) - lwt = _llmd_lws(engine, replica)["spec"]["leaderWorkerTemplate"] - for tmpl in (lwt["leaderTemplate"], lwt["workerTemplate"]): - env = tmpl["spec"]["containers"][0]["env"] - assert env[0] == {"name": "MODELPLANE_LEADER_ADDRESS", "value": "$(LWS_LEADER_ADDRESS)"} - assert env[1] == {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"} - - -def test_llmd_no_modelexpress_env_even_with_a_cache() -> None: - """The llm-d backend injects no ModelExpress env, even for a replica with a cache.""" - # The llm-d (Standard) backend never injects ModelExpress env: that P2P - # wiring is the Grove (Dynamo) backend's, gated on the cluster stack. - engine = _gang_engine() - replica = v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="c"), - engines=[engine], - ), - ) - lwt = _llmd_lws(engine, replica)["spec"]["leaderWorkerTemplate"] - for tmpl in (lwt["leaderTemplate"], lwt["workerTemplate"]): - container = tmpl["spec"]["containers"][0] - env_names = [e["name"] for e in container["env"]] - # HF_HUB_CACHE is the cache's own env (every stack); the MX bundle - # is not. - assert env_names == ["MODELPLANE_LEADER_ADDRESS", "MODELPLANE_RANK", "HF_HUB_CACHE"] - assert "MX_SERVER_ADDRESS" not in env_names - assert "securityContext" not in container - - -def test_llmd_workload_readiness_uses_available_cel() -> None: - """The LeaderWorkerSet's readiness derives from its Available condition.""" - engine = _gang_engine() - replica = _replica(engines=[engine]) - out = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - readiness = out["model-serving-main"].spec.readiness - assert readiness is not None - assert readiness.policy == "DeriveFromCelQuery" - assert readiness.celQuery == base.AVAILABLE_CEL - - -def test_select_backend_standalone_engine_is_native() -> None: - """A Standalone engine selects the native backend.""" - # A Standalone engine is native regardless of the cluster's stack. - assert base.select_backend(_standalone_engine(), "Standard") == base.NATIVE - assert base.select_backend(_standalone_engine(), "Dynamo") == base.NATIVE -def test_select_backend_leader_worker_engine_is_llmd() -> None: - """A Leader/Worker engine on a Standard cluster selects the llm-d backend.""" - assert base.select_backend(_gang_engine(), "Standard") == base.LLMD +def _prefill_claim_template() -> dict: + """The Object composing the ResourceClaimTemplate for the prefill engine's member, for 1 GPU.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": "r-prefill-standalone-devices-f62af", "namespace": "mp-ml-team-51733"}, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ] + } + } + }, + } + }, + } + } -def test_select_backend_leader_worker_engine_is_grove() -> None: - """A Leader/Worker engine on a Dynamo cluster selects the Grove backend.""" - assert base.select_backend(_gang_engine(), "Dynamo") == base.GROVE +def _decode_claim_template() -> dict: + """The Object composing the ResourceClaimTemplate for the decode engine's member, for 1 GPU.""" + return { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": "r-decode-standalone-devices-63392", "namespace": "mp-ml-team-51733"}, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ] + } + } + }, + } + }, + } + } -def _cache_replica( - *, cache: str | None = None, args: list[str] | None = None, command: list[str] | None = None +def _replica( + *, + name: str, + namespace: str, + model_cache_ref: v1alpha1.ModelCacheRef | None, + serving: v1alpha1.Serving | None, + engines: list[v1alpha1.Engine], ) -> v1alpha1.ModelReplica: - """A replica with one Standalone engine, referencing cache if one's given.""" - engine = _standalone_engine(args=args or [], command=command) - modelcache = v1alpha1.ModelCacheRef(name=cache) if cache else None + """A ModelReplica on cluster-a.""" return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(namespace="ml-team"), - spec=v1alpha1.SpecModel(clusterName="c", modelCacheRef=modelcache, engines=[engine]), + metadata=metav1.ObjectMeta(name=name, namespace=namespace), + spec=v1alpha1.SpecModel( + clusterName="cluster-a", + modelCacheRef=model_cache_ref, + serving=serving, + engines=engines, + ), ) -def test_no_cache_no_mounts() -> None: - """A replica with no cache mounts nothing.""" - volumes, mounts = base.cache_mounts(_cache_replica()) - assert (volumes, mounts) == ([], []) - - -def test_cache_adds_volume_and_mount() -> None: - """A replica with a cache mounts the cache's PVC.""" - volumes, mounts = base.cache_mounts(_cache_replica(cache="qwen")) - assert volumes == [{"name": "model-cache", "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}}] - assert mounts == [{"name": "model-cache", "mountPath": "/mnt/models"}] - +def _unnamed_replica(*, model_cache_ref: v1alpha1.ModelCacheRef | None) -> v1alpha1.ModelReplica: + """A ModelReplica with no name in namespace ml-team, on cluster c, with one Standalone engine.""" + return v1alpha1.ModelReplica( + metadata=metav1.ObjectMeta(namespace="ml-team"), + spec=v1alpha1.SpecModel( + clusterName="c", + modelCacheRef=model_cache_ref, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest", args=[]) + ] + ) + ), + ) + ], + ) + ], + ), + ) -def test_cache_env_points_huggingface_at_the_mount() -> None: - """A replica with a cache points HF_HUB_CACHE at the mount.""" - # The cache is staged in HuggingFace's cache layout, so pointing - # HF_HUB_CACHE at the mount is what lets an engine's own --model= - # resolve against it instead of pulling from HuggingFace (#407). - assert base.cache_env(_cache_replica(cache="qwen")) == [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}] +def _to_dicts(objs: dict[str, k8sobjv1alpha1.Object]) -> dict[str, dict]: + """objs as dicts of only the fields the code set.""" + return {key: obj.model_dump(exclude_unset=True, by_alias=True) for key, obj in objs.items()} -def test_cache_env_empty_without_cache() -> None: - """A replica with no cache gets no cache env.""" - assert base.cache_env(_cache_replica()) == [] +def _sorted(d: dict) -> dict: + """d with its keys sorted, so pytest's diff of two lines them up.""" + return json.loads(json.dumps(d, sort_keys=True)) -def test_cache_env_sets_no_offline_flag() -> None: - """A replica with a cache doesn't set HF_HUB_OFFLINE.""" - # HF_HUB_OFFLINE would break an engine that fetches a *different* repo - # at startup (kimi-k2's separately-gated tokenizer), and resolution - # doesn't need it. - names = {e["name"] for e in base.cache_env(_cache_replica(cache="qwen"))} - assert "HF_HUB_OFFLINE" not in names +BUILD_CASES = [ + # A Deployment reports readiness from its Available condition, and a claim + # template is ready once it's created. The device request's CEL selector, + # here and throughout, is as compose-model-deployment stamps it. + BuildCase( + name="a Standalone engine composes a Deployment", + backend=native.NativeBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _main_deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _main_standalone_claim_template( + name="r-main-standalone-devices-f456f", + requests=[ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ], + ), + }, + ), + # The member's labels merge with the managed ones (#378). + BuildCase( + name="a Standalone member's template metadata lands on its pod template", + backend=native.NativeBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + metadata=v1alpha1.Metadata( + labels={"example.com/role": "standalone"}, + annotations={"example.com/config": "standalone"}, + ), + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ), + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _main_deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "example.com/role": "standalone", + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + }, + "annotations": {"example.com/config": "standalone"}, + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _main_standalone_claim_template( + name="r-main-standalone-devices-f456f", + requests=[ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ], + ), + }, + ), + # Two replicas of one deployment on the same cluster, this one and + # dep-clusterB, must compose distinct resource names there. + BuildCase( + name="a co-located replica's resource names are qualified by its own name (dep-clusterA)", + backend=native.NativeBackend(), + replica=_replica( + name="dep-clusterA", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="dep-clusterA", + stack="Standard", + want={ + "model-serving-main": _main_deployment( + name="dep-clusterA-main-3d1d5", + claim_template_name="dep-clusterA-main-standalone-devices-145eb", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/serving": "dep-clusterA", + "modelplane.ai/workload": "dep-clusterA-main-3d1d5", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _main_standalone_claim_template( + name="dep-clusterA-main-standalone-devices-145eb", + requests=[ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ], + ), + }, + ), + BuildCase( + name="a co-located replica's resource names are qualified by its own name (dep-clusterB)", + backend=native.NativeBackend(), + replica=_replica( + name="dep-clusterB", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="dep-clusterB", + stack="Standard", + want={ + "model-serving-main": _main_deployment( + name="dep-clusterB-main-d6c52", + claim_template_name="dep-clusterB-main-standalone-devices-5a8a8", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/serving": "dep-clusterB", + "modelplane.ai/workload": "dep-clusterB-main-d6c52", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _main_standalone_claim_template( + name="dep-clusterB-main-standalone-devices-5a8a8", + requests=[ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ], + ), + }, + ), + # Workload and claim template names are qualified by the engine, so a + # multi-engine replica's don't collide on the remote cluster. + BuildCase( + name="each engine of a multi-engine replica composes distinctly named resources", + backend=native.NativeBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="prefill", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + v1alpha1.Engine( + name="decode", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-prefill": _prefill_deployment( + labels={ + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-prefill-d90b0", + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-prefill-standalone": _prefill_claim_template(), + "model-serving-decode": _decode_deployment( + labels={ + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-decode-4b27b", + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-decode-standalone": _decode_claim_template(), + }, + ), + # resources.claims is a list-map keyed on name alone, so N device requests + # mustn't compose N container claims all named "devices". The container + # references the whole pod claim once, and the template carries every + # request. + BuildCase( + name="several device requests share one container claim", + backend=native.NativeBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest(name="gpu", deviceClassName="gpu.nvidia.com", count=8), + v1alpha1.DeviceRequest(name="nic", deviceClassName="nic.nvidia.com", count=8), + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _main_deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _main_standalone_claim_template( + name="r-main-standalone-devices-f456f", + requests=[ + { + "name": "gpu", + "exactly": {"deviceClassName": "gpu.nvidia.com", "count": 8}, + }, + { + "name": "nic", + "exactly": {"deviceClassName": "nic.nvidia.com", "count": 8}, + }, + ], + ), + }, + ), + # HF_HUB_CACHE makes the engine's own --model= resolve against the + # cache. Modelplane injects no --model of its own: naming the model is the + # command's job. + # + # On a Standard cluster there's no ModelExpress env and no security context. + # HF_HUB_CACHE isn't part of the ModelExpress bundle, and applies on every + # stack. + BuildCase( + name="a cache mounts its PVC and points HF_HUB_CACHE at it", + backend=native.NativeBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=v1alpha1.ModelCacheRef(name="qwen"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest", args=[]) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _main_deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": [], + "ports": [{"containerPort": 8000}], + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}], + } + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}, + }, + ], + ), + "resource-claim-main-standalone": _main_standalone_claim_template( + name="r-main-standalone-devices-f456f", + requests=[ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ], + ), + }, + ), + # Kubernetes expands $(VAR) left to right, so Modelplane's own entries must + # precede the user's for a user entry to reference them. + BuildCase( + name="a Standalone member's own env follows the cache env", + backend=native.NativeBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=v1alpha1.ModelCacheRef(name="qwen"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=[], + env=[v1alpha1.EnvItem(name="HF_TOKEN", value="x")], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _main_deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": [], + "ports": [{"containerPort": 8000}], + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + {"name": "HF_TOKEN", "value": "x"}, + ], + } + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}, + }, + ], + ), + "resource-claim-main-standalone": _main_standalone_claim_template( + name="r-main-standalone-devices-f456f", + requests=[ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ], + ), + }, + ), + # A Standalone engine on a Dynamo cluster with a cache is as valid a P2P + # peer set as a gang, so it gets the ModelExpress env and the IPC_LOCK + # security context. MX_SERVER_ADDRESS is the per-cluster shared server's + # well-known Service, qualified by its namespace because the engine runs in + # its team's. MX_MODEL_REVISION isolates this cache's P2P source identity, + # qualified by the Modelplane namespace like the cache's PVC name, so two + # namespaces' caches of the same name can't collide at the cluster's one + # shared server. + # + # HF_HUB_CACHE appears once. It's the cache's own env, and ModelExpress + # reads it only as a fallback for its cache root. + BuildCase( + name="a cached Standalone engine on Dynamo gets the ModelExpress env", + backend=native.NativeBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=v1alpha1.ModelCacheRef(name="qwen"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest", args=[]) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _main_deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": [], + "ports": [{"containerPort": 8000}], + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_SERVER_ADDRESS", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MODEL_EXPRESS_URL", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MX_MODEL_REVISION", + "value": "modelcache-ml-team-qwen-17db2", + }, + {"name": "MX_P2P_METADATA", "value": "1"}, + { + "name": "POD_NAME", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.name"}}, + }, + { + "name": "POD_UID", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.uid"}}, + }, + { + "name": "POD_NAMESPACE", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, + }, + ], + "securityContext": {"capabilities": {"add": ["IPC_LOCK"]}}, + } + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}, + }, + ], + ), + "resource-claim-main-standalone": _main_standalone_claim_template( + name="r-main-standalone-devices-f456f", + requests=[ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ], + ), + }, + ), + # The members' commands pass through verbatim, with no flag injection or + # bootstrap. The worker addresses the leader through + # $(MODELPLANE_LEADER_ADDRESS), which concatenates Grove's PCSG vars because + # they vary per gang. The PCS-scoped ones are identical across gangs, and + # would point every copy at gang 0's leader. There's no MODELPLANE_RANK: + # Grove exposes no group-wide pod index yet (grove#755), so a gang engine's + # command computes its own rank from GROVE_PCLQ_POD_INDEX. + # + # A PodCliqueSet publishes no Available condition, so its readiness derives + # from its replica counters. A worker with no template metadata carries only + # the queue label, and with no cache there's no ModelExpress env or security + # context. + BuildCase( + name="a Leader/Worker engine on Grove composes a PodCliqueSet, commands verbatim", + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _main_pod_clique_set( + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": {"kai.scheduler/queue": "modelplane"}, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ] + ), + "resource-claim-main-leader": _main_leader_claim_template(), + "resource-claim-main-worker": _main_worker_claim_template(), + }, + ), + BuildCase( + name="a Grove member's own env follows the leader address alias", + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + env=[v1alpha1.EnvItem(name="HF_TOKEN", value="x")], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _main_pod_clique_set( + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + {"name": "HF_TOKEN", "value": "x"}, + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": {"kai.scheduler/queue": "modelplane"}, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ] + ), + "resource-claim-main-leader": _main_leader_claim_template(), + "resource-claim-main-worker": _main_worker_claim_template(), + }, + ), + # Multi-NIC RDMA nodes need VLLM_HOST_IP from status.podIP so the engine + # binds the right interface (#141). + BuildCase( + name="a Grove member's pod-field env passes through", + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + env=[ + v1alpha1.EnvItem( + name="VLLM_HOST_IP", + valueFrom=v1alpha1.ValueFrom( + fieldRef=v1alpha1.FieldRef(fieldPath="status.podIP") + ), + ) + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _main_pod_clique_set( + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + { + "name": "VLLM_HOST_IP", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": {"kai.scheduler/queue": "modelplane"}, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ] + ), + "resource-claim-main-leader": _main_leader_claim_template(), + "resource-claim-main-worker": _main_worker_claim_template(), + }, + ), + # Neither member's metadata leaks into the other's clique. Grove propagates + # a clique's labels and annotations to its pods (#378). + BuildCase( + name="each Grove member's template metadata lands on its own clique", + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + metadata=v1alpha1.Metadata( + labels={"example.com/role": "leader"}, + annotations={"example.com/config": "leader"}, + ), + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ), + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + metadata=v1alpha1.Metadata( + labels={"example.com/role": "worker"}, + annotations={"example.com/config": "worker"}, + ), + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ), + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _main_pod_clique_set( + cliques=[ + { + "name": "leader", + "labels": { + "example.com/role": "leader", + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "annotations": {"example.com/config": "leader"}, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": { + "example.com/role": "worker", + "kai.scheduler/queue": "modelplane", + }, + "annotations": {"example.com/config": "worker"}, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ] + ), + "resource-claim-main-leader": _main_leader_claim_template(), + "resource-claim-main-worker": _main_worker_claim_template(), + }, + ), + # A coordinator-only leader, like a vLLM DP head running + # --data-parallel-size-local=0, has no deviceRequests. It gets no pod claim, + # no container claim and no ResourceClaimTemplate, but it still pins to its + # pool and tolerates the GPU taint. + BuildCase( + name="a claimless Grove leader composes no claim, but still pins and tolerates", + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _main_pod_clique_set( + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": {"kai.scheduler/queue": "modelplane"}, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ] + ), + "resource-claim-main-worker": _main_worker_claim_template(), + }, + ), + # The scheduler may split a gang across pools when no single pool satisfies + # every member, so each member's pods pin to that member's pool rather than + # an engine-wide one. + BuildCase( + name="each Grove member pins to its own pool", + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="head", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _main_pod_clique_set( + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "head"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": {"kai.scheduler/queue": "modelplane"}, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ] + ), + "resource-claim-main-leader": _main_leader_claim_template(), + "resource-claim-main-worker": _main_worker_claim_template(), + }, + ), + # Both cliques' HF_HUB_CACHE points at the mount, so their own + # --model= resolves against it. Modelplane adds no --model itself. + BuildCase( + name="a cache mounts on both Grove cliques", + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=v1alpha1.ModelCacheRef(name="kimi"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest", args=[]) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=["/bin/sh", "-c", "join"], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _main_pod_clique_set( + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_SERVER_ADDRESS", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MODEL_EXPRESS_URL", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MX_MODEL_REVISION", + "value": "modelcache-ml-team-kimi-aa322", + }, + {"name": "MX_P2P_METADATA", "value": "1"}, + { + "name": "POD_NAME", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.name"}}, + }, + { + "name": "POD_UID", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.uid"}}, + }, + { + "name": "POD_NAMESPACE", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, + }, + ], + "securityContext": {"capabilities": {"add": ["IPC_LOCK"]}}, + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-kimi-aa322"}, + }, + ], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": {"kai.scheduler/queue": "modelplane"}, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "command": ["/bin/sh", "-c", "join"], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_SERVER_ADDRESS", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MODEL_EXPRESS_URL", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MX_MODEL_REVISION", + "value": "modelcache-ml-team-kimi-aa322", + }, + {"name": "MX_P2P_METADATA", "value": "1"}, + { + "name": "POD_NAME", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.name"}}, + }, + { + "name": "POD_UID", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.uid"}}, + }, + { + "name": "POD_NAMESPACE", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, + }, + ], + "securityContext": {"capabilities": {"add": ["IPC_LOCK"]}}, + } + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-kimi-aa322"}, + }, + ], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ] + ), + "resource-claim-main-leader": _main_leader_claim_template(), + "resource-claim-main-worker": _main_worker_claim_template(), + }, + ), + # A member that points at the cache with its own flag gets no injected + # --model. + BuildCase( + name="a Grove member with its own command mounts a cache, command verbatim", + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=v1alpha1.ModelCacheRef(name="kimi"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "python3 -m sglang.launch_server --model-path /mnt/models --tp 16", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=["/bin/sh", "-c", "join"], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _main_pod_clique_set( + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "python3 -m sglang.launch_server --model-path /mnt/models --tp 16", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_SERVER_ADDRESS", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MODEL_EXPRESS_URL", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MX_MODEL_REVISION", + "value": "modelcache-ml-team-kimi-aa322", + }, + {"name": "MX_P2P_METADATA", "value": "1"}, + { + "name": "POD_NAME", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.name"}}, + }, + { + "name": "POD_UID", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.uid"}}, + }, + { + "name": "POD_NAMESPACE", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, + }, + ], + "securityContext": {"capabilities": {"add": ["IPC_LOCK"]}}, + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-kimi-aa322"}, + }, + ], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": {"kai.scheduler/queue": "modelplane"}, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "command": ["/bin/sh", "-c", "join"], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_SERVER_ADDRESS", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MODEL_EXPRESS_URL", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MX_MODEL_REVISION", + "value": "modelcache-ml-team-kimi-aa322", + }, + {"name": "MX_P2P_METADATA", "value": "1"}, + { + "name": "POD_NAME", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.name"}}, + }, + { + "name": "POD_UID", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.uid"}}, + }, + { + "name": "POD_NAMESPACE", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, + }, + ], + "securityContext": {"capabilities": {"add": ["IPC_LOCK"]}}, + } + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-kimi-aa322"}, + }, + ], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ] + ), + "resource-claim-main-leader": _main_leader_claim_template(), + "resource-claim-main-worker": _main_worker_claim_template(), + }, + ), + # Each pod publishes itself as a source independently, so the worker gets + # the ModelExpress env too. The leader address alias leads, ahead of the + # cache env and the ModelExpress bundle. + BuildCase( + name="a cached Grove gang on Dynamo gets the ModelExpress env on both cliques", + backend=grove.GroveBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=v1alpha1.ModelCacheRef(name="qwen"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Dynamo", + want={ + "model-serving-main": _main_pod_clique_set( + cliques=[ + { + "name": "leader", + "labels": { + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_SERVER_ADDRESS", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MODEL_EXPRESS_URL", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MX_MODEL_REVISION", + "value": "modelcache-ml-team-qwen-17db2", + }, + {"name": "MX_P2P_METADATA", "value": "1"}, + { + "name": "POD_NAME", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.name"}}, + }, + { + "name": "POD_UID", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.uid"}}, + }, + { + "name": "POD_NAMESPACE", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, + }, + ], + "securityContext": {"capabilities": {"add": ["IPC_LOCK"]}}, + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}, + }, + ], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-leader-devices-f58c6", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": {"kai.scheduler/queue": "modelplane"}, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_SERVER_ADDRESS", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MODEL_EXPRESS_URL", + "value": "modelexpress-server.default.svc:8001", + }, + { + "name": "MX_MODEL_REVISION", + "value": "modelcache-ml-team-qwen-17db2", + }, + {"name": "MX_P2P_METADATA", "value": "1"}, + { + "name": "POD_NAME", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.name"}}, + }, + { + "name": "POD_UID", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.uid"}}, + }, + { + "name": "POD_NAMESPACE", + "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, + }, + ], + "securityContext": {"capabilities": {"add": ["IPC_LOCK"]}}, + } + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}, + }, + ], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-main-worker-devices-99b8a", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ] + ), + "resource-claim-main-leader": _main_leader_claim_template(), + "resource-claim-main-worker": _main_worker_claim_template(), + }, + ), + # Only the leader carries the serving label. The worker followers never + # serve, so they carry no metadata at all. Every gang container leads with + # the backend-neutral coordination vars aliasing LWS_LEADER_ADDRESS and + # LWS_WORKER_INDEX. A LeaderWorkerSet reports readiness from its Available + # condition. + BuildCase( + name="a Leader/Worker engine on llm-d composes a LeaderWorkerSet", + backend=llmd.LLMDBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _main_leader_worker_set( + replicas=1, + size=2, + leader_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + leader_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + worker_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + } + ], + worker_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-leader": _main_leader_claim_template(), + "resource-claim-main-worker": _main_worker_claim_template(), + }, + ), + BuildCase( + name="a LeaderWorkerSet's gang is the leader plus its worker's nodes", + backend=llmd.LLMDBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=2, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=3), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _main_leader_worker_set( + replicas=2, + size=4, + leader_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + leader_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + worker_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + } + ], + worker_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-leader": _main_leader_claim_template(), + "resource-claim-main-worker": _main_worker_claim_template(), + }, + ), + # Only the native and Grove backends wire ModelExpress, and only on Dynamo, + # which never selects llm-d. HF_HUB_CACHE is the cache's own env, on every + # stack. + BuildCase( + name="llm-d injects no ModelExpress env, even with a cache", + backend=llmd.LLMDBackend(), + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=v1alpha1.ModelCacheRef(name="c"), + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + serving_label="r", + stack="Standard", + want={ + "model-serving-main": _main_leader_worker_set( + replicas=1, + size=2, + leader_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + leader_volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-c-c5cc6"}, + }, + ], + worker_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [ + {"name": "dshm", "mountPath": "/dev/shm"}, + {"name": "model-cache", "mountPath": "/mnt/models"}, + ], + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + ], + } + ], + worker_volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + { + "name": "model-cache", + "persistentVolumeClaim": {"claimName": "modelcache-ml-team-c-c5cc6"}, + }, + ], + ), + "resource-claim-main-leader": _main_leader_claim_template(), + "resource-claim-main-worker": _main_worker_claim_template(), + }, + ), +] -def _native_cache_replica() -> v1alpha1.ModelReplica: - """A replica with one Standalone engine, referencing the qwen cache.""" - engine = _standalone_engine(args=[]) - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen"), - engines=[engine], + +@pytest.mark.parametrize("case", BUILD_CASES, ids=lambda case: case.name) +def test_build(case: BuildCase) -> None: + """A backend composes an engine's workload, and the claim templates its members need.""" + # build composes one engine, so build each of the replica's engines in turn. + got: dict[str, k8sobjv1alpha1.Object] = {} + for engine in case.replica.spec.engines: + got.update(case.backend.build(case.replica, engine, case.provider_config, case.serving_label, case.stack)) + assert _sorted(_to_dicts(got)) == _sorted(case.want) + + +SELECT_BACKEND_CASES = [ + SelectBackendCase( + name="a Standalone engine on Standard is native", + engine=v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", image="vllm/vllm-openai:latest", args=["--model=Qwen/Qwen3-0.6B"] + ) + ] + ) + ), + ) + ], + ), + stack="Standard", + want="native", + ), + # A Standalone engine is native regardless of the cluster's stack. + SelectBackendCase( + name="a Standalone engine on Dynamo is native", + engine=v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", image="vllm/vllm-openai:latest", args=["--model=Qwen/Qwen3-0.6B"] + ) + ] + ) + ), + ) + ], + ), + stack="Dynamo", + want="native", + ), + SelectBackendCase( + name="a Leader/Worker engine on Standard is llm-d", + engine=v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + ], + ), + stack="Standard", + want="llmd", + ), + SelectBackendCase( + name="a Leader/Worker engine on Dynamo is Grove", + engine=v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[v1alpha1.Container(name="engine", image="vllm/vllm-openai:latest")] + ) + ), + ), + ], + ), + stack="Dynamo", + want="grove", + ), +] + + +@pytest.mark.parametrize("case", SELECT_BACKEND_CASES, ids=lambda case: case.name) +def test_select_backend(case: SelectBackendCase) -> None: + """An engine's member roles and its cluster's stack select its backend.""" + assert base.select_backend(case.engine, case.stack) == case.want + + +CACHE_MOUNTS_CASES = [ + CacheMountsCase( + name="no cache mounts nothing", + replica=_unnamed_replica(model_cache_ref=None), + want=([], []), + ), + CacheMountsCase( + name="a cache mounts its PVC", + replica=_unnamed_replica(model_cache_ref=v1alpha1.ModelCacheRef(name="qwen")), + want=( + [{"name": "model-cache", "persistentVolumeClaim": {"claimName": "modelcache-ml-team-qwen-17db2"}}], + [{"name": "model-cache", "mountPath": "/mnt/models"}], + ), + ), +] + + +@pytest.mark.parametrize("case", CACHE_MOUNTS_CASES, ids=lambda case: case.name) +def test_cache_mounts(case: CacheMountsCase) -> None: + """A replica's cache contributes a volume and a mount.""" + assert base.cache_mounts(case.replica) == case.want + + +CACHE_ENV_CASES = [ + CacheEnvCase( + name="no cache sets no env", + replica=_unnamed_replica(model_cache_ref=None), + want=[], + ), + # The cache is staged in HuggingFace's cache layout, so pointing + # HF_HUB_CACHE at the mount is what lets an engine's own --model= + # resolve against it instead of pulling from HuggingFace (#407). There's no + # HF_HUB_OFFLINE: it would break an engine that fetches a different repo at + # startup, like kimi-k2's separately gated tokenizer, and resolution doesn't + # need it. + CacheEnvCase( + name="a cache points HF_HUB_CACHE at the mount", + replica=_unnamed_replica(model_cache_ref=v1alpha1.ModelCacheRef(name="qwen")), + want=[{"name": "HF_HUB_CACHE", "value": "/mnt/models"}], + ), +] + + +@pytest.mark.parametrize("case", CACHE_ENV_CASES, ids=lambda case: case.name) +def test_cache_env(case: CacheEnvCase) -> None: + """A replica's cache contributes the env that resolves a model against it.""" + assert base.cache_env(case.replica) == case.want + + +APPLY_CASES = [ + # PrefillDecode role-labels the two engines, sidecars decode, and fronts + # both with an InferencePool and endpoint picker rather than a Service. Both + # engines get the NIXL plumbing the schema can't express: a Memory /dev/shm, + # and VLLM_NIXL_SIDE_CHANNEL_HOST set to the pod IP. The decode engine moves + # to port 8001, behind the pd-sidecar on 8000. + # + # PrefillDecode silently serves decode-only unless the picker's config arms + # the prefix-based PD decider. That needs nonCachedTokens > 0, the + # approx-prefix-cache-producer that populates the attribute it reads, pinned + # to autoTune: false, and no prepareDataPlugins feature gate, which the + # v0.8.0 EPP image rejects and crashloops on. The picker watches + # InferenceObjectives, so its Role must allow that. + # + # A wrong picker or sidecar image tag, or a stale EndpointPickerConfig API + # group, would otherwise only surface as a crashloop at deploy. Pinning them + # here means a bump shows up to be reviewed. + ApplyCase( + name="PrefillDecode fronts prefill and decode engines with a pool and endpoint picker", + composed={ + "model-serving-prefill": _prefill_deployment( + labels={ + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-prefill-d90b0", + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-prefill-standalone": _prefill_claim_template(), + "model-serving-decode": _decode_deployment( + labels={ + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-decode-4b27b", + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-decode-standalone": _decode_claim_template(), + }, + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=v1alpha1.Serving(mode="PrefillDecode"), + engines=[ + v1alpha1.Engine( + name="prefill", + phase="Prefill", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + v1alpha1.Engine( + name="decode", + phase="Decode", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + ], + ), + provider_config="cluster-a-pc", + want={ + "model-serving-prefill": _prefill_deployment( + labels={ + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-prefill-d90b0", + "llm-d.ai/role": "prefill", + "llm-d.ai/inference-serving": "true", + "app": "r", + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + {"name": "VLLM_NIXL_SIDE_CHANNEL_PORT", "value": "5557"}, + ], + } + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + ), + "resource-claim-prefill-standalone": _prefill_claim_template(), + "model-serving-decode": _decode_deployment( + labels={ + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-decode-4b27b", + "llm-d.ai/role": "decode", + "llm-d.ai/inference-serving": "true", + "app": "r", + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8001}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8001}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + {"name": "VLLM_NIXL_SIDE_CHANNEL_PORT", "value": "5557"}, + ], + }, + { + "name": "pd-sidecar", + "image": "ghcr.io/llm-d/llm-d-router-disagg-sidecar:v0.9.0", + "args": [ + "--secure-proxy=false", + "--kv-connector=nixlv2", + "--vllm-port=8001", + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + }, + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + ), + "resource-claim-decode-standalone": _decode_claim_template(), + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: disagg-headers-handler\n" + "- type: queue-scorer\n" + "- type: prefill-filter\n" + "- type: decode-filter\n" + "- type: max-score-picker\n" + "- type: prefix-based-pd-decider\n" + " parameters:\n" + " nonCachedTokens: 16\n" + "- type: disagg-profile-handler\n" + " parameters:\n" + " deciders:\n" + " prefill: prefix-based-pd-decider\n" + "schedulingProfiles:\n" + "- name: prefill\n" + " plugins:\n" + " - pluginRef: prefill-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + "- name: decode\n" + " plugins:\n" + " - pluginRef: decode-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp(config_checksum="f715f3024f37e7042d42628f10cbb04c7da78b96dc65953dac0f289ef2c7ef98"), + "epp-service": _epp_service(), + }, + ), + # The sidecar and the decode container's port track the user's --port rather + # than a hardcoded one. + ApplyCase( + name="the decode port follows the user's --port", + composed={ + "model-serving-prefill": _prefill_deployment( + labels={ + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-prefill-d90b0", + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-prefill-standalone": _prefill_claim_template(), + "model-serving-decode": _decode_deployment( + labels={ + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-decode-4b27b", + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=m", "--port=9000"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-decode-standalone": _decode_claim_template(), + }, + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=v1alpha1.Serving(mode="PrefillDecode"), + engines=[ + v1alpha1.Engine( + name="prefill", + phase="Prefill", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + v1alpha1.Engine( + name="decode", + phase="Decode", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=m", "--port=9000"], + ) + ] + ) + ), + ) + ], + ), + ], + ), + provider_config="cluster-a-pc", + want={ + "model-serving-prefill": _prefill_deployment( + labels={ + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-prefill-d90b0", + "llm-d.ai/role": "prefill", + "llm-d.ai/inference-serving": "true", + "app": "r", + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + {"name": "VLLM_NIXL_SIDE_CHANNEL_PORT", "value": "5557"}, + ], + } + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + ), + "resource-claim-prefill-standalone": _prefill_claim_template(), + "model-serving-decode": _decode_deployment( + labels={ + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-decode-4b27b", + "llm-d.ai/role": "decode", + "llm-d.ai/inference-serving": "true", + "app": "r", + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=m", "--port=9000"], + "ports": [{"containerPort": 9000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 9000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + {"name": "VLLM_NIXL_SIDE_CHANNEL_PORT", "value": "5557"}, + ], + }, + { + "name": "pd-sidecar", + "image": "ghcr.io/llm-d/llm-d-router-disagg-sidecar:v0.9.0", + "args": [ + "--secure-proxy=false", + "--kv-connector=nixlv2", + "--vllm-port=9000", + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + }, + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + ), + "resource-claim-decode-standalone": _decode_claim_template(), + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: disagg-headers-handler\n" + "- type: queue-scorer\n" + "- type: prefill-filter\n" + "- type: decode-filter\n" + "- type: max-score-picker\n" + "- type: prefix-based-pd-decider\n" + " parameters:\n" + " nonCachedTokens: 16\n" + "- type: disagg-profile-handler\n" + " parameters:\n" + " deciders:\n" + " prefill: prefix-based-pd-decider\n" + "schedulingProfiles:\n" + "- name: prefill\n" + " plugins:\n" + " - pluginRef: prefill-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + "- name: decode\n" + " plugins:\n" + " - pluginRef: decode-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp(config_checksum="f715f3024f37e7042d42628f10cbb04c7da78b96dc65953dac0f289ef2c7ef98"), + "epp-service": _epp_service(), + }, + ), + # alpha is Decode, so it gets the sidecar, and beta is Prefill, whatever + # their names suggest. + ApplyCase( + name="PrefillDecode takes each engine's role from its phase, not its name", + composed={ + "model-serving-alpha": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": "r-alpha-b36ce", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": 1, + "selector": {"matchLabels": {"modelplane.ai/workload": "r-alpha-b36ce"}}, + "template": { + "metadata": { + "labels": { + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-alpha-b36ce", + } + }, + "spec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-alpha-standalone-devices-9ec1c", + } + ], + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ], + }, + }, + }, + } + }, + } + }, + "resource-claim-alpha-standalone": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": "r-alpha-standalone-devices-9ec1c", "namespace": "mp-ml-team-51733"}, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ] + } + } + }, + } + }, + } + }, + "model-serving-beta": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": "r-beta-52d85", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": 1, + "selector": {"matchLabels": {"modelplane.ai/workload": "r-beta-52d85"}}, + "template": { + "metadata": { + "labels": { + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-beta-52d85", + } + }, + "spec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-beta-standalone-devices-9d8b1", + } + ], + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ], + }, + }, + }, + } + }, + } + }, + "resource-claim-beta-standalone": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": "r-beta-standalone-devices-9d8b1", "namespace": "mp-ml-team-51733"}, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ] + } + } + }, + } + }, + } + }, + }, + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=v1alpha1.Serving(mode="PrefillDecode"), + engines=[ + v1alpha1.Engine( + name="alpha", + phase="Decode", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + v1alpha1.Engine( + name="beta", + phase="Prefill", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + ], ), - ) + provider_config="cluster-a-pc", + want={ + "model-serving-alpha": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": "r-alpha-b36ce", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": 1, + "selector": {"matchLabels": {"modelplane.ai/workload": "r-alpha-b36ce"}}, + "template": { + "metadata": { + "labels": { + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-alpha-b36ce", + "llm-d.ai/role": "decode", + "llm-d.ai/inference-serving": "true", + "app": "r", + } + }, + "spec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8001}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8001}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + {"name": "VLLM_NIXL_SIDE_CHANNEL_PORT", "value": "5557"}, + ], + }, + { + "name": "pd-sidecar", + "image": "ghcr.io/llm-d/llm-d-router-disagg-sidecar:v0.9.0", + "args": [ + "--secure-proxy=false", + "--kv-connector=nixlv2", + "--vllm-port=8001", + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + }, + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-alpha-standalone-devices-9ec1c", + } + ], + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ], + }, + }, + }, + } + }, + } + }, + "resource-claim-alpha-standalone": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": "r-alpha-standalone-devices-9ec1c", "namespace": "mp-ml-team-51733"}, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ] + } + } + }, + } + }, + } + }, + "model-serving-beta": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": "r-beta-52d85", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": 1, + "selector": {"matchLabels": {"modelplane.ai/workload": "r-beta-52d85"}}, + "template": { + "metadata": { + "labels": { + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-beta-52d85", + "llm-d.ai/role": "prefill", + "llm-d.ai/inference-serving": "true", + "app": "r", + } + }, + "spec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + {"name": "VLLM_NIXL_SIDE_CHANNEL_PORT", "value": "5557"}, + ], + } + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-beta-standalone-devices-9d8b1", + } + ], + "tolerations": [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ], + }, + }, + }, + } + }, + } + }, + "resource-claim-beta-standalone": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": "r-beta-standalone-devices-9d8b1", "namespace": "mp-ml-team-51733"}, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ] + } + } + }, + } + }, + } + }, + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: disagg-headers-handler\n" + "- type: queue-scorer\n" + "- type: prefill-filter\n" + "- type: decode-filter\n" + "- type: max-score-picker\n" + "- type: prefix-based-pd-decider\n" + " parameters:\n" + " nonCachedTokens: 16\n" + "- type: disagg-profile-handler\n" + " parameters:\n" + " deciders:\n" + " prefill: prefix-based-pd-decider\n" + "schedulingProfiles:\n" + "- name: prefill\n" + " plugins:\n" + " - pluginRef: prefill-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + "- name: decode\n" + " plugins:\n" + " - pluginRef: decode-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp(config_checksum="f715f3024f37e7042d42628f10cbb04c7da78b96dc65953dac0f289ef2c7ef98"), + "epp-service": _epp_service(), + }, + ), + # Routing decorates a Grove PodCliqueSet's leader clique with the role and + # serving labels, the pd-sidecar and the NIXL plumbing, exactly as it does a + # Deployment's pod template. The worker clique never serves, so routing + # leaves it as the Grove backend composed it. + ApplyCase( + name="a PrefillDecode decode engine can be a Grove gang", + composed={ + "model-serving-prefill": _prefill_deployment( + labels={ + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-prefill-d90b0", + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-prefill-standalone": _prefill_claim_template(), + "model-serving-decode": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status) && has(object.status.observedGeneration) && object.status.observedGeneration == object.metadata.generation && object.spec.replicas > 0 && has(object.status.availableReplicas) && object.status.availableReplicas >= object.spec.replicas", + }, + "forProvider": { + "manifest": { + "apiVersion": "grove.io/v1alpha1", + "kind": "PodCliqueSet", + "metadata": {"name": "r-decode-4b27b", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": 1, + "template": { + "cliqueStartupType": "CliqueStartupTypeExplicit", + "terminationDelay": "4h", + "headlessServiceConfig": {"publishNotReadyAddresses": True}, + "cliques": [ + { + "name": "leader", + "labels": { + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-decode-leader-devices-d2e70", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": {"kai.scheduler/queue": "modelplane"}, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-decode-worker-devices-08d44", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ], + "podCliqueScalingGroups": [ + { + "name": "gang", + "cliqueNames": ["leader", "worker"], + "replicas": 1, + "minAvailable": 1, + } + ], + }, + }, + } + }, + } + }, + "resource-claim-decode-leader": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": "r-decode-leader-devices-d2e70", "namespace": "mp-ml-team-51733"}, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ] + } + } + }, + } + }, + } + }, + "resource-claim-decode-worker": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": "r-decode-worker-devices-08d44", "namespace": "mp-ml-team-51733"}, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ] + } + } + }, + } + }, + } + }, + }, + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=v1alpha1.Serving(mode="PrefillDecode"), + engines=[ + v1alpha1.Engine( + name="prefill", + phase="Prefill", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ), + v1alpha1.Engine( + name="decode", + phase="Decode", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ), + ], + ), + provider_config="cluster-a-pc", + want={ + "model-serving-prefill": _prefill_deployment( + labels={ + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-prefill-d90b0", + "llm-d.ai/role": "prefill", + "llm-d.ai/inference-serving": "true", + "app": "r", + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + "env": [ + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": {"fieldRef": {"fieldPath": "status.podIP"}}, + }, + {"name": "VLLM_NIXL_SIDE_CHANNEL_PORT", "value": "5557"}, + ], + } + ], + volumes=[ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + ), + "resource-claim-prefill-standalone": _prefill_claim_template(), + "model-serving-decode": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status) && has(object.status.observedGeneration) && object.status.observedGeneration == object.metadata.generation && object.spec.replicas > 0 && has(object.status.availableReplicas) && object.status.availableReplicas >= object.spec.replicas", + }, + "forProvider": { + "manifest": { + "apiVersion": "grove.io/v1alpha1", + "kind": "PodCliqueSet", + "metadata": {"name": "r-decode-4b27b", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": 1, + "template": { + "cliqueStartupType": "CliqueStartupTypeExplicit", + "terminationDelay": "4h", + "headlessServiceConfig": {"publishNotReadyAddresses": True}, + "cliques": [ + { + "name": "leader", + "labels": { + "modelplane.ai/serving": "r", + "kai.scheduler/queue": "modelplane", + "modelplane.ai/clique-role": "leader", + "llm-d.ai/role": "decode", + "llm-d.ai/inference-serving": "true", + "app": "r", + }, + "spec": { + "roleName": "leader", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + }, + { + "name": "VLLM_NIXL_SIDE_CHANNEL_HOST", + "valueFrom": { + "fieldRef": {"fieldPath": "status.podIP"} + }, + }, + { + "name": "VLLM_NIXL_SIDE_CHANNEL_PORT", + "value": "5557", + }, + ], + "ports": [{"containerPort": 8001}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8001}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + }, + { + "name": "pd-sidecar", + "image": "ghcr.io/llm-d/llm-d-router-disagg-sidecar:v0.9.0", + "args": [ + "--secure-proxy=false", + "--kv-connector=nixlv2", + "--vllm-port=8001", + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + }, + ], + "volumes": [ + {"name": "dshm", "emptyDir": {"medium": "Memory"}}, + {"name": "nixl-shm", "emptyDir": {"medium": "Memory"}}, + ], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-decode-leader-devices-d2e70", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + { + "name": "worker", + "labels": {"kai.scheduler/queue": "modelplane"}, + "spec": { + "roleName": "worker", + "replicas": 1, + "minAvailable": 1, + "podSpec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(GROVE_PCSG_NAME)-$(GROVE_PCSG_INDEX)-leader-0.$(GROVE_HEADLESS_SERVICE)", + } + ], + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "schedulerName": "kai-scheduler", + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "r-decode-worker-devices-08d44", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], + }, + }, + }, + ], + "podCliqueScalingGroups": [ + { + "name": "gang", + "cliqueNames": ["leader", "worker"], + "replicas": 1, + "minAvailable": 1, + } + ], + }, + }, + } + }, + } + }, + "resource-claim-decode-leader": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": "r-decode-leader-devices-d2e70", "namespace": "mp-ml-team-51733"}, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ] + } + } + }, + } + }, + } + }, + "resource-claim-decode-worker": { + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": {"name": "r-decode-worker-devices-08d44", "namespace": "mp-ml-team-51733"}, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 8, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ] + } + } + }, + } + }, + } + }, + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: disagg-headers-handler\n" + "- type: queue-scorer\n" + "- type: prefill-filter\n" + "- type: decode-filter\n" + "- type: max-score-picker\n" + "- type: prefix-based-pd-decider\n" + " parameters:\n" + " nonCachedTokens: 16\n" + "- type: disagg-profile-handler\n" + " parameters:\n" + " deciders:\n" + " prefill: prefix-based-pd-decider\n" + "schedulingProfiles:\n" + "- name: prefill\n" + " plugins:\n" + " - pluginRef: prefill-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + "- name: decode\n" + " plugins:\n" + " - pluginRef: decode-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp(config_checksum="f715f3024f37e7042d42628f10cbb04c7da78b96dc65953dac0f289ef2c7ef98"), + "epp-service": _epp_service(), + }, + ), + # A replica with no serving block is served Unified, here and in the cases + # that follow. + # + # A single serving pod has nothing to pick between, but still gets the pool. + # Always fronting with one avoids swapping a Service for a pool when a + # second pod appears, a swap that would drop in-flight requests. The pool + # selects the pods by the serving label they already carry. + # + # The picker scores by prefix cache and queue depth in a single profile, + # with no prefill/decode split, fed by the approx-prefix-cache-producer. It + # reads its config once at startup, so its pod template carries a sha256 of + # the config, and a config change rolls the pod. Its image and the + # EndpointPickerConfig API group are pinned here too, so a wrong tag or a + # stale group fails here rather than crashlooping at deploy. + ApplyCase( + name="Unified fronts one pod with a pool and endpoint picker", + composed={ + "model-serving-main": _main_deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _main_standalone_claim_template( + name="r-main-standalone-devices-f456f", + requests=[ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ], + ), + }, + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + want={ + "model-serving-main": _main_deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=1, + pod_metadata={ + "labels": { + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _main_standalone_claim_template( + name="r-main-standalone-devices-f456f", + requests=[ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ], + ), + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: queue-scorer\n" + "- type: max-score-picker\n" + "schedulingProfiles:\n" + "- name: default\n" + " plugins:\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp(config_checksum="20c1dfea3fc4ad41e335cc74edbeb1e8689a607bc4bf7395849ffd2cf0bb2ae1"), + "epp-service": _epp_service(), + }, + ), + ApplyCase( + name="Unified fronts several pods with a pool and endpoint picker", + composed={ + "model-serving-main": _main_deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=2, + pod_metadata={ + "labels": { + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _main_standalone_claim_template( + name="r-main-standalone-devices-f456f", + requests=[ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ], + ), + }, + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=2, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + provider_config="cluster-a-pc", + want={ + "model-serving-main": _main_deployment( + name="r-main-bb4e3", + claim_template_name="r-main-standalone-devices-f456f", + replicas=2, + pod_metadata={ + "labels": { + "modelplane.ai/serving": "r", + "modelplane.ai/workload": "r-main-bb4e3", + } + }, + containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-standalone": _main_standalone_claim_template( + name="r-main-standalone-devices-f456f", + requests=[ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ + { + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } + ], + }, + } + ], + ), + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: queue-scorer\n" + "- type: max-score-picker\n" + "schedulingProfiles:\n" + "- name: default\n" + " plugins:\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp(config_checksum="20c1dfea3fc4ad41e335cc74edbeb1e8689a607bc4bf7395849ffd2cf0bb2ae1"), + "epp-service": _epp_service(), + }, + ), + # Unified routing reads the engine args for the KV block size through + # _serving_pod_templates, which normalizes a LeaderWorkerSet's + # leaderTemplate alongside a Deployment's pod template and a Grove + # PodCliqueSet's leader clique. A regression case for a normalization that + # only knew Deployment and PodCliqueSet, and raised KeyError on a + # LeaderWorkerSet. + ApplyCase( + name="Unified fronts a LeaderWorkerSet", + composed={ + "model-serving-main": _main_leader_worker_set( + replicas=1, + size=2, + leader_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + leader_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + worker_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + } + ], + worker_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-leader": _main_leader_claim_template(), + "resource-claim-main-worker": _main_worker_claim_template(), + }, + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Leader", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + ) + ] + ) + ), + ), + v1alpha1.Member( + role="Worker", + nodePoolName="frontier", + worker=v1alpha1.Worker(nodes=1), + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=8, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + command=[ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + ) + ] + ) + ), + ), + ], + ) + ], + ), + provider_config="cluster-a-pc", + want={ + "model-serving-main": _main_leader_worker_set( + replicas=1, + size=2, + leader_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "ray start --head --port=6379; exec vllm serve --model=meta-llama/Llama-3.1-405B --tensor-parallel-size=8 --pipeline-parallel-size=2 --port=8000", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + "ports": [{"containerPort": 8000}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, + }, + } + ], + leader_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + worker_containers=[ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "resources": {"claims": [{"name": "devices"}]}, + "command": [ + "/bin/sh", + "-c", + "exec ray start --address=$(MODELPLANE_LEADER_ADDRESS):6379 --block", + ], + "env": [ + { + "name": "MODELPLANE_LEADER_ADDRESS", + "value": "$(LWS_LEADER_ADDRESS)", + }, + {"name": "MODELPLANE_RANK", "value": "$(LWS_WORKER_INDEX)"}, + ], + } + ], + worker_volumes=[{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + ), + "resource-claim-main-leader": _main_leader_claim_template(), + "resource-claim-main-worker": _main_worker_claim_template(), + "inference-pool": _inference_pool(), + "model-route": _route(), + "epp-serviceaccount": _epp_service_account(), + "epp-role": _epp_role(), + "epp-rolebinding": _epp_role_binding(), + "epp-config": _epp_config( + config="apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: queue-scorer\n" + "- type: max-score-picker\n" + "schedulingProfiles:\n" + "- name: default\n" + " plugins:\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ), + "epp": _epp(config_checksum="20c1dfea3fc4ad41e335cc74edbeb1e8689a607bc4bf7395849ffd2cf0bb2ae1"), + "epp-service": _epp_service(), + }, + ), +] -def test_native_cache_mounts_pvc_and_sets_cache_env() -> None: - """The native backend mounts a cache and points HF_HUB_CACHE at it, injecting no --model.""" - # A cache contributes a volume, a mount, and the HF_HUB_CACHE that makes - # the engine's own --model= resolve against it. Modelplane injects - # no --model of its own: naming the model is the command's job. - replica = _native_cache_replica() - out = native.NativeBackend().build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Standard") - dep = out["model-serving-main"].spec.forProvider.manifest - pod = dep["spec"]["template"]["spec"] - vol_names = {v["name"] for v in pod["volumes"]} - assert "model-cache" in vol_names - container = pod["containers"][0] - assert {"name": "model-cache", "mountPath": "/mnt/models"} in container["volumeMounts"] - assert {"name": "HF_HUB_CACHE", "value": "/mnt/models"} in container["env"] - assert container["args"] == [] - - -def test_native_cache_user_env_comes_after_cache_env() -> None: - """A Standalone member's own env follows the cache env.""" - # Kubernetes expands $(VAR) left to right, so Modelplane's own entries - # must precede the user's for a user entry to reference them. - replica = _native_cache_replica() - engine = replica.spec.engines[0] - spec = engine.members[0].template.spec - assert spec is not None - spec.containers[0].env = [v1alpha1.EnvItem(name="HF_TOKEN", value="x")] - out = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] - assert container["env"] == [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}, {"name": "HF_TOKEN", "value": "x"}] - - -def _grove_cache_replica( - *, - leader_command: list[str] | None = None, - worker_command: list[str] | None = None, - leader_args: list[str] | None = None, - worker_args: list[str] | None = None, -) -> v1alpha1.ModelReplica: - """A replica with one Leader/Worker engine, referencing the kimi cache.""" - engine = _gang_engine( - leader_command=leader_command, - worker_command=worker_command, - leader_args=leader_args, - worker_args=worker_args, - ) - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="kimi"), - engines=[engine], - ), - ) +@pytest.mark.parametrize("case", APPLY_CASES, ids=lambda case: case.name) +def test_apply(case: ApplyCase) -> None: + """routing.apply fronts a replica's engines with the routing its serving mode selects.""" + composed = {key: k8sobjv1alpha1.Object.model_validate(obj) for key, obj in case.composed.items()} + got = routing.apply(composed, case.replica, case.provider_config) + assert _sorted(_to_dicts(got)) == _sorted(case.want) -def test_grove_cache_both_cliques_mount_cache() -> None: - """The Grove backend mounts a cache on both cliques.""" - replica = _grove_cache_replica(leader_args=[], worker_command=["/bin/sh", "-c", "join"]) - manifest = ( - grove.GroveBackend() - .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] - .spec.forProvider.manifest - ) - for clique_name in ("leader", "worker"): - pod = _clique(manifest, clique_name)["spec"]["podSpec"] - assert "model-cache" in {v["name"] for v in pod["volumes"]} - assert {"name": "model-cache", "mountPath": "/mnt/models"} in pod["containers"][0]["volumeMounts"] - - -def test_grove_cache_sets_cache_env_on_every_clique_and_injects_no_model() -> None: - """The Grove backend points both cliques' HF_HUB_CACHE at a cache, injecting no --model.""" - # A cache gives both cliques HF_HUB_CACHE so their own --model= - # resolves against the mount; Modelplane adds no --model itself. - replica = _grove_cache_replica(leader_args=[], worker_command=["/bin/sh", "-c", "join"]) - manifest = ( - grove.GroveBackend() - .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] - .spec.forProvider.manifest - ) - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - assert {"name": "HF_HUB_CACHE", "value": "/mnt/models"} in container["env"] - assert "--model=/mnt/models" not in container.get("args", []) - - -def test_grove_cache_command_engine_mounts_cache_without_injecting_model() -> None: - """A Grove member with its own command mounts a cache and keeps its command verbatim.""" - # A member with its own command keeps it verbatim and gets no injected - # --model (it points at the cache with its own flag). - leader_cmd = [ - "/bin/sh", - "-c", - "python3 -m sglang.launch_server --model-path /mnt/models --tp 16", - ] - replica = _grove_cache_replica(leader_command=leader_cmd, worker_command=["/bin/sh", "-c", "join"]) - manifest = ( - grove.GroveBackend() - .build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo")["model-serving-main"] - .spec.forProvider.manifest - ) - leader = _clique(manifest, "leader")["spec"]["podSpec"]["containers"][0] - assert {"name": "model-cache", "mountPath": "/mnt/models"} in leader["volumeMounts"] - assert leader["command"] == leader_cmd - - -# serving.mode: PrefillDecode routing layers an InferencePool + endpoint -# picker over two engines, role-labels them, and sidecars decode — no unified -# Service. Mirrors how fn.py composes engines then calls routing.apply. - - -def _disaggregated_apply() -> dict[str, k8sobjv1alpha1.Object]: - """Routing for a PrefillDecode replica of two native engines.""" - prefill = _standalone_engine(name="prefill") - prefill.phase = "Prefill" - decode = _standalone_engine(name="decode") - decode.phase = "Decode" - replica = _replica(engines=[prefill, decode]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = {} - for engine in replica.spec.engines: - composed.update(native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard")) - return routing.apply(composed, replica, _PC) - - -def _serving_pod(out: dict[str, k8sobjv1alpha1.Object], engine_name: str) -> dict: - """The pod template of an engine's Deployment.""" - return out[f"model-serving-{engine_name}"].spec.forProvider.manifest["spec"]["template"] - - -def test_disaggregated_replaces_unified_service_with_pool_and_epp() -> None: - """PrefillDecode routing fronts the engines with an InferencePool and endpoint picker.""" - out = _disaggregated_apply() - assert "inference-pool" in out - assert "epp" in out - assert "epp-config" in out - pool = out["inference-pool"].spec.forProvider.manifest - assert pool["kind"] == "InferencePool" - assert pool["spec"]["endpointPickerRef"]["name"] == "r-epp" - - -def test_disaggregated_injects_nixl_plumbing() -> None: - """Both PrefillDecode engines get the NIXL plumbing the schema can't express.""" - # The plumbing is a Memory /dev/shm and VLLM_NIXL_SIDE_CHANNEL_HOST = pod IP. - out = _disaggregated_apply() - for role in ("prefill", "decode"): - pod = _serving_pod(out, role)["spec"] - assert any(v.get("emptyDir", {}).get("medium") == "Memory" for v in pod["volumes"]), ( - f"{role} missing Memory /dev/shm volume" - ) - engine = next(c for c in pod["containers"] if c["name"] == "engine") - assert "/dev/shm" in [m["mountPath"] for m in engine["volumeMounts"]] - host = next((e for e in engine["env"] if e["name"] == "VLLM_NIXL_SIDE_CHANNEL_HOST"), None) - assert host is not None, f"{role} missing VLLM_NIXL_SIDE_CHANNEL_HOST" - assert host["valueFrom"]["fieldRef"]["fieldPath"] == "status.podIP" - assert "VLLM_NIXL_SIDE_CHANNEL_PORT" in [e["name"] for e in engine["env"]] - - -def test_disaggregated_epp_config_arms_the_pd_decider() -> None: - """PrefillDecode silently serves decode-only unless the PD decider is armed.""" - # Selective prefix-based-pd-decider needs all of: nonCachedTokens > 0 (0 = - # disabled), the approx-prefix-cache-producer plugin that populates the - # attribute it reads, and that producer pinned to autoTune: false (the - # true default never populates). And it must NOT carry the prepareDataPlugins - # feature gate, which the v0.8.0 EPP image rejects and crashloops on. - cfg = _disaggregated_apply()["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - assert "prefix-based-pd-decider" in cfg - assert "nonCachedTokens: 16" in cfg - assert "approx-prefix-cache-producer" in cfg - assert "autoTune: false" in cfg - assert "nonCachedTokens: 0" not in cfg - assert "prepareDataPlugins" not in cfg - - -def test_disaggregated_epp_and_sidecar_images_and_config_group_are_pinned() -> None: - """Lock the picker and sidecar images and the EndpointPickerConfig API group.""" - # Nothing else asserts these, so a wrong tag/registry path or a stale config - # group passes CI and only surfaces as an EPP/sidecar crashloop at deploy. - # These are deliberate literals, not routing._* constants: comparing to the - # constant would be tautological (it can't catch a typo in the constant), and - # a literal forces a bump to show up here and be reviewed. - out = _disaggregated_apply() - epp = out["epp"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"] - assert next(c["image"] for c in epp if c["name"] == "epp") == "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0" - sidecar = next(c for c in _serving_pod(out, "decode")["spec"]["containers"] if c["name"] == "pd-sidecar") - assert sidecar["image"] == "ghcr.io/llm-d/llm-d-router-disagg-sidecar:v0.9.0" - cfg = out["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - assert "apiVersion: llm-d.ai/v1alpha1" in cfg - - -def test_disaggregated_epp_role_watches_inferenceobjectives() -> None: - """The picker watches InferenceObjectives (GIE x-k8s.io group); the Role must allow it.""" - rules = _disaggregated_apply()["epp-role"].spec.forProvider.manifest["rules"] - assert any( - "inference.networking.x-k8s.io" in r["apiGroups"] and "inferenceobjectives" in r["resources"] for r in rules - ), f"EPP Role missing inferenceobjectives watch: {rules}" - - -def test_disaggregated_decode_port_follows_user_arg() -> None: - """The sidecar and the decode container port track the user's --port, not a hardcoded one.""" - prefill = _standalone_engine(name="prefill") - prefill.phase = "Prefill" - decode = _standalone_engine(name="decode", args=["--model=m", "--port=9000"]) - decode.phase = "Decode" - replica = _replica(engines=[prefill, decode]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = {} - for e in replica.spec.engines: - composed.update(native.NativeBackend().build(replica, e, _PC, base.serving_label(replica), "Standard")) - out = routing.apply(composed, replica, _PC) - containers = _serving_pod(out, "decode")["spec"]["containers"] - engine = next(c for c in containers if c["name"] == "engine") - sidecar = next(c for c in containers if c["name"] == "pd-sidecar") - assert engine["ports"][0]["containerPort"] == 9000 - assert "--vllm-port=9000" in sidecar["args"] - assert sidecar["ports"][0]["containerPort"] == 8000 - - -def test_disaggregated_engines_role_labeled() -> None: - """PrefillDecode routing labels each engine's pods with its role.""" - out = _disaggregated_apply() - assert _serving_pod(out, "prefill")["metadata"]["labels"]["llm-d.ai/role"] == "prefill" - decode_labels = _serving_pod(out, "decode")["metadata"]["labels"] - assert decode_labels["llm-d.ai/role"] == "decode" - assert decode_labels["app"] == "r" - - -def test_disaggregated_decode_gets_sidecar_and_moves_engine_port() -> None: - """The decode engine gets the pd-sidecar on the serving port, and moves to another.""" - out = _disaggregated_apply() - containers = _serving_pod(out, "decode")["spec"]["containers"] - names = [c["name"] for c in containers] - assert names == ["engine", "pd-sidecar"] - engine = next(c for c in containers if c["name"] == "engine") - assert engine["ports"][0]["containerPort"] == 8001 - assert engine["readinessProbe"]["timeoutSeconds"] == 5 - sidecar = next(c for c in containers if c["name"] == "pd-sidecar") - assert sidecar["ports"][0]["containerPort"] == 8000 - assert sidecar["readinessProbe"]["timeoutSeconds"] == 5 - assert "--secure-proxy=false" in sidecar["args"] - - -def test_disaggregated_prefill_has_no_sidecar() -> None: - """The prefill engine gets no sidecar.""" - containers = _serving_pod(_disaggregated_apply(), "prefill")["spec"]["containers"] - assert [c["name"] for c in containers] == ["engine"] - - -def test_disaggregated_route_targets_inference_pool() -> None: - """PrefillDecode routing points the HTTPRoute at the InferencePool, with no request timeout.""" - route = _disaggregated_apply()[base.ROUTE_KEY].spec.forProvider.manifest - rule = route["spec"]["rules"][0] - ref = rule["backendRefs"][0] - assert ref["kind"] == "InferencePool" - assert ref["name"] == "r-pool" - # Disable the request timeout so long token streams aren't severed. - assert rule["timeouts"]["request"] == "0s" - - -def test_disaggregated_selects_engines_by_phase_not_name() -> None: - """Roles come from each engine's phase, not its name.""" - decode = _standalone_engine(name="alpha") - decode.phase = "Decode" - prefill = _standalone_engine(name="beta") - prefill.phase = "Prefill" - replica = _replica(engines=[decode, prefill]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = {} - for e in replica.spec.engines: - composed.update(native.NativeBackend().build(replica, e, _PC, base.serving_label(replica), "Standard")) - out = routing.apply(composed, replica, _PC) - # alpha is Decode -> sidecar; beta is Prefill -> none, despite their names. - assert [c["name"] for c in _serving_pod(out, "alpha")["spec"]["containers"]] == ["engine", "pd-sidecar"] - assert [c["name"] for c in _serving_pod(out, "beta")["spec"]["containers"]] == ["engine"] - assert _serving_pod(out, "alpha")["metadata"]["labels"]["llm-d.ai/role"] == "decode" - assert _serving_pod(out, "beta")["metadata"]["labels"]["llm-d.ai/role"] == "prefill" - - -def test_disaggregated_decode_can_be_a_grove_gang() -> None: - """PrefillDecode routing decorates a Grove decode gang's leader clique, and leaves its worker alone.""" - # A PrefillDecode engine can itself be a Leader/Worker gang, so routing - # must decorate a Grove PodCliqueSet's leader clique - role label, serving - # label, pd-sidecar, NIXL plumbing - exactly like a Deployment's pod - # template. Exercises the _serving_pod_templates normalization that lets - # one routing layer decorate both workload shapes. - prefill = _standalone_engine(name="prefill") - prefill.phase = "Prefill" - decode = _gang_engine(name="decode", leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - decode.phase = "Decode" - replica = _replica(engines=[prefill, decode]) - replica.spec.serving = v1alpha1.Serving(mode="PrefillDecode") - composed = { - **native.NativeBackend().build(replica, prefill, _PC, base.serving_label(replica), "Standard"), - **grove.GroveBackend().build(replica, decode, _PC, base.serving_label(replica), "Dynamo"), - } - out = routing.apply(composed, replica, _PC) - - manifest = out["model-serving-decode"].spec.forProvider.manifest - leader_clique = _clique(manifest, "leader") - assert leader_clique["labels"]["llm-d.ai/role"] == "decode" - assert leader_clique["labels"]["app"] == "r" - leader = leader_clique["spec"]["podSpec"] - assert [c["name"] for c in leader["containers"]] == ["engine", "pd-sidecar"] - assert any(v.get("emptyDir", {}).get("medium") == "Memory" for v in leader["volumes"]), ( - "leader clique missing Memory /dev/shm volume for NIXL" - ) - engine = next(c for c in leader["containers"] if c["name"] == "engine") - assert "VLLM_NIXL_SIDE_CHANNEL_HOST" in [e["name"] for e in engine["env"]] - - # The worker clique never serves; routing must not touch it at all - - # its labels stay exactly what the Grove backend composed (just the - # queue label), with no role or serving label added. - worker_clique = _clique(manifest, "worker") - worker = worker_clique["spec"]["podSpec"] - assert [c["name"] for c in worker["containers"]] == ["engine"] - assert worker_clique["labels"] == {_QUEUE_LABEL: _QUEUE} - - -# Unified serving (or no serving block) fronts the pods with an -# InferencePool + endpoint picker in place of a plain Service, so requests -# route by prefix cache and load rather than round-robin - one pod or many. -# Mirrors how fn.py composes engines then calls routing.apply. - - -def _unified_apply(copies: int = 1) -> dict[str, k8sobjv1alpha1.Object]: - """Routing for a Unified replica of one native engine.""" - engine = _standalone_engine(copies=copies) - replica = _replica(engines=[engine]) - composed = native.NativeBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - return routing.apply(composed, replica, _PC) - - -def test_unified_fronts_with_pool_and_epp() -> None: - """Unified routing fronts the engine with an InferencePool and endpoint picker.""" - out = _unified_apply() - assert "inference-pool" in out - assert "epp" in out - assert "epp-config" in out - pool = out["inference-pool"].spec.forProvider.manifest - assert pool["kind"] == "InferencePool" - assert pool["spec"]["endpointPickerRef"]["name"] == "r-epp" - - -@pytest.mark.parametrize("copies", [1, 2]) -def test_unified_single_pod_also_pools(copies: int) -> None: - """A single serving pod gets a pool too, as several do.""" - # A single serving pod has nothing to pick between, but still gets the - # pool. Always fronting with one avoids swapping a Service for a pool when a - # second pod appears - a swap that would drop in-flight requests. - out = _unified_apply(copies=copies) - assert "inference-pool" in out - assert "epp" in out - - -def test_unified_fronts_a_leader_worker_set() -> None: - """Unified routing fronts a LeaderWorkerSet.""" - # A Standard multi-node engine composes a LeaderWorkerSet, and unified - # routing must handle that shape too: it reads the engine args for the KV - # block size through _serving_pod_templates, which has to normalize a - # LeaderWorkerSet's leaderTemplate alongside a Deployment's pod template - # and a Grove PodCliqueSet's leader clique. Regression for a shape - # normalization that only knew Deployment and PodCliqueSet and raised - # KeyError on a LeaderWorkerSet. - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _replica(engines=[engine]) - composed = llmd.LLMDBackend().build(replica, engine, _PC, base.serving_label(replica), "Standard") - out = routing.apply(composed, replica, _PC) - assert "inference-pool" in out - assert out["model-serving-main"].spec.forProvider.manifest["kind"] == "LeaderWorkerSet" - - -def test_unified_pool_selects_pods_by_the_serving_label() -> None: - """The pool selects the pods by the serving label they already carry, so no relabeling is needed.""" - pool = _unified_apply()["inference-pool"].spec.forProvider.manifest - assert pool["spec"]["selector"]["matchLabels"] == {base.LABEL_SERVING: "r"} - - -def test_unified_route_targets_inference_pool() -> None: - """Unified routing points the HTTPRoute at the InferencePool.""" - route = _unified_apply()[base.ROUTE_KEY].spec.forProvider.manifest - ref = route["spec"]["rules"][0]["backendRefs"][0] - assert ref["kind"] == "InferencePool" - assert ref["name"] == "r-pool" - - -def test_unified_epp_config_is_unified_not_disaggregated() -> None: - """The unified picker scores by prefix cache and queue depth, with no prefill/decode split.""" - # It scores in a single profile, and still needs the - # approx-prefix-cache-producer that feeds the prefix-cache scorer. - cfg = _unified_apply()["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - assert "prefix-cache-scorer" in cfg - assert "queue-scorer" in cfg - assert "approx-prefix-cache-producer" in cfg - assert "prefill" not in cfg - assert "decider" not in cfg - - -def test_unified_epp_image_and_config_group_are_pinned() -> None: - """Lock the picker image and the EndpointPickerConfig API group for the unified path too.""" - # A deliberate literal (not routing._EPP_IMAGE) so a wrong - # tag/registry or a stale config group is caught in review, not as a - # deploy-time crashloop. Unified has no sidecar, so only the EPP is checked. - out = _unified_apply() - epp = out["epp"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"] - assert next(c["image"] for c in epp if c["name"] == "epp") == "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0" - cfg = out["epp-config"].spec.forProvider.manifest["data"]["epp-config.yaml"] - assert "apiVersion: llm-d.ai/v1alpha1" in cfg - - -def test_unified_epp_pod_carries_config_checksum() -> None: - """The EPP pod template carries a sha256 of its config, so a config change rolls the pod.""" - # The EPP reads its config once at startup, so a config change must roll - # the pod. The pod template carries a sha256 of the rendered config to drive - # that rollout. - template = _unified_apply()["epp"].spec.forProvider.manifest["spec"]["template"] - checksum = template["metadata"]["annotations"]["modelplane.ai/epp-config-checksum"] - assert len(checksum) == 64 - - -# On a Dynamo cluster the native (Standalone) and Grove (Leader/Worker) -# backends inject the ModelExpress P2P env (MX_SERVER_ADDRESS/MODEL_EXPRESS_URL/ -# MX_MODEL_REVISION/MX_P2P_METADATA/POD_*) and the IPC_LOCK security context -# into every engine container of a replica that references a cache. The env is -# inert unless the engine command opts in with --load-format modelexpress. It's -# gated on the cluster's Dynamo stack: on Standard neither backend injects it -# (the portable engine command falls back), and the llm-d backend never does. -# -# HF_HUB_CACHE is deliberately NOT in this set: it's the cache's own env, on -# every stack (see base.cache_env), and ModelExpress reads it only as a -# fallback for its cache root. Keeping it out here is what makes these -# assertions fail if it ever leaks back into modelexpress_env as a duplicate. - -_MODELEXPRESS_ENV_NAMES = { - "MX_SERVER_ADDRESS", - "MODEL_EXPRESS_URL", - "MX_MODEL_REVISION", - "MX_P2P_METADATA", - "POD_NAME", - "POD_UID", - "POD_NAMESPACE", -} -# What a cache-referencing engine carries on Dynamo: the cache's env plus -# the MX bundle, and nothing else. -_CACHE_ENV_NAME = "HF_HUB_CACHE" - - -def _modelexpress_replica(*, cache: bool = True, engines: list[v1alpha1.Engine] | None = None) -> v1alpha1.ModelReplica: - """A replica of engines, referencing the qwen cache unless cache is False.""" - engines = engines if engines is not None else [_standalone_engine(args=[])] - return v1alpha1.ModelReplica( - metadata=metav1.ObjectMeta(name="r", namespace="ml-team"), - spec=v1alpha1.SpecModel( - clusterName="cluster-a", - modelCacheRef=v1alpha1.ModelCacheRef(name="qwen") if cache else None, - engines=engines, +# The EPP prefix-cache producer's blockSizeTokens is derived best-effort from +# the engine flags (#179), so it matches the engine's KV block size. +KV_BLOCK_SIZE_CASES = [ + KvBlockSizeCase( + name="defaults to 16 with no args", + engine_args=[], + want=16, + ), + KvBlockSizeCase( + name="defaults to 16 with no block size flag", + engine_args=["--model=/mnt/models"], + want=16, + ), + KvBlockSizeCase( + name="reads vLLM's --block-size", + engine_args=["--block-size", "32"], + want=32, + ), + KvBlockSizeCase( + name="reads vLLM's --block-size=", + engine_args=["--model=/m", "--block-size=8"], + want=8, + ), + KvBlockSizeCase( + name="reads SGLang's --page-size=", + engine_args=["--page-size=64"], + want=64, + ), + KvBlockSizeCase( + name="a non-integer block size falls back to 16", + engine_args=["--block-size", "auto"], + want=16, + ), +] + + +@pytest.mark.parametrize("case", KV_BLOCK_SIZE_CASES, ids=lambda case: case.name) +def test_kv_block_size(case: KvBlockSizeCase) -> None: + """The KV block size comes from the engine's flags.""" + assert routing._kv_block_size(case.engine_args) == case.want + + +DISAGGREGATED_EPP_CONFIG_YAML_CASES = [ + DisaggregatedEppConfigYamlCase( + name="renders the block size in place of its placeholder", + block_size=32, + want=( + "apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 32\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: disagg-headers-handler\n" + "- type: queue-scorer\n" + "- type: prefill-filter\n" + "- type: decode-filter\n" + "- type: max-score-picker\n" + "- type: prefix-based-pd-decider\n" + " parameters:\n" + " nonCachedTokens: 16\n" + "- type: disagg-profile-handler\n" + " parameters:\n" + " deciders:\n" + " prefill: prefix-based-pd-decider\n" + "schedulingProfiles:\n" + "- name: prefill\n" + " plugins:\n" + " - pluginRef: prefill-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + "- name: decode\n" + " plugins:\n" + " - pluginRef: decode-filter\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" ), - ) + ), +] -def test_grove_gang_gets_modelexpress_env_on_both_cliques() -> None: - """A cached Grove gang on Dynamo gets the ModelExpress env and IPC_LOCK on both cliques.""" - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _modelexpress_replica(engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - # Grove also gets the leader-address alias, unconditional on a cache, - # ahead of the cache env and the ModelExpress bundle. - want_env_names = _MODELEXPRESS_ENV_NAMES | {base.LEADER_ADDRESS_ENV, _CACHE_ENV_NAME} - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - env_names = {e["name"] for e in container["env"]} - assert env_names == want_env_names, f"{clique_name}: {env_names}" - assert container["env"][0] == base.grove_leader_address_env() - server_env = next(e for e in container["env"] if e["name"] == "MX_SERVER_ADDRESS") - # The per-cluster shared server's well-known Service, qualified by - # its namespace because the engine runs in its team's namespace. - assert server_env["value"] == "modelexpress-server.default.svc:8001" - mxurl_env = next(e for e in container["env"] if e["name"] == "MODEL_EXPRESS_URL") - assert mxurl_env["value"] == server_env["value"] - assert container["securityContext"] == {"capabilities": {"add": ["IPC_LOCK"]}} - - -def test_grove_gang_without_cache_gets_no_modelexpress_env() -> None: - """A Grove gang with no cache gets only the leader address alias, and no security context.""" - # No cache means no ModelExpress env or security context, but the - # leader-address alias is unconditional (it doesn't depend on a cache). - engine = _gang_engine(leader_command=_LEADER_CMD, worker_command=_WORKER_CMD) - replica = _modelexpress_replica(cache=False, engines=[engine]) - out = grove.GroveBackend().build(replica, engine, _PC, base.serving_label(replica), "Dynamo") - manifest = out["model-serving-main"].spec.forProvider.manifest - for clique_name in ("leader", "worker"): - container = _clique(manifest, clique_name)["spec"]["podSpec"]["containers"][0] - assert container["env"] == [base.grove_leader_address_env()] - assert "securityContext" not in container - - -def test_native_engine_gets_modelexpress_env_on_dynamo() -> None: - """A cached Standalone engine on Dynamo gets the ModelExpress env and IPC_LOCK.""" - # A Standalone engine on a Dynamo cluster with a cache is as valid a P2P - # peer set as a gang, so it gets the full ModelExpress env and the - # IPC_LOCK security context on its engine container. - replica = _modelexpress_replica() - out = native.NativeBackend().build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Dynamo") - container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] - env = {e["name"]: e for e in container["env"]} - assert set(env) == _MODELEXPRESS_ENV_NAMES | {_CACHE_ENV_NAME} - assert env["MX_SERVER_ADDRESS"]["value"] == "modelexpress-server.default.svc:8001" - assert env["MX_P2P_METADATA"]["value"] == "1" - assert env["HF_HUB_CACHE"]["value"] == "/mnt/models" - # Isolates this cache's P2P source identity, qualified by the - # Modelplane namespace (like cache_pvc_name) so two namespaces' caches - # of the same name can't collide at the cluster's one shared server. - assert env["MX_MODEL_REVISION"]["value"] == base.cache_pvc_name("ml-team", "qwen") - for name, field in ( - ("POD_NAME", "metadata.name"), - ("POD_UID", "metadata.uid"), - ("POD_NAMESPACE", "metadata.namespace"), - ): - assert env[name]["valueFrom"]["fieldRef"]["fieldPath"] == field - assert container["securityContext"] == {"capabilities": {"add": ["IPC_LOCK"]}} - - -def test_native_engine_gets_no_modelexpress_env_on_standard() -> None: - """A cached Standalone engine on Standard gets only the cache env, and no security context.""" - # The same cached Standalone engine on a Standard cluster gets no - # ModelExpress env and no security context: the portable engine command - # falls back. It keeps the cache's own HF_HUB_CACHE, which is not part - # of the ModelExpress bundle and applies on every stack. - replica = _modelexpress_replica() - out = native.NativeBackend().build(replica, replica.spec.engines[0], _PC, base.serving_label(replica), "Standard") - container = out["model-serving-main"].spec.forProvider.manifest["spec"]["template"]["spec"]["containers"][0] - assert container["env"] == [{"name": "HF_HUB_CACHE", "value": "/mnt/models"}] - assert "securityContext" not in container - assert container["args"] == [] - - -# The EPP prefix-cache producer's blockSizeTokens is derived best-effort -# from the engine flags (#179) so it matches the engine's KV block size. - - -def test_kv_block_size_defaults_to_16_when_absent() -> None: - """The KV block size defaults to 16 when no flag sets it.""" - assert routing._kv_block_size([]) == 16 - assert routing._kv_block_size(["--model=/mnt/models"]) == 16 - - -def test_kv_block_size_reads_vllm_block_size() -> None: - """The KV block size comes from vLLM's --block-size.""" - assert routing._kv_block_size(["--block-size", "32"]) == 32 - assert routing._kv_block_size(["--model=/m", "--block-size=8"]) == 8 - - -def test_kv_block_size_reads_sglang_page_size() -> None: - """The KV block size comes from SGLang's --page-size.""" - assert routing._kv_block_size(["--page-size=64"]) == 64 - - -def test_kv_block_size_non_integer_falls_back_to_default() -> None: - """A non-integer block size falls back to 16.""" - assert routing._kv_block_size(["--block-size", "auto"]) == 16 - - -def test_kv_block_size_rendered_config_uses_block_size() -> None: - """The rendered EPP config carries the block size in place of its placeholder.""" - cfg = routing._disaggregated_epp_config_yaml(32) - assert "blockSizeTokens: 32" in cfg - assert "BLOCK_SIZE_TOKENS" not in cfg - - -# The mirrored namespace a replica's objects land in. The expected names are -# spelled out, because compose-inference-cluster creates the namespace and -# compose-model-route and compose-model-cache land objects in it by the same -# derivation, and all four must agree. +@pytest.mark.parametrize("case", DISAGGREGATED_EPP_CONFIG_YAML_CASES, ids=lambda case: case.name) +def test_disaggregated_epp_config_yaml(case: DisaggregatedEppConfigYamlCase) -> None: + """The disaggregated EPP config renders with the engine's KV block size.""" + assert routing._disaggregated_epp_config_yaml(case.block_size) == case.want + +# compose-inference-cluster creates the namespace a replica's objects land in, +# and compose-model-route and compose-model-cache land objects in it by the same +# derivation, so all four must agree on it. REMOTE_NAMESPACE_CASES = [ - pytest.param("ml-team", "mp-ml-team-51733", id="a short namespace keeps its name, prefixed and hashed"), - pytest.param( - # 63 is the longest a namespace can be, so mp- plus it can't be - # used as is. It's truncated to leave room for the hash. - "a" * 63, - "mp-" + "a" * 54 + "-38bfb", - id="the longest valid namespace still yields a valid one", + RemoteNamespaceCase( + name="a short namespace keeps its name, prefixed and hashed", + replica=_replica( + name="r", + namespace="ml-team", + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + want="mp-ml-team-51733", + ), + # 63 is the longest a namespace can be, so mp- plus it can't be used as + # is. It's truncated to leave room for the hash, to 63 characters. The + # lengths are the point, so the names are written as repeats rather than + # 63-character literals. + RemoteNamespaceCase( + name="the longest valid namespace still yields a valid one", + replica=_replica( + name="r", + namespace="a" * 63, + model_cache_ref=None, + serving=None, + engines=[ + v1alpha1.Engine( + name="main", + copies=1, + members=[ + v1alpha1.Member( + role="Standalone", + nodePoolName="frontier", + deviceRequests=[ + v1alpha1.DeviceRequest( + name="gpu", + deviceClassName="gpu.nvidia.com", + count=1, + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], + ) + ], + template=v1alpha1.Template( + spec=v1alpha1.Spec( + containers=[ + v1alpha1.Container( + name="engine", + image="vllm/vllm-openai:latest", + args=["--model=Qwen/Qwen3-0.6B"], + ) + ] + ) + ), + ) + ], + ) + ], + ), + want="mp-" + "a" * 54 + "-38bfb", ), ] -@pytest.mark.parametrize(("namespace", "want"), REMOTE_NAMESPACE_CASES) -def test_remote_namespace(namespace: str, want: str) -> None: +@pytest.mark.parametrize("case", REMOTE_NAMESPACE_CASES, ids=lambda case: case.name) +def test_remote_namespace(case: RemoteNamespaceCase) -> None: """A replica's objects land in a namespace mirroring its own.""" - got = base.remote_namespace(_replica(namespace=namespace)) - assert got == want - assert len(got) <= 63 + assert base.remote_namespace(case.replica) == case.want diff --git a/functions/compose-model-replica/tests/test_fn.py b/functions/compose-model-replica/tests/test_fn.py index 4d2aedd80..44ff065d2 100644 --- a/functions/compose-model-replica/tests/test_fn.py +++ b/functions/compose-model-replica/tests/test_fn.py @@ -28,24 +28,6 @@ from models.ai.modelplane.modelreplica import v1alpha1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -# A GPU device request CEL selector, as compose-model-deployment stamps it. -_GPU_CEL = 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' - -# Unified routing fronts the serving pods with an InferencePool + endpoint -# picker; their manifests are asserted in detail in test_backends. Here we -# only check the function wired the whole set in (and dropped the plain -# Service), then drop their manifests so the golden covers the dispatch, -# wiring and readiness the function itself owns. -_ROUTING_KEYS = { - "inference-pool", - "epp", - "epp-config", - "epp-role", - "epp-rolebinding", - "epp-serviceaccount", - "epp-service", -} - @dataclasses.dataclass class Case: @@ -56,31 +38,8 @@ class Case: want: fnv1.RunFunctionResponse -def _observed_object(*, ready: bool) -> fnv1.Resource: - """A composed provider-kubernetes Object as observed back, with the Ready - condition its readiness policy derives.""" - return fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True" if ready else "False", - "reason": "Available" if ready else "Unavailable", - "lastTransitionTime": "2025-01-01T00:00:00Z", - }, - ], - }, - } - ), - ) - - -def _compose_cases() -> list[Case]: - """The compose cases. Later cases are built from earlier ones.""" +def _model_replica() -> fnv1.Resource: + """The ModelReplica XR test-replica in ml-team, with one Standalone engine.""" xr = v1alpha1.ModelReplica( metadata=metav1.ObjectMeta( name="test-replica", @@ -105,7 +64,11 @@ def _compose_cases() -> list[Case]: name="gpu", deviceClassName="gpu.nvidia.com", count=1, - selectors=[v1alpha1.Selector(cel=_GPU_CEL)], + selectors=[ + v1alpha1.Selector( + cel='device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + ) + ], ), ], template=v1alpha1.Template( @@ -124,447 +87,537 @@ def _compose_cases() -> list[Case]: ), ], ), - ).model_dump(exclude_none=True, mode="json") - - cluster_requirement = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="InferenceCluster", - match_name="cluster-a", ) + return fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json", by_alias=True))) - # Case 1: cluster resolved with providerConfigRef — composes native - # Deployment. First reconcile: none of the composed resources are in - # observed yet, so none are marked ready (the function only asserts - # readiness for a resource it can see in observed state). - req1 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), - ), - ) - req1.required_resources["cluster"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceCluster", - "metadata": {"name": "cluster-a"}, - "spec": { - "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k"}}}, - }, - "status": { - "providerConfigRef": {"name": "cluster-a-pc"}, - "gateway": {"address": "10.0.0.1"}, - }, - } - ) - ) - ) - want1 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - resources={ - "model-serving-main": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", +def _cluster(*, provider_config_ref: str | None) -> fnv1.Resource: + """The InferenceCluster cluster-a, with no status until it reports a providerConfigRef.""" + cluster = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "cluster-a"}, + "spec": { + "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k"}}}, + }, + } + if provider_config_ref is not None: + cluster["status"] = { + "providerConfigRef": {"name": provider_config_ref}, + "gateway": {"address": "10.0.0.1"}, + } + return fnv1.Resource(resource=resource.dict_to_struct(cluster)) + + +def _workload(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the engine's Deployment.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": { + "name": "test-replica-main-6b608", + "namespace": "mp-ml-team-51733", + }, "spec": { - "providerConfigRef": { - "kind": "ClusterProviderConfig", - "name": "cluster-a-pc", - }, - "readiness": { - "policy": "DeriveFromCelQuery", - "celQuery": ( - "has(object.status.conditions) && " - "object.status.conditions.exists(" - 'c, c.type == "Available" && c.status == "True")' - ), - }, - "forProvider": { - "manifest": { - "apiVersion": "apps/v1", - "kind": "Deployment", - "metadata": { - "name": resource.child_name("test-replica", "main"), - "namespace": "mp-ml-team-51733", - }, - "spec": { - "replicas": 1, - "selector": { - "matchLabels": { - "modelplane.ai/workload": resource.child_name( - "test-replica", "main" - ), - }, - }, - "template": { - "metadata": { - "labels": { - "modelplane.ai/serving": "test-replica", - "modelplane.ai/workload": resource.child_name( - "test-replica", "main" - ), - }, - }, - "spec": { - "containers": [ - { - "name": "engine", - "image": "vllm/vllm-openai:latest", - "args": ["--model=Qwen/Qwen3-0.6B"], - "ports": [{"containerPort": 8000}], - "resources": {"claims": [{"name": "devices"}]}, - "volumeMounts": [ - {"name": "dshm", "mountPath": "/dev/shm"}, - ], - "readinessProbe": { - "httpGet": {"path": "/health", "port": 8000}, - "initialDelaySeconds": 30, - "periodSeconds": 10, - "timeoutSeconds": 5, - }, - }, - ], - "volumes": [ - {"name": "dshm", "emptyDir": {"medium": "Memory"}}, - ], - "nodeSelector": {"modelplane.ai/pool": "frontier"}, - "resourceClaims": [ - { - "name": "devices", - "resourceClaimTemplateName": resource.child_name( - "test-replica", "main", "standalone", "devices" - ), - }, - ], - "tolerations": [ - { - "key": "nvidia.com/gpu", - "operator": "Exists", - "effect": "NoSchedule", - }, - ], + "replicas": 1, + "selector": {"matchLabels": {"modelplane.ai/workload": "test-replica-main-6b608"}}, + "template": { + "metadata": { + "labels": { + "modelplane.ai/serving": "test-replica", + "modelplane.ai/workload": "test-replica-main-6b608", + } + }, + "spec": { + "containers": [ + { + "name": "engine", + "image": "vllm/vllm-openai:latest", + "args": ["--model=Qwen/Qwen3-0.6B"], + "ports": [{"containerPort": 8000}], + "volumeMounts": [{"name": "dshm", "mountPath": "/dev/shm"}], + "readinessProbe": { + "httpGet": {"path": "/health", "port": 8000}, + "initialDelaySeconds": 30, + "periodSeconds": 10, + "timeoutSeconds": 5, }, - }, - }, + "resources": {"claims": [{"name": "devices"}]}, + } + ], + "volumes": [{"name": "dshm", "emptyDir": {"medium": "Memory"}}], + "nodeSelector": {"modelplane.ai/pool": "frontier"}, + "resourceClaims": [ + { + "name": "devices", + "resourceClaimTemplateName": "test-replica-main-standalone-devices-e609d", + } + ], + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + } + ], }, }, }, } - ), - ), - "model-route": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", + }, + }, + } + ), + ready=ready, + ) + + +def _route(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the HTTPRoute to the InferencePool.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "HTTPRoute", + "metadata": {"name": "test-replica", "namespace": "mp-ml-team-51733"}, "spec": { - "providerConfigRef": { - "kind": "ClusterProviderConfig", - "name": "cluster-a-pc", - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "gateway.networking.k8s.io/v1", - "kind": "HTTPRoute", - "metadata": { - "name": "test-replica", - "namespace": "mp-ml-team-51733", - }, - "spec": { - "parentRefs": [ - { - "name": "cluster-gateway", - "namespace": "modelplane-system", + "parentRefs": [{"name": "cluster-gateway", "namespace": "modelplane-system"}], + "rules": [ + { + "matches": [ + { + "path": { + "type": "PathPrefix", + "value": "/ml-team/test-replica/", + } + } + ], + "timeouts": {"request": "0s"}, + "filters": [ + { + "type": "URLRewrite", + "urlRewrite": { + "path": { + "type": "ReplacePrefixMatch", + "replacePrefixMatch": "/", + } }, - ], - "rules": [ - { - "matches": [ - { - "path": { - "type": "PathPrefix", - "value": "/ml-team/test-replica/", - }, - }, - ], - "timeouts": {"request": "0s"}, - "filters": [ - { - "type": "URLRewrite", - "urlRewrite": { - "path": { - "type": "ReplacePrefixMatch", - "replacePrefixMatch": "/", - }, - }, - }, - ], - "backendRefs": [ + } + ], + "backendRefs": [ + { + "group": "inference.networking.k8s.io", + "kind": "InferencePool", + "name": "test-replica-pool", + } + ], + } + ], + }, + } + }, + }, + } + ), + ready=ready, + ) + + +def _claim_template(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the engine's ResourceClaimTemplate.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "resource.k8s.io/v1", + "kind": "ResourceClaimTemplate", + "metadata": { + "name": "test-replica-main-standalone-devices-e609d", + "namespace": "mp-ml-team-51733", + }, + "spec": { + "spec": { + "devices": { + "requests": [ + { + "name": "gpu", + "exactly": { + "deviceClassName": "gpu.nvidia.com", + "count": 1, + "selectors": [ { - "group": "inference.networking.k8s.io", - "kind": "InferencePool", - "name": "test-replica-pool", - }, + "cel": { + "expression": 'device.capacity["gpu.nvidia.com"].memory.compareTo(quantity("80Gi")) >= 0' + } + } ], }, - ], - }, - }, - }, + } + ] + } + } }, } - ), - ), - "resource-claim-main-standalone": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", + }, + }, + } + ), + ready=ready, + ) + + +def _inference_pool(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the InferencePool fronting the engine.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "inference.networking.k8s.io/v1", + "kind": "InferencePool", + "metadata": {"name": "test-replica-pool", "namespace": "mp-ml-team-51733"}, "spec": { - "providerConfigRef": { - "kind": "ClusterProviderConfig", - "name": "cluster-a-pc", + "selector": {"matchLabels": {"modelplane.ai/serving": "test-replica"}}, + "targetPorts": [{"number": 8000}], + "endpointPickerRef": { + "name": "test-replica-epp", + "port": {"number": 9002}, + "failureMode": "FailOpen", }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "resource.k8s.io/v1", - "kind": "ResourceClaimTemplate", - "metadata": { - "name": resource.child_name( - "test-replica", "main", "standalone", "devices" - ), - "namespace": "mp-ml-team-51733", - }, - "spec": { - "spec": { - "devices": { - "requests": [ - { - "name": "gpu", - "exactly": { - "deviceClassName": "gpu.nvidia.com", - "count": 1, - "selectors": [ - {"cel": {"expression": _GPU_CEL}}, - ], - }, - }, - ], - }, - }, + }, + } + }, + }, + } + ), + ready=ready, + ) + + +def _epp(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the endpoint picker's Deployment.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": "test-replica-epp", "namespace": "mp-ml-team-51733"}, + "spec": { + "replicas": 1, + "selector": {"matchLabels": {"app": "test-replica-epp"}}, + "template": { + "metadata": { + "labels": {"app": "test-replica-epp"}, + "annotations": { + "modelplane.ai/epp-config-checksum": "20c1dfea3fc4ad41e335cc74edbeb1e8689a607bc4bf7395849ffd2cf0bb2ae1" }, }, + "spec": { + "serviceAccountName": "test-replica-epp", + "containers": [ + { + "name": "epp", + "image": "ghcr.io/llm-d/llm-d-router-endpoint-picker:v0.9.0", + "args": [ + "--pool-name=test-replica-pool", + "--pool-namespace=mp-ml-team-51733", + "--pool-group=inference.networking.k8s.io", + "--config-file=/config/epp-config.yaml", + "--grpc-port=9002", + ], + "ports": [ + {"name": "grpc", "containerPort": 9002}, + {"name": "grpc-health", "containerPort": 9003}, + ], + "volumeMounts": [{"name": "config", "mountPath": "/config"}], + } + ], + "volumes": [ + { + "name": "config", + "configMap": {"name": "test-replica-epp"}, + } + ], + }, }, }, } - ), - ), - }, + }, + }, + } ), - conditions=[ - fnv1.Condition( - type="ModelAccepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="Deploying", - ), - fnv1.Condition( - type="ModelReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForModel", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Composing vllm/vllm-openai:latest on cluster-a", - ), - ], - context=structpb.Struct(), + ready=ready, ) - want1.requirements.resources["cluster"].CopyFrom(cluster_requirement) - # Case 2: cluster not resolved — early return with conditions. - req2 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), + +def _epp_config(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the endpoint picker's ConfigMap.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "test-replica-epp", "namespace": "mp-ml-team-51733"}, + "data": { + "epp-config.yaml": ( + "apiVersion: llm-d.ai/v1alpha1\n" + "kind: EndpointPickerConfig\n" + "plugins:\n" + "- type: approx-prefix-cache-producer\n" + " parameters:\n" + " autoTune: false\n" + " blockSizeTokens: 16\n" + " maxPrefixBlocksToMatch: 256\n" + " lruCapacityPerServer: 31250\n" + "- type: prefix-cache-scorer\n" + "- type: queue-scorer\n" + "- type: max-score-picker\n" + "schedulingProfiles:\n" + "- name: default\n" + " plugins:\n" + " - pluginRef: max-score-picker\n" + " - pluginRef: prefix-cache-scorer\n" + " weight: 2\n" + " - pluginRef: queue-scorer\n" + " weight: 1\n" + ) + }, + } + }, + }, + } ), + ready=ready, ) - want2 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - # Nothing is composed while waiting, so the XR is marked not ready - # rather than left to aggregate to trivially ready. - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - conditions=[ - fnv1.Condition( - type="ModelAccepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - fnv1.Condition( - type="ModelReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForModel", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for cluster to be resolved", - ), - ], - context=structpb.Struct(), - ) - want2.requirements.resources["cluster"].CopyFrom(cluster_requirement) - # Case 3: cluster resolved but no providerConfigRef — early return. - req3 = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(xr)), +def _epp_role(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the endpoint picker's Role.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "Role", + "metadata": {"name": "test-replica-epp", "namespace": "mp-ml-team-51733"}, + "rules": [ + { + "apiGroups": [""], + "resources": ["pods"], + "verbs": ["get", "watch", "list"], + }, + { + "apiGroups": ["inference.networking.k8s.io"], + "resources": ["inferencepools"], + "verbs": ["get", "watch", "list"], + }, + { + "apiGroups": ["inference.networking.x-k8s.io"], + "resources": ["inferenceobjectives"], + "verbs": ["get", "watch", "list"], + }, + ], + } + }, + }, + } ), + ready=ready, ) - req3.required_resources["cluster"].items.append( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "InferenceCluster", - "metadata": {"name": "cluster-a"}, - "spec": { - "cluster": {"source": "Existing", "existing": {"secretRef": {"name": "k"}}}, + + +def _epp_role_binding(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the endpoint picker's RoleBinding.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "RoleBinding", + "metadata": {"name": "test-replica-epp", "namespace": "mp-ml-team-51733"}, + "subjects": [ + { + "kind": "ServiceAccount", + "name": "test-replica-epp", + "namespace": "mp-ml-team-51733", + } + ], + "roleRef": { + "apiGroup": "rbac.authorization.k8s.io", + "kind": "Role", + "name": "test-replica-epp", + }, + } }, - } - ) - ) + }, + } + ), + ready=ready, ) - want3 = fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), - conditions=[ - fnv1.Condition( - type="ModelAccepted", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForCluster", - ), - fnv1.Condition( - type="ModelReady", - status=fnv1.STATUS_CONDITION_FALSE, - reason="WaitingForModel", - ), - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for cluster providerConfigRef", - ), - ], - context=structpb.Struct(), + +def _epp_service_account(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the endpoint picker's ServiceAccount.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ServiceAccount", + "metadata": {"name": "test-replica-epp", "namespace": "mp-ml-team-51733"}, + } + }, + }, + } + ), + ready=ready, ) - want3.requirements.resources["cluster"].CopyFrom(cluster_requirement) - - # The routing objects' manifests are dropped from the golden (see - # _ROUTING_KEYS). - for key in _ROUTING_KEYS: - want1.desired.resources[key].CopyFrom(fnv1.Resource()) - - # Case 4: the resources from case 1 now exist in observed, and the - # workload Object reports Available (so its derived Ready is True). The - # function marks each observed resource ready once its Object reports - # Ready: the workload because it's serving and the rest because existing - # is being ready for them. Built from case 1, mutating only what the - # observed-ready transition changes: the three ready flags, the - # acceptance/readiness conditions, and the dropped first-reconcile event. - req4 = fnv1.RunFunctionRequest() - req4.CopyFrom(req1) - # The workload Object as provider-kubernetes observes it back: applied - # (atProvider.manifest populated) and Available (its derived Ready=True). - req4.observed.resources["model-serving-main"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "spec": {"forProvider": {"manifest": {"kind": "Deployment"}}}, - "status": { - "atProvider": {"manifest": {"kind": "Deployment"}}, - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2025-01-01T00:00:00Z", + + +def _epp_service(*, ready: fnv1.Ready) -> fnv1.Resource: + """The Object composing the endpoint picker's Service.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "cluster-a-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Service", + "metadata": {"name": "test-replica-epp", "namespace": "mp-ml-team-51733"}, + "spec": { + "selector": {"app": "test-replica-epp"}, + "ports": [ + { + "name": "grpc-ext-proc", + "port": 9002, + "targetPort": 9002, + "appProtocol": "http2", + } + ], }, - ], + } }, - } - ), - ) + }, + } + ), + ready=ready, + ) + + +def _observed_workload() -> fnv1.Resource: + """The workload Object as observed back, applied and Available, so its derived Ready is True.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": {"forProvider": {"manifest": {"kind": "Deployment"}}}, + "status": { + "atProvider": {"manifest": {"kind": "Deployment"}}, + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2025-01-01T00:00:00Z", + }, + ], + }, + } + ), ) - # The other two have no runtime readiness to wait on, so under their - # SuccessfulCreate policy provider-kubernetes reports them Ready once - # applied. (The InferencePool + endpoint picker resources aren't - # observed here, so they stay unready.) - for key in ("model-route", "resource-claim-main-standalone"): - req4.observed.resources[key].CopyFrom(_observed_object(ready=True)) - - want4 = fnv1.RunFunctionResponse() - want4.CopyFrom(want1) - for key in ("model-serving-main", "model-route", "resource-claim-main-standalone"): - want4.desired.resources[key].ready = fnv1.READY_TRUE - del want4.conditions[:] - want4.conditions.extend( - [ - fnv1.Condition(type="ModelAccepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Accepted"), - fnv1.Condition(type="ModelReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Serving"), - ] + + +def _observed_object(*, ready: bool) -> fnv1.Resource: + """A composed Object as observed back, with the Ready condition its readiness policy derives.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True" if ready else "False", + "reason": "Available" if ready else "Unavailable", + "lastTransitionTime": "2025-01-01T00:00:00Z", + }, + ], + }, + } + ), ) - # The "Composing ..." event fires only the first reconcile (model-serving - # not yet observed), so it's gone now. - del want4.results[:] - - # Case 5: everything from case 4 plus the routing objects is observed, - # but the endpoint picker's Service failed to apply, say because its - # name was invalid, so its Object isn't Ready. Being observed isn't - # being applied, so it stays unready and holds the replica unready with - # it. Built from case 4, mutating only the routing objects' ready flags. - req5 = fnv1.RunFunctionRequest() - req5.CopyFrom(req4) - for key in _ROUTING_KEYS: - req5.observed.resources[key].CopyFrom(_observed_object(ready=key != "epp-service")) - - want5 = fnv1.RunFunctionResponse() - want5.CopyFrom(want4) - for key in _ROUTING_KEYS - {"epp-service"}: - want5.desired.resources[key].ready = fnv1.READY_TRUE - - # Case 6: as case 5, but everything applied and the endpoint picker's - # Deployment isn't Available yet, so its Object's CEL-derived Ready is - # False. The gateway fails closed without a picker, so the replica stays - # unready with it. - req6 = fnv1.RunFunctionRequest() - req6.CopyFrom(req4) - for key in _ROUTING_KEYS: - req6.observed.resources[key].CopyFrom(_observed_object(ready=key != "epp")) - - want6 = fnv1.RunFunctionResponse() - want6.CopyFrom(want4) - for key in _ROUTING_KEYS - {"epp"}: - want6.desired.resources[key].ready = fnv1.READY_TRUE - - return [ - Case(name="cluster ready composes native Deployment", req=req1, want=want1), - Case(name="cluster not resolved returns waiting conditions", req=req2, want=want2), - Case(name="cluster without providerConfigRef returns waiting conditions", req=req3, want=want3), - Case(name="observed resources are marked ready", req=req4, want=want4), - Case(name="an object that failed to apply stays unready", req=req5, want=want5), - Case(name="an unavailable endpoint picker stays unready", req=req6, want=want6), - ] def _to_dict(msg: message.Message) -> dict: @@ -572,19 +625,321 @@ def _to_dict(msg: message.Message) -> dict: return json.loads(json_format.MessageToJson(msg, sort_keys=True)) -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +COMPOSE_CASES = [ + # The cluster is resolved with a providerConfigRef, so the function composes + # a native Deployment. On this first reconcile none of the composed + # resources are observed yet, so none are marked ready: the function only + # asserts readiness for a resource it can see in observed state. + # + # Unified routing fronts the serving pods with an InferencePool and endpoint + # picker rather than a plain Service, and every object lands in the + # namespace mirroring the replica's. The device request's CEL selector, here + # and throughout, is as compose-model-deployment stamps it. + Case( + name="cluster with providerConfigRef composes native Deployment", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_model_replica()), + required_resources={"cluster": fnv1.Resources(items=[_cluster(provider_config_ref="cluster-a-pc")])}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + resources={ + "model-serving-main": _workload(ready=fnv1.READY_UNSPECIFIED), + "model-route": _route(ready=fnv1.READY_UNSPECIFIED), + "resource-claim-main-standalone": _claim_template(ready=fnv1.READY_UNSPECIFIED), + "inference-pool": _inference_pool(ready=fnv1.READY_UNSPECIFIED), + "epp": _epp(ready=fnv1.READY_UNSPECIFIED), + "epp-config": _epp_config(ready=fnv1.READY_UNSPECIFIED), + "epp-role": _epp_role(ready=fnv1.READY_UNSPECIFIED), + "epp-rolebinding": _epp_role_binding(ready=fnv1.READY_UNSPECIFIED), + "epp-serviceaccount": _epp_service_account(ready=fnv1.READY_UNSPECIFIED), + "epp-service": _epp_service(ready=fnv1.READY_UNSPECIFIED), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Composing vllm/vllm-openai:latest on cluster-a", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "cluster": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_name="cluster-a", + ), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelAccepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="Deploying", + ), + fnv1.Condition( + type="ModelReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForModel", + ), + ], + ), + ), + # The cluster isn't resolved yet, so the function returns early with waiting + # conditions. Nothing is composed while waiting, so the XR is marked not + # ready rather than left to aggregate to trivially ready. + Case( + name="cluster not resolved returns waiting conditions", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_model_replica()), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for cluster to be resolved", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "cluster": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_name="cluster-a", + ), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelAccepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + fnv1.Condition( + type="ModelReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForModel", + ), + ], + ), + ), + # The cluster is resolved but has no providerConfigRef yet, so the function + # returns early. + Case( + name="cluster without providerConfigRef returns waiting conditions", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_model_replica()), + required_resources={"cluster": fnv1.Resources(items=[_cluster(provider_config_ref=None)])}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=fnv1.Resource(ready=fnv1.READY_FALSE)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for cluster providerConfigRef", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "cluster": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_name="cluster-a", + ), + }, + ), + conditions=[ + fnv1.Condition( + type="ModelAccepted", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForCluster", + ), + fnv1.Condition( + type="ModelReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForModel", + ), + ], + ), + ), + # The workload, route and claim template from the first reconcile are now + # observed, and the workload Object reports Available, so its derived Ready + # is True. The function marks each observed resource ready once its Object + # reports Ready: the workload because it's serving, and the route and claim + # template because, with no runtime readiness to wait on, their + # SuccessfulCreate Objects report Ready once applied. The InferencePool and + # endpoint picker objects aren't observed yet, so they stay unready. The + # "Composing ..." event fires only on the first reconcile, before the + # workload is observed, so there are no results. + Case( + name="observed resources are marked ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_replica(), + resources={ + "model-serving-main": _observed_workload(), + "model-route": _observed_object(ready=True), + "resource-claim-main-standalone": _observed_object(ready=True), + }, + ), + required_resources={"cluster": fnv1.Resources(items=[_cluster(provider_config_ref="cluster-a-pc")])}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + resources={ + "model-serving-main": _workload(ready=fnv1.READY_TRUE), + "model-route": _route(ready=fnv1.READY_TRUE), + "resource-claim-main-standalone": _claim_template(ready=fnv1.READY_TRUE), + "inference-pool": _inference_pool(ready=fnv1.READY_UNSPECIFIED), + "epp": _epp(ready=fnv1.READY_UNSPECIFIED), + "epp-config": _epp_config(ready=fnv1.READY_UNSPECIFIED), + "epp-role": _epp_role(ready=fnv1.READY_UNSPECIFIED), + "epp-rolebinding": _epp_role_binding(ready=fnv1.READY_UNSPECIFIED), + "epp-serviceaccount": _epp_service_account(ready=fnv1.READY_UNSPECIFIED), + "epp-service": _epp_service(ready=fnv1.READY_UNSPECIFIED), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "cluster": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_name="cluster-a", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ModelAccepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Accepted"), + fnv1.Condition(type="ModelReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Serving"), + ], + ), + ), + # Everything is observed, but the endpoint picker's Service Object isn't + # Ready, as when the Service failed to apply, say because its name was + # invalid. Being observed isn't being applied, so it stays unready, and + # Crossplane holds the XR unready with it. + Case( + name="an object that failed to apply stays unready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_replica(), + resources={ + "model-serving-main": _observed_workload(), + "model-route": _observed_object(ready=True), + "resource-claim-main-standalone": _observed_object(ready=True), + "inference-pool": _observed_object(ready=True), + "epp": _observed_object(ready=True), + "epp-config": _observed_object(ready=True), + "epp-role": _observed_object(ready=True), + "epp-rolebinding": _observed_object(ready=True), + "epp-serviceaccount": _observed_object(ready=True), + "epp-service": _observed_object(ready=False), + }, + ), + required_resources={"cluster": fnv1.Resources(items=[_cluster(provider_config_ref="cluster-a-pc")])}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + resources={ + "model-serving-main": _workload(ready=fnv1.READY_TRUE), + "model-route": _route(ready=fnv1.READY_TRUE), + "resource-claim-main-standalone": _claim_template(ready=fnv1.READY_TRUE), + "inference-pool": _inference_pool(ready=fnv1.READY_TRUE), + "epp": _epp(ready=fnv1.READY_TRUE), + "epp-config": _epp_config(ready=fnv1.READY_TRUE), + "epp-role": _epp_role(ready=fnv1.READY_TRUE), + "epp-rolebinding": _epp_role_binding(ready=fnv1.READY_TRUE), + "epp-serviceaccount": _epp_service_account(ready=fnv1.READY_TRUE), + "epp-service": _epp_service(ready=fnv1.READY_UNSPECIFIED), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "cluster": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_name="cluster-a", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ModelAccepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Accepted"), + fnv1.Condition(type="ModelReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Serving"), + ], + ), + ), + # Everything applied, but the endpoint picker's Deployment isn't Available + # yet, so its Object's CEL-derived Ready is False and it stays unready. + # Crossplane holds the XR unready with it, because the gateway fails closed + # without a picker, whatever the InferencePool's failureMode says. + Case( + name="an unavailable endpoint picker stays unready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_replica(), + resources={ + "model-serving-main": _observed_workload(), + "model-route": _observed_object(ready=True), + "resource-claim-main-standalone": _observed_object(ready=True), + "inference-pool": _observed_object(ready=True), + "epp": _observed_object(ready=False), + "epp-config": _observed_object(ready=True), + "epp-role": _observed_object(ready=True), + "epp-rolebinding": _observed_object(ready=True), + "epp-serviceaccount": _observed_object(ready=True), + "epp-service": _observed_object(ready=True), + }, + ), + required_resources={"cluster": fnv1.Resources(items=[_cluster(provider_config_ref="cluster-a-pc")])}, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + resources={ + "model-serving-main": _workload(ready=fnv1.READY_TRUE), + "model-route": _route(ready=fnv1.READY_TRUE), + "resource-claim-main-standalone": _claim_template(ready=fnv1.READY_TRUE), + "inference-pool": _inference_pool(ready=fnv1.READY_TRUE), + "epp": _epp(ready=fnv1.READY_UNSPECIFIED), + "epp-config": _epp_config(ready=fnv1.READY_TRUE), + "epp-role": _epp_role(ready=fnv1.READY_TRUE), + "epp-rolebinding": _epp_role_binding(ready=fnv1.READY_TRUE), + "epp-serviceaccount": _epp_service_account(ready=fnv1.READY_TRUE), + "epp-service": _epp_service(ready=fnv1.READY_TRUE), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "cluster": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="InferenceCluster", + match_name="cluster-a", + ), + }, + ), + conditions=[ + fnv1.Condition(type="ModelAccepted", status=fnv1.STATUS_CONDITION_TRUE, reason="Accepted"), + fnv1.Condition(type="ModelReady", status=fnv1.STATUS_CONDITION_TRUE, reason="Serving"), + ], + ), + ), +] + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: - """The function dispatches to a backend to compose serving resources on a remote cluster.""" + """RunFunction dispatches to a backend to compose serving resources on a remote cluster.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - got_dict = _to_dict(got) - resources = got_dict.get("desired", {}).get("resources", {}) - if "model-serving-main" in resources: - assert set(resources) >= _ROUTING_KEYS - assert "model-service" not in resources - # The routing objects land in the mirrored namespace too, - # before they're dropped from the golden below. - for key in _ROUTING_KEYS: - manifest = resources[key]["resource"]["spec"]["forProvider"]["manifest"] - assert manifest["metadata"]["namespace"] == "mp-ml-team-51733", key - del resources[key]["resource"] - assert got_dict == _to_dict(case.want) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-model-route/tests/test_fn.py b/functions/compose-model-route/tests/test_fn.py index c19c64d14..c0e4e3a64 100644 --- a/functions/compose-model-route/tests/test_fn.py +++ b/functions/compose-model-route/tests/test_fn.py @@ -15,7 +15,6 @@ """Tests for the compose-model-route function.""" import asyncio -import base64 import dataclasses import json @@ -26,23 +25,8 @@ from google.protobuf import duration_pb2 as durationpb from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb -from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 -from models.ai.modelplane.inferencegateway import v1alpha1 as igv1alpha1 -from models.ai.modelplane.modelendpoint import v1alpha1 as mev1alpha1 from models.ai.modelplane.modelroute import v1alpha1 - -_NS = "ml-team" -_SVC = "assistant" -_MODEL = f"{_NS}/{_SVC}" -_GW = "eu" -_CLUSTER_CA = "-----BEGIN CERTIFICATE-----\ncluster\n-----END CERTIFICATE-----\n" -_CLIENT_CA = "-----BEGIN CERTIFICATE-----\nclient\n-----END CERTIFICATE-----\n" - - -def _be(ep: str) -> str: - """A composed backend object's name: child_name of the ModelRoute's own name - (service-gateway) and the endpoint's.""" - return resource.child_name(f"{_SVC}-{_GW}", ep) +from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @dataclasses.dataclass @@ -54,200 +38,479 @@ class Case: want: fnv1.RunFunctionResponse -def _entry( - label: str, *, name: str | None = None, priority: int | None = None, weight: int | None = None -) -> v1alpha1.Endpoint: - kwargs = {} - if priority is not None: - kwargs["priority"] = priority - if weight is not None: - kwargs["weight"] = weight - return v1alpha1.Endpoint( - name=name or label, - selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": label}), - **kwargs, +def _model_route(*, endpoints: list[v1alpha1.Endpoint]) -> fnv1.Resource: + """The ModelRoute XR pinning ml-team's assistant service to gateway eu.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ModelRoute( + apiVersion="modelplane.ai/v1alpha1", + kind="ModelRoute", + metadata=metav1.ObjectMeta(name="assistant-eu", namespace="ml-team"), + spec=v1alpha1.Spec( + gatewayName="eu", + serviceName="assistant", + endpoints=endpoints, + timeouts=v1alpha1.Timeouts(request="600s", idle="0s"), + ), + ).model_dump(exclude_none=True, mode="json", by_alias=True) + ) ) -def _route_xr(entries: list[v1alpha1.Endpoint], *, gateway: str = _GW) -> dict: - xr = v1alpha1.ModelRoute( - apiVersion="modelplane.ai/v1alpha1", - kind="ModelRoute", - metadata={"name": f"{_SVC}-{gateway}", "namespace": _NS}, - spec=v1alpha1.Spec( - gatewayName=gateway, - serviceName=_SVC, - endpoints=entries, - timeouts=v1alpha1.Timeouts(request="600s", idle="0s"), - ), +def _desired_model_route(*, address: str | None, total_endpoints: int, ready_endpoints: int) -> fnv1.Resource: + """The desired ModelRoute XR, not yet ready, reporting its model, its gateway's address and its endpoint counts.""" + status: dict = {"model": "ml-team/assistant"} + if address is not None: + status["address"] = address + status["endpoints"] = {"total": total_endpoints, "ready": ready_endpoints} + return fnv1.Resource(resource=resource.dict_to_struct({"status": status}), ready=fnv1.READY_FALSE) + + +def _inference_gateway(*, tls: bool, address: str, client_ca_published: bool) -> fnv1.Resource: + """The InferenceGateway eu on cluster gw-eu, as the gateway requirement returns it.""" + spec: dict = {"clusterName": "gw-eu"} + if tls: + spec["tls"] = {"certificateRefs": [{"name": "eu-tls"}]} + status: dict = {"address": address} + if client_ca_published: + status["clientCACertificate"] = "-----BEGIN CERTIFICATE-----\nclient\n-----END CERTIFICATE-----\n" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceGateway", + "metadata": {"name": "eu"}, + "spec": spec, + "status": status, + } + ) ) - return xr.model_dump(exclude_none=True, mode="json", by_alias=True) -def _endpoint( - name: str, +def _inference_cluster(*, gateway_ca_published: bool) -> fnv1.Resource: + """The InferenceCluster gw-eu the gateway runs on, as the clusters requirement returns it.""" + status: dict = {"providerConfigRef": {"name": "gw-eu-pc"}} + if gateway_ca_published: + status["gateway"] = {"caCertificate": "-----BEGIN CERTIFICATE-----\ncluster\n-----END CERTIFICATE-----\n"} + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceCluster", + "metadata": {"name": "gw-eu"}, + "spec": { + "cluster": { + "source": "Existing", + "existing": {"secretRef": {"name": "gw-eu-kubeconfig", "key": "kubeconfig"}}, + }, + "stack": "Standard", + }, + "status": status, + } + ) + ) + + +def _composed_endpoint(*, model: str | None) -> fnv1.Resource: + """The ready ModelEndpoint self, which Modelplane composed on gw-eu, as an endpoints requirement returns it.""" + spec: dict = {"origin": "https://gw-eu.example.com"} + if model is not None: + spec["model"] = model + spec["api"] = {"schema": "OpenAI", "prefix": "/v1"} + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": { + "name": "self", + "namespace": "ml-team", + "labels": {"modelplane.ai/cluster": "gw-eu", "modelplane.ai/deployment": "d"}, + }, + "spec": spec, + "status": { + "conditions": [ + { + "type": "EndpointReady", + "status": "True", + "reason": "EndpointUsable", + "lastTransitionTime": "2026-06-08T00:00:00Z", + } + ] + }, + } + ) + ) + + +def _third_party_endpoint( *, + name: str, origin: str, - model: str | None = None, - api: mev1alpha1.Api | None = None, - credential: str | None = None, - ready: bool = True, - composed: bool = False, -) -> dict: - ep = mev1alpha1.ModelEndpoint( - apiVersion="modelplane.ai/v1alpha1", - kind="ModelEndpoint", - metadata={"name": name, "namespace": _NS}, - spec=mev1alpha1.Spec( - origin=origin, - **({"model": model} if model else {}), - **({"api": api} if api else {}), - **( - { - "credential": mev1alpha1.Credential( - method="APIKey", apiKey=mev1alpha1.ApiKey(secretRef=mev1alpha1.SecretRef(name=credential)) - ) - } - if credential - else {} - ), - ), + model: str | None, + schema: str, + api_key_secret: str | None, + ready: bool, + reason: str, +) -> fnv1.Resource: + """A ModelEndpoint without the cluster label, so third-party; ready and reason set its EndpointReady condition.""" + spec: dict = {"origin": origin} + if model is not None: + spec["model"] = model + spec["api"] = {"schema": schema, "prefix": "/v1"} + if api_key_secret is not None: + spec["credential"] = {"method": "APIKey", "apiKey": {"secretRef": {"name": api_key_secret, "key": "apiKey"}}} + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelEndpoint", + "metadata": {"name": name, "namespace": "ml-team"}, + "spec": spec, + "status": { + "conditions": [ + { + "type": "EndpointReady", + "status": "True" if ready else "False", + "reason": reason, + "lastTransitionTime": "2026-06-08T00:00:00Z", + } + ] + }, + } + ) ) - d = ep.model_dump(exclude_none=True, mode="json", by_alias=True) - if composed: - d["metadata"]["labels"] = {"modelplane.ai/cluster": "gw-eu", "modelplane.ai/deployment": "d"} - d["status"] = { - "conditions": [ + + +def _api_key_secret(*, name: str, data: dict[str, str]) -> fnv1.Resource: + """A Secret in ml-team holding an endpoint's API key, as a credential requirement returns it.""" + return fnv1.Resource( + resource=resource.dict_to_struct( { - "type": "EndpointReady", - "status": "True" if ready else "False", - "reason": "EndpointUsable" if ready else "CredentialMissing", - "lastTransitionTime": "2026-06-08T00:00:00Z", + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": name, "namespace": "ml-team"}, + "data": data, } - ] - } - return d + ) + ) -def _gateway(*, client_ca: str | None = _CLIENT_CA, address: str | None = "203.0.113.1", tls: bool = False) -> dict: - spec = igv1alpha1.Spec(clusterName="gw-eu") - if tls: - spec.tls = igv1alpha1.Tls(certificateRefs=[igv1alpha1.CertificateRef(name="eu-tls")]) - gw = igv1alpha1.InferenceGateway( - apiVersion="modelplane.ai/v1alpha1", - kind="InferenceGateway", - metadata={"name": _GW}, - spec=spec, +def _composed_endpoint_backend() -> fnv1.Resource: + """The composed Backend for self, pinning this route's copy of gw-eu's CA and presenting its client certificate.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "Backend", + "metadata": {"name": "assistant-eu-self-a42e6", "namespace": "mp-ml-team-51733"}, + "spec": { + "endpoints": [{"fqdn": {"hostname": "gw-eu.example.com", "port": 443}}], + "tls": { + "caCertificateRefs": [ + {"kind": "ConfigMap", "group": "", "name": "assistant-eu-gw-eu-ca-3dd16"} + ], + "sni": "gw-eu.example.com", + "clientCertificateRef": { + "kind": "Secret", + "group": "", + "name": "assistant-eu-client-08324", + }, + }, + }, + } + }, + }, + } + ) ) - d = gw.model_dump(exclude_none=True, mode="json", by_alias=True) - status: dict = {} - if address: - status["address"] = address - if client_ca: - status["clientCACertificate"] = client_ca - if status: - d["status"] = status - return d - - -def _cluster(name: str, *, provider_config: str | None = "gw-eu-pc", ca: str | None = _CLUSTER_CA) -> dict: - c = icv1alpha1.InferenceCluster( - apiVersion="modelplane.ai/v1alpha1", - kind="InferenceCluster", - metadata={"name": name}, - spec=icv1alpha1.Spec( - cluster=icv1alpha1.Cluster( - source="Existing", - existing=icv1alpha1.Existing( - secretRef=icv1alpha1.SecretRef(name=f"{name}-kubeconfig", key="kubeconfig") - ), - ) - ), + + +def _composed_endpoint_ai_backend() -> fnv1.Resource: + """The composed AIServiceBackend for self, which keeps the caller header because Modelplane operates self.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "aigateway.envoyproxy.io/v1beta1", + "kind": "AIServiceBackend", + "metadata": {"name": "assistant-eu-self-a42e6", "namespace": "mp-ml-team-51733"}, + "spec": { + "schema": {"name": "OpenAI", "prefix": "/v1"}, + "backendRef": { + "group": "gateway.envoyproxy.io", + "kind": "Backend", + "name": "assistant-eu-self-a42e6", + }, + }, + } + }, + }, + } + ) + ) + + +def _third_party_backend(*, name: str, hostname: str) -> fnv1.Resource: + """A composed Backend for a third-party endpoint, which trusts the system CAs.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "Backend", + "metadata": {"name": name, "namespace": "mp-ml-team-51733"}, + "spec": { + "endpoints": [{"fqdn": {"hostname": hostname, "port": 443}}], + "tls": {"wellKnownCACertificates": "System", "sni": hostname}, + }, + } + }, + }, + } + ) + ) + + +def _third_party_ai_backend(*, name: str, schema: str) -> fnv1.Resource: + """A composed AIServiceBackend for a third-party endpoint, which strips the caller header.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "aigateway.envoyproxy.io/v1beta1", + "kind": "AIServiceBackend", + "metadata": {"name": name, "namespace": "mp-ml-team-51733"}, + "spec": { + "schema": {"name": schema, "prefix": "/v1"}, + "backendRef": {"group": "gateway.envoyproxy.io", "kind": "Backend", "name": name}, + "headerMutation": {"remove": ["x-modelplane-caller"]}, + }, + } + }, + }, + } + ) ) - d = c.model_dump(exclude_none=True, mode="json", by_alias=True) - status: dict = {} - if provider_config: - status["providerConfigRef"] = {"name": provider_config} - if ca: - status["gateway"] = {"caCertificate": ca} - if status: - d["status"] = status - return d - - -def _secret(name: str, data: dict[str, str]) -> dict: - return { - "apiVersion": "v1", - "kind": "Secret", - "metadata": {"name": name, "namespace": _NS}, - "data": {k: base64.b64encode(v.encode()).decode() for k, v in data.items()}, - } - - -def _required(**resources) -> dict: # noqa: ANN003 - return { - name: fnv1.Resources(items=[fnv1.Resource(resource=resource.dict_to_struct(r)) for r in items]) - for name, items in resources.items() - } - - -def _requirements(entries: list[v1alpha1.Endpoint], *, credentials: dict[str, str] | None = None) -> fnv1.Requirements: - """The requirements the function emits: the named gateway, every cluster, a - ModelEndpoint selector per entry, and a Secret per endpoint that names a - credential (endpoint name -> Secret name).""" - reqs = { - "gateway": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name=_GW), - "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), - } - for entry in entries: - reqs[f"endpoints-{entry.name}"] = fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", - kind="ModelEndpoint", - namespace=_NS, - match_labels=fnv1.MatchLabels(labels=dict(entry.selector.matchLabels)), + + +def _credential(*, name: str, api_key: str) -> fnv1.Resource: + """The composed copy of an endpoint's API key Secret, under the apiKey key the AI Gateway reads.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": name, "namespace": "mp-ml-team-51733"}, + "type": "Opaque", + "data": {"apiKey": api_key}, + } + }, + }, + } ) - for endpoint, secret in (credentials or {}).items(): - reqs[f"credential-{endpoint}"] = fnv1.ResourceSelector( - api_version="v1", kind="Secret", namespace=_NS, match_name=secret + ) + + +def _credential_policy(*, name: str, auth: dict) -> fnv1.Resource: + """The composed BackendSecurityPolicy sending an endpoint's API key the way auth says.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "aigateway.envoyproxy.io/v1beta1", + "kind": "BackendSecurityPolicy", + "metadata": {"name": name, "namespace": "mp-ml-team-51733"}, + "spec": { + **auth, + "targetRefs": [ + {"group": "aigateway.envoyproxy.io", "kind": "AIServiceBackend", "name": name} + ], + }, + } + }, + }, + } ) - return fnv1.Requirements(resources=reqs) - - -def _not_ready( - status: dict, reason: str, message: str, requirements: fnv1.Requirements, *, warning: str | None = None -) -> fnv1.RunFunctionResponse: - """The whole response for a pass that composes nothing: the status counts so - far, a not-ready composite, one RoutingReady=False condition, and the reason - as a result. A warning about endpoints dropped before the tier emptied - precedes it.""" - results = [] - if warning is not None: - results.append(fnv1.Result(severity=fnv1.SEVERITY_WARNING, message=warning)) - results.append(fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message=message)) - return fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct({"status": status}), - ready=fnv1.READY_FALSE, - ) - ), - context=structpb.Struct(), - requirements=requirements, - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=reason, - message=message, - ) - ], - results=results, ) -def _manifest(rsp: fnv1.RunFunctionResponse, key: str) -> dict: - return resource.struct_to_dict(rsp.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] +def _cluster_ca() -> fnv1.Resource: + """The composed ConfigMap holding gw-eu's gateway CA, named for this route so no other route composes it.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "assistant-eu-gw-eu-ca-3dd16", "namespace": "mp-ml-team-51733"}, + "data": {"ca.crt": "-----BEGIN CERTIFICATE-----\ncluster\n-----END CERTIFICATE-----\n"}, + } + }, + }, + } + ) + ) + + +def _client_certificate() -> fnv1.Resource: + """The composed client Certificate the composed endpoint's backend presents.""" + # Issued from the gateway's CA ClusterIssuer into this namespace. Named for + # this route, so no other route in the namespace composes it, and deleted + # with the route, so it sets no managementPolicies. + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.conditions) && " + "object.status.conditions.exists(c, c.type == 'Ready' && c.status == 'True')" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "Certificate", + "metadata": {"name": "assistant-eu-client-08324", "namespace": "mp-ml-team-51733"}, + "spec": { + "secretName": "assistant-eu-client-08324", + "commonName": "inference-gateway-eu", + "usages": ["client auth", "digital signature", "key encipherment"], + "duration": "2160h", + "renewBefore": "720h", + "privateKey": {"algorithm": "ECDSA", "size": 256, "rotationPolicy": "Always"}, + "issuerRef": { + "name": "inference-gateway-ca", + "kind": "ClusterIssuer", + "group": "cert-manager.io", + }, + }, + } + }, + }, + } + ) + ) + + +def _ai_gateway_route(*, section_name: str, backend_refs: list[dict]) -> fnv1.Resource: + """The composed AIGatewayRoute matching ml-team/assistant, bound to the gateway's section_name listener.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "gw-eu-pc"}, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status) && has(object.status.conditions) && " + "object.status.conditions.exists(c, c.type == 'Accepted' && c.status == 'True')" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "aigateway.envoyproxy.io/v1beta1", + "kind": "AIGatewayRoute", + "metadata": {"name": "assistant", "namespace": "mp-ml-team-51733"}, + "spec": { + # The route lives in the team's namespace but + # attaches across to the gateway. + "parentRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "inference-gateway", + "namespace": "modelplane-system", + "sectionName": section_name, + } + ], + "rules": [ + { + "matches": [ + { + "headers": [ + { + "type": "Exact", + "name": "x-ai-eg-model", + "value": "ml-team/assistant", + } + ] + } + ], + "backendRefs": backend_refs, + "timeouts": {"request": "600s"}, + "streamIdleTimeout": "0s", + "modelsOwnedBy": "ml-team", + } + ], + # Declaring the token costs is what makes the + # ext-proc ask a backend for usage on a streamed + # response, which otherwise reports none, and is + # where the metered counts in the access log + # come from. + "llmRequestCosts": [ + {"metadataKey": "llm_input_token", "type": "InputToken"}, + {"metadataKey": "llm_output_token", "type": "OutputToken"}, + {"metadataKey": "llm_total_token", "type": "TotalToken"}, + ], + }, + } + }, + }, + } + ) + ) def _to_dict(msg: message.Message) -> dict: @@ -255,475 +518,1528 @@ def _to_dict(msg: message.Message) -> dict: return json.loads(json_format.MessageToJson(msg, sort_keys=True)) -# Passes where a route can't be composed compose nothing and say why. Asserting -# the whole response proves nothing is composed against a cluster the route -# can't yet reach, rather than a subset being applied. -def _gates_cases() -> list[Case]: - composed = _endpoint("self", origin="https://gw-eu.example.com", composed=True) - return [ - Case( - name="the gateway's client PKI hasn't issued, so nothing can name its certificate", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) - ), - required_resources=_required( - gateway=[_gateway(client_ca=None)], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [composed]}, - ), - ), - want=_not_ready( - {"model": _MODEL, "endpoints": {"total": 0, "ready": 0}}, - fn.CONDITION_REASON_WAITING_FOR_GATEWAY, - "InferenceGateway eu has not published its client CA", - _requirements([_entry("d")]), - ), - ), - Case( - name="no selected endpoint is ready", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) - ), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", ready=False)]}, - ), - ), - want=_not_ready( - { - "model": _MODEL, - "address": "203.0.113.1", - "endpoints": {"total": 1, "ready": 0}, - }, - fn.CONDITION_REASON_NO_ENDPOINTS, - "None of the 1 selected ModelEndpoints is ready to carry traffic", - _requirements([_entry("d")]), +# Every ModelRoute here sets timeouts other than the ModelService's defaults, so +# the AIGatewayRoute's timeouts can only have come from the ModelRoute. Every +# object composed onto the gateway's cluster lands in mp-ml-team-51733, the +# namespace mirroring the route's own. compose-inference-cluster composes that +# namespace, so it isn't among the composed resources. +# +# The cases where the route can't be composed compose nothing and say why. Their +# whole responses show that no subset of the route is applied. +# +# Secret data is base64 encoded, as the API server stores it: c2stMQ== is +# "sk-1", c2stdG9n "sk-tog" and c2stcHJvdmlkZXI= "sk-provider". +COMPOSE_CASES = [ + Case( + name="the gateway's client PKI hasn't issued, so nothing can name its certificate", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=False)] + ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources(items=[_composed_endpoint(model=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State(composite=_desired_model_route(address=None, total_endpoints=0, ready_endpoints=0)), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, message="InferenceGateway eu has not published its client CA" + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + } ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateway", + message="InferenceGateway eu has not published its client CA", + ) + ], ), - Case( - name="a composed endpoint whose cluster withdrew its CA is dropped", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")]))) + ), + Case( + name="no selected endpoint is ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] ), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu", ca=None)], - **{"endpoints-d": [composed]}, + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources( + items=[ + _third_party_endpoint( + name="self", + origin="https://gw-eu.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=False, + reason="CredentialMissing", + ) + ] ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=0) ), - want=_not_ready( - { - "model": _MODEL, - "address": "203.0.113.1", - "endpoints": {"total": 1, "ready": 0}, - }, - fn.CONDITION_REASON_NO_ENDPOINTS, - "None of the 1 selected ModelEndpoints is ready to carry traffic", - _requirements([_entry("d")]), - warning="Endpoints left out of the route, their cluster has published no gateway CA: self", - ), - ), - Case( - name="a credential Secret missing its key drops the endpoint", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("a")]))) - ), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-a": [_endpoint("wrongkey", origin="https://a.example.com", credential="k")], - "credential-wrongkey": [_secret("k", {"token": "sk-1"})], - }, + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="None of the 1 selected ModelEndpoints is ready to carry traffic", + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReadyEndpoints", + message="None of the 1 selected ModelEndpoints is ready to carry traffic", + ) + ], + ), + ), + Case( + name="a composed endpoint whose cluster has published no gateway CA is dropped", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=False)]), + "endpoints-d": fnv1.Resources(items=[_composed_endpoint(model=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=0) ), - want=_not_ready( - { - "model": _MODEL, - "address": "203.0.113.1", - "endpoints": {"total": 1, "ready": 0}, - }, - fn.CONDITION_REASON_NO_ENDPOINTS, - "None of the 1 selected ModelEndpoints is ready to carry traffic", - _requirements([_entry("a")], credentials={"wrongkey": "k"}), - warning=( - "Endpoints left out of the route, their credential Secret missing or missing its key: wrongkey" + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="Endpoints left out of the route, their cluster has published no gateway CA: self", + ), + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="None of the 1 selected ModelEndpoints is ready to carry traffic", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReadyEndpoints", + message="None of the 1 selected ModelEndpoints is ready to carry traffic", + ) + ], + ), + ), + # The endpoint's credential names the apiKey key, which its Secret lacks. + Case( + name="a credential Secret missing its key drops the endpoint", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="a", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "a"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-a": fnv1.Resources( + items=[ + _third_party_endpoint( + name="wrongkey", + origin="https://a.example.com", + model=None, + schema="OpenAI", + api_key_secret="k", + ready=True, + reason="EndpointUsable", + ) + ] + ), + "credential-wrongkey": fnv1.Resources(items=[_api_key_secret(name="k", data={"token": "c2stMQ=="})]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=0) ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="Endpoints left out of the route, their credential Secret missing or missing its key: wrongkey", + ), + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="None of the 1 selected ModelEndpoints is ready to carry traffic", + ), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-a": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "a"}), + ), + "credential-wrongkey": fnv1.ResourceSelector( + api_version="v1", kind="Secret", namespace="ml-team", match_name="k" + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoReadyEndpoints", + message="None of the 1 selected ModelEndpoints is ready to carry traffic", + ) + ], ), - ] - - -@pytest.mark.parametrize("case", _gates_cases(), ids=lambda case: case.name) -def test_gates(case: Case) -> None: - """A pass where a route can't be composed composes nothing and says why.""" - got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) - - -def test_compose() -> None: - """A composed endpoint and a third-party provider get backends, a credential, a cluster CA and a route.""" - # A composed self-hosted endpoint at priority 0 and a third-party provider - # at priority 1. - entries = [_entry("d", priority=0), _entry("together", priority=1)] - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-d": [ - _endpoint("self", origin="https://gw-eu.example.com", model="d", composed=True), - ], - "endpoints-together": [ - _endpoint( - "together", - origin="https://api.together.xyz", - model="Qwen/Qwen2.5", - credential="together-key", - ), - ], - "credential-together": [_secret("together-key", {"apiKey": "sk-tog"})], + ), + # A composed self-hosted endpoint at priority 0 and a third-party provider at + # priority 1. + Case( + name="a composed endpoint and a third-party provider get backends, a credential, a cluster CA and a route", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + priority=0, + ), + v1alpha1.Endpoint( + name="together", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "together"}), + priority=1, + ), + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources(items=[_composed_endpoint(model="d")]), + "endpoints-together": fnv1.Resources( + items=[ + _third_party_endpoint( + name="together", + origin="https://api.together.xyz", + model="Qwen/Qwen2.5", + schema="OpenAI", + api_key_secret="together-key", + ready=True, + reason="EndpointUsable", + ) + ] + ), + "credential-together": fnv1.Resources( + items=[_api_key_secret(name="together-key", data={"apiKey": "c2stdG9n"})] + ), }, ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - # The exact set, so an unexpected extra object fails the test. The - # mirrored namespace isn't here: compose-inference-cluster composes it. - assert set(got.desired.resources) == { - "client-certificate", - "backend-self", - "aibackend-self", - "backend-together", - "aibackend-together", - "credential-together", - "credpolicy-together", - "cluster-ca-gw-eu", - "route", - } - - route = _manifest(got, "route") - # The timeouts are the ModelRoute's, which _route_xr sets to values - # other than the ModelService's defaults. - assert route["spec"]["rules"] == [ - { - "matches": [{"headers": [{"type": "Exact", "name": "x-ai-eg-model", "value": _MODEL}]}], - "backendRefs": [ - {"name": _be("self"), "weight": 1, "priority": 0, "modelNameOverride": "d"}, - { - "name": _be("together"), - "weight": 1, - "priority": 1, - "modelNameOverride": "Qwen/Qwen2.5", + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=2, ready_endpoints=2), + resources={ + "backend-self": _composed_endpoint_backend(), + "aibackend-self": _composed_endpoint_ai_backend(), + "backend-together": _third_party_backend( + name="assistant-eu-together-20044", hostname="api.together.xyz" + ), + "aibackend-together": _third_party_ai_backend(name="assistant-eu-together-20044", schema="OpenAI"), + "credential-together": _credential( + name="assistant-eu-together-credential-fe51d", api_key="c2stdG9n" + ), + "credpolicy-together": _credential_policy( + name="assistant-eu-together-20044", + auth={ + "type": "APIKey", + "apiKey": {"secretRef": {"name": "assistant-eu-together-credential-fe51d"}}, + }, + ), + "cluster-ca-gw-eu": _cluster_ca(), + "client-certificate": _client_certificate(), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[ + {"name": "assistant-eu-self-a42e6", "weight": 1, "priority": 0, "modelNameOverride": "d"}, + { + "name": "assistant-eu-together-20044", + "weight": 1, + "priority": 1, + "modelNameOverride": "Qwen/Qwen2.5", + }, + ], + ), }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") ], - "timeouts": {"request": "600s"}, - "streamIdleTimeout": "0s", - "modelsOwnedBy": _NS, - } - ] - # Declaring the token costs is what makes the ext-proc ask a backend for - # usage on a streamed response, which otherwise reports none, and is - # where the metered counts in the access log come from. - assert route["spec"]["llmRequestCosts"] == [ - {"metadataKey": "llm_input_token", "type": "InputToken"}, - {"metadataKey": "llm_output_token", "type": "OutputToken"}, - {"metadataKey": "llm_total_token", "type": "TotalToken"}, - ] - - # The composed backend pins its cluster's CA and presents the client - # certificate, both this route's own; the third-party one uses the - # system trust store. - assert _manifest(got, "backend-self")["spec"]["tls"] == { - "caCertificateRefs": [{"kind": "ConfigMap", "group": "", "name": "assistant-eu-gw-eu-ca-3dd16"}], - "sni": "gw-eu.example.com", - "clientCertificateRef": {"kind": "Secret", "group": "", "name": "assistant-eu-client-08324"}, - } - assert _manifest(got, "backend-together")["spec"]["tls"] == { - "wellKnownCACertificates": "System", - "sni": "api.together.xyz", - } - - # The caller header is stripped only for the backend we don't operate. - assert "headerMutation" not in _manifest(got, "aibackend-self")["spec"] - assert _manifest(got, "aibackend-together")["spec"]["headerMutation"] == {"remove": ["x-modelplane-caller"]} - - # The credential is republished under the fixed apiKey key. - assert _manifest(got, "credential-together")["data"] == {"apiKey": base64.b64encode(b"sk-tog").decode()} - # Named for this route, so no other route in the namespace composes it. - assert _manifest(got, "cluster-ca-gw-eu") == { - "apiVersion": "v1", - "kind": "ConfigMap", - "metadata": {"name": "assistant-eu-gw-eu-ca-3dd16", "namespace": "mp-ml-team-51733"}, - "data": {"ca.crt": _CLUSTER_CA}, - } - - # Every composed object lands in the namespace mirroring the route's own, - # which compose-inference-cluster composes. - for key in ("backend-self", "backend-together", "credential-together", "cluster-ca-gw-eu", "route"): - assert _manifest(got, key)["metadata"]["namespace"] == "mp-ml-team-51733", key - - # The route lives in the team namespace but attaches across to the gateway. - assert route["spec"]["parentRefs"][0]["namespace"] == "modelplane-system" - - # The client certificate the backends present, issued from the gateway's - # CA ClusterIssuer into this namespace. Named for this route, so no other - # route in the namespace composes it, and deleted with the route. - assert ( - "managementPolicies" - not in resource.struct_to_dict(got.desired.resources["client-certificate"].resource)["spec"] - ) - assert _manifest(got, "client-certificate") == { - "apiVersion": "cert-manager.io/v1", - "kind": "Certificate", - "metadata": {"name": "assistant-eu-client-08324", "namespace": "mp-ml-team-51733"}, - "spec": { - "secretName": "assistant-eu-client-08324", - "commonName": "inference-gateway-eu", - "usages": ["client auth", "digital signature", "key encipherment"], - "duration": "2160h", - "renewBefore": "720h", - "privateKey": {"algorithm": "ECDSA", "size": 256, "rotationPolicy": "Always"}, - "issuerRef": {"name": "inference-gateway-ca", "kind": "ClusterIssuer", "group": "cert-manager.io"}, - }, - } - - -@pytest.mark.parametrize(("tls", "want"), [(False, "http"), (True, "https")]) -def test_route_binds_to_the_listener_matching_the_gateways_tls(tls: bool, want: str) -> None: - """The route binds to the HTTPS listener on a TLS gateway, and to the HTTP listener otherwise.""" + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + "endpoints-together": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "together"}), + ), + "credential-together": fnv1.ResourceSelector( + api_version="v1", kind="Secret", namespace="ml-team", match_name="together-key" + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + # Without TLS there's only the HTTP listener. + Case( + name="the route binds to the HTTP listener of a gateway without TLS", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources(items=[_composed_endpoint(model=None)]), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=1), + resources={ + "backend-self": _composed_endpoint_backend(), + "aibackend-self": _composed_endpoint_ai_backend(), + "cluster-ca-gw-eu": _cluster_ca(), + "client-certificate": _client_certificate(), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[{"name": "assistant-eu-self-a42e6", "weight": 1, "priority": 0}], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), # A TLS gateway serves inference on its HTTPS listener alone, so the route - # binds there; without TLS there's only the HTTP listener. Binding to :80 on - # a TLS gateway would carry credentials in the clear. - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")])))), - required_resources=_required( - gateway=[_gateway(tls=tls)], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", composed=True)]}, + # binds there. Binding to :80 on a TLS gateway would carry credentials in the + # clear. + Case( + name="the route binds to the HTTPS listener of a TLS gateway", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=True, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources(items=[_composed_endpoint(model=None)]), + }, ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert _manifest(got, "route")["spec"]["parentRefs"][0]["sectionName"] == want - - -def test_status_reports_address_and_counts() -> None: - """The status reports the model name, the gateway's address, and the endpoint counts.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")])))), - required_resources=_required( - gateway=[_gateway(address="203.0.113.9")], - clusters=[_cluster("gw-eu")], - **{"endpoints-d": [_endpoint("self", origin="https://gw-eu.example.com", composed=True)]}, + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=1), + resources={ + "backend-self": _composed_endpoint_backend(), + "aibackend-self": _composed_endpoint_ai_backend(), + "cluster-ca-gw-eu": _cluster_ca(), + "client-certificate": _client_certificate(), + "route": _ai_gateway_route( + section_name="https", + backend_refs=[{"name": "assistant-eu-self-a42e6", "weight": 1, "priority": 0}], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert resource.struct_to_dict(got.desired.composite.resource)["status"] == { - "model": _MODEL, - "address": "203.0.113.9", - "endpoints": {"total": 1, "ready": 1}, - } - - -def test_an_endpoint_matched_twice_belongs_to_the_first_entry() -> None: - """An endpoint two entries match belongs to the first of them.""" - # A canary entry and a catch-all entry must not both weight one endpoint; - # the first that matches it wins. - entries = [_entry("kimi", name="canary", priority=0), _entry("kimi", name="catchall", priority=1)] - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-canary": [_endpoint("kimi-a", origin="https://a.example.com")], - "endpoints-catchall": [_endpoint("kimi-a", origin="https://a.example.com")], + ), + Case( + name="the status reports the gateway's address and the endpoint counts", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.9", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources(items=[_composed_endpoint(model=None)]), }, ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - # Neither endpoint is Modelplane-composed, so no client certificate is - # issued. - assert "client-certificate" not in got.desired.resources - refs = _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] - assert refs == [{"name": _be("kimi-a"), "weight": 1, "priority": 0}] - assert resource.struct_to_dict(got.desired.composite.resource)["status"]["endpoints"] == {"total": 1, "ready": 1} - - -def test_priorities_are_renumbered_without_gaps() -> None: - """The priorities of the tiers that have a ready endpoint are renumbered from 0, without gaps.""" + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.9", total_endpoints=1, ready_endpoints=1), + resources={ + "backend-self": _composed_endpoint_backend(), + "aibackend-self": _composed_endpoint_ai_backend(), + "cluster-ca-gw-eu": _cluster_ca(), + "client-certificate": _client_certificate(), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[{"name": "assistant-eu-self-a42e6", "weight": 1, "priority": 0}], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + # A canary entry and a catch-all entry must not both weight one endpoint; the + # first that matches it wins. The endpoint isn't Modelplane-composed, so no + # client certificate is issued. + Case( + name="an endpoint matched twice belongs to the first entry", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="canary", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "kimi"}), + priority=0, + ), + v1alpha1.Endpoint( + name="catchall", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "kimi"}), + priority=1, + ), + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-canary": fnv1.Resources( + items=[ + _third_party_endpoint( + name="kimi-a", + origin="https://a.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ) + ] + ), + "endpoints-catchall": fnv1.Resources( + items=[ + _third_party_endpoint( + name="kimi-a", + origin="https://a.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ) + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=1), + resources={ + "backend-kimi-a": _third_party_backend(name="assistant-eu-kimi-a-bf6de", hostname="a.example.com"), + "aibackend-kimi-a": _third_party_ai_backend(name="assistant-eu-kimi-a-bf6de", schema="OpenAI"), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[{"name": "assistant-eu-kimi-a-bf6de", "weight": 1, "priority": 0}], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-canary": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "kimi"}), + ), + "endpoints-catchall": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "kimi"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), # A ModelService's priorities are an ordering; Envoy's are levels it walks - # from 0. A user writing 0 and 5, or a tier gone unready during a roll, - # would otherwise leave gaps in what Envoy gets. - entries = [_entry("a", priority=0), _entry("b", priority=5), _entry("c", priority=9)] - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(entries)))), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - # The middle tier has no ready endpoint, so it drops out and - # must not leave a hole behind it. - "endpoints-a": [_endpoint("a-0", origin="https://a.example.com")], - "endpoints-b": [_endpoint("b-0", origin="https://b.example.com", ready=False)], - "endpoints-c": [_endpoint("c-0", origin="https://c.example.com")], + # from 0. A user writing 0 and 5, or a tier gone unready during a roll, would + # otherwise leave gaps in what Envoy gets. The middle tier here has no ready + # endpoint, so it drops out and must not leave a hole behind it: two tiers + # survive, renumbered 0 and 1. + Case( + name="priorities are renumbered without gaps", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="a", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "a"}), + priority=0, + ), + v1alpha1.Endpoint( + name="b", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "b"}), + priority=5, + ), + v1alpha1.Endpoint( + name="c", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "c"}), + priority=9, + ), + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-a": fnv1.Resources( + items=[ + _third_party_endpoint( + name="a-0", + origin="https://a.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ) + ] + ), + "endpoints-b": fnv1.Resources( + items=[ + _third_party_endpoint( + name="b-0", + origin="https://b.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=False, + reason="CredentialMissing", + ) + ] + ), + "endpoints-c": fnv1.Resources( + items=[ + _third_party_endpoint( + name="c-0", + origin="https://c.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ) + ] + ), }, ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - refs = _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] - assert [r["priority"] for r in refs] == [0, 1], "two tiers survive, renumbered 0 and 1" - - -@dataclasses.dataclass -class WeightCase: - """A weight-distribution case: the entries and the endpoints each matched, - and the whole backendRefs list the route should carry.""" - - name: str - entries: list[v1alpha1.Endpoint] - endpoints: dict[str, list[dict]] - want_refs: list[dict] - - -def _weight_distribution_cases() -> list[WeightCase]: - def _origins(*names_: str) -> list[dict]: - return [_endpoint(n, origin=f"https://{n}.example.com") for n in names_] - - def _ref(ep: str, weight: int, priority: int = 0) -> dict: - return {"name": _be(ep), "weight": weight, "priority": priority} - - return [ - WeightCase( - # An entry's weight is written once but applied per backend, so it - # spreads over the endpoints it matched while the ratio between - # entries survives: 90 over three is 30 each, 10 over one is 10, - # reduced by the gcd to the smallest equivalent integers. - name="a weight spreads across a tier's endpoints, ratio preserved", - entries=[_entry("big", weight=90), _entry("small", weight=10)], - endpoints={ - "endpoints-big": _origins("big-0", "big-1", "big-2"), - "endpoints-small": _origins("small-0"), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=3, ready_endpoints=2), + resources={ + "backend-a-0": _third_party_backend(name="assistant-eu-a-0-b66b8", hostname="a.example.com"), + "aibackend-a-0": _third_party_ai_backend(name="assistant-eu-a-0-b66b8", schema="OpenAI"), + "backend-c-0": _third_party_backend(name="assistant-eu-c-0-782db", hostname="c.example.com"), + "aibackend-c-0": _third_party_ai_backend(name="assistant-eu-c-0-782db", schema="OpenAI"), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[ + {"name": "assistant-eu-a-0-b66b8", "weight": 1, "priority": 0}, + {"name": "assistant-eu-c-0-782db", "weight": 1, "priority": 1}, + ], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-a": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "a"}), + ), + "endpoints-b": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "b"}), + ), + "endpoints-c": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "c"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + # An entry's weight is written once but applied per backend, so it spreads + # over the endpoints it matched while the ratio between entries survives: 90 + # over three is 30 each, 10 over one is 10, reduced by the gcd to the + # smallest equivalent integers. + Case( + name="a weight spreads across a tier's endpoints, ratio preserved", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="big", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "big"}), + weight=90, + ), + v1alpha1.Endpoint( + name="small", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "small"}), + weight=10, + ), + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-big": fnv1.Resources( + items=[ + _third_party_endpoint( + name="big-0", + origin="https://big-0.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ), + _third_party_endpoint( + name="big-1", + origin="https://big-1.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ), + _third_party_endpoint( + name="big-2", + origin="https://big-2.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ), + ] + ), + "endpoints-small": fnv1.Resources( + items=[ + _third_party_endpoint( + name="small-0", + origin="https://small-0.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ) + ] + ), }, - want_refs=[ - _ref("big-0", 3), - _ref("big-1", 3), - _ref("big-2", 3), - _ref("small-0", 1), - ], - ), - WeightCase( - # Weight 1 over five endpoints must floor none of them to 0, which - # would drop them from the load assignment rather than share. - name="a weight below its endpoint count floors no endpoint", - entries=[_entry("many", weight=1)], - endpoints={"endpoints-many": _origins("many-0", "many-1", "many-2", "many-3", "many-4")}, - want_refs=[_ref(f"many-{i}", 1) for i in range(5)], - ), - WeightCase( - # A max-weight entry beside a tiny one spread over two endpoints - # scales past the per-backendRef limit even though every weight is - # in bounds, so it rescales to the limit rather than composing a - # route the API server rejects. - name="an extreme but valid ratio is clamped to the limit", - entries=[_entry("big", weight=1000000, priority=0), _entry("small", weight=1, priority=0)], - endpoints={"endpoints-big": _origins("big-0"), "endpoints-small": _origins("small-0", "small-1")}, - want_refs=[_ref("big-0", 1000000), _ref("small-0", 1), _ref("small-1", 1)], - ), - WeightCase( - # The remainder is handed to the first endpoints of a tier, so the - # order must be the endpoints' names rather than the API server's - # unspecified list order, or the composed weights churn. - name="endpoints are ordered by name for a stable split", - entries=[_entry("d")], - endpoints={"endpoints-d": _origins("z", "a", "m")}, - want_refs=[_ref("a", 1), _ref("m", 1), _ref("z", 1)], - ), - ] - - -@pytest.mark.parametrize("case", _weight_distribution_cases(), ids=lambda case: case.name) -def test_weight_distribution(case: WeightCase) -> None: - """Each entry's weight is distributed across the endpoints it matched.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr(case.entries)))), - required_resources=_required(gateway=[_gateway()], clusters=[_cluster("gw-eu")], **case.endpoints), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert _manifest(got, "route")["spec"]["rules"][0]["backendRefs"] == case.want_refs - - -@dataclasses.dataclass -class CredentialCase: - """A credential case: the API a keyed backend speaks, and the whole - BackendSecurityPolicy composed for it.""" - - name: str - api: mev1alpha1.Api | None - want: dict - - -def _credential_policy_cases() -> list[CredentialCase]: - secret = resource.child_name(f"{_SVC}-{_GW}", "provider", "credential") - target = {"group": "aigateway.envoyproxy.io", "kind": "AIServiceBackend", "name": _be("provider")} - return [ - CredentialCase( - name="a backend speaking OpenAI's API gets the key as a bearer token", - api=None, - want={ - "apiVersion": "aigateway.envoyproxy.io/v1beta1", - "kind": "BackendSecurityPolicy", - "metadata": {"name": _be("provider"), "namespace": "mp-ml-team-51733"}, - "spec": { - "type": "APIKey", - "apiKey": {"secretRef": {"name": secret}}, - "targetRefs": [target], + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=4, ready_endpoints=4), + resources={ + "backend-big-0": _third_party_backend( + name="assistant-eu-big-0-1e944", hostname="big-0.example.com" + ), + "aibackend-big-0": _third_party_ai_backend(name="assistant-eu-big-0-1e944", schema="OpenAI"), + "backend-big-1": _third_party_backend( + name="assistant-eu-big-1-91055", hostname="big-1.example.com" + ), + "aibackend-big-1": _third_party_ai_backend(name="assistant-eu-big-1-91055", schema="OpenAI"), + "backend-big-2": _third_party_backend( + name="assistant-eu-big-2-5c954", hostname="big-2.example.com" + ), + "aibackend-big-2": _third_party_ai_backend(name="assistant-eu-big-2-5c954", schema="OpenAI"), + "backend-small-0": _third_party_backend( + name="assistant-eu-small-0-60d20", hostname="small-0.example.com" + ), + "aibackend-small-0": _third_party_ai_backend(name="assistant-eu-small-0-60d20", schema="OpenAI"), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[ + {"name": "assistant-eu-big-0-1e944", "weight": 3, "priority": 0}, + {"name": "assistant-eu-big-1-91055", "weight": 3, "priority": 0}, + {"name": "assistant-eu-big-2-5c954", "weight": 3, "priority": 0}, + {"name": "assistant-eu-small-0-60d20", "weight": 1, "priority": 0}, + ], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-big": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "big"}), + ), + "endpoints-small": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "small"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + # Weight 1 over five endpoints must floor none of them to 0, which would drop + # them from the load assignment rather than share. + Case( + name="a weight below its endpoint count floors no endpoint", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="many", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "many"}), + weight=1, + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-many": fnv1.Resources( + items=[ + _third_party_endpoint( + name="many-0", + origin="https://many-0.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ), + _third_party_endpoint( + name="many-1", + origin="https://many-1.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ), + _third_party_endpoint( + name="many-2", + origin="https://many-2.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ), + _third_party_endpoint( + name="many-3", + origin="https://many-3.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ), + _third_party_endpoint( + name="many-4", + origin="https://many-4.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ), + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=5, ready_endpoints=5), + resources={ + "backend-many-0": _third_party_backend( + name="assistant-eu-many-0-e4e60", hostname="many-0.example.com" + ), + "aibackend-many-0": _third_party_ai_backend(name="assistant-eu-many-0-e4e60", schema="OpenAI"), + "backend-many-1": _third_party_backend( + name="assistant-eu-many-1-b8a15", hostname="many-1.example.com" + ), + "aibackend-many-1": _third_party_ai_backend(name="assistant-eu-many-1-b8a15", schema="OpenAI"), + "backend-many-2": _third_party_backend( + name="assistant-eu-many-2-a7a3b", hostname="many-2.example.com" + ), + "aibackend-many-2": _third_party_ai_backend(name="assistant-eu-many-2-a7a3b", schema="OpenAI"), + "backend-many-3": _third_party_backend( + name="assistant-eu-many-3-db5a5", hostname="many-3.example.com" + ), + "aibackend-many-3": _third_party_ai_backend(name="assistant-eu-many-3-db5a5", schema="OpenAI"), + "backend-many-4": _third_party_backend( + name="assistant-eu-many-4-bd6f9", hostname="many-4.example.com" + ), + "aibackend-many-4": _third_party_ai_backend(name="assistant-eu-many-4-bd6f9", schema="OpenAI"), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[ + {"name": "assistant-eu-many-0-e4e60", "weight": 1, "priority": 0}, + {"name": "assistant-eu-many-1-b8a15", "weight": 1, "priority": 0}, + {"name": "assistant-eu-many-2-a7a3b", "weight": 1, "priority": 0}, + {"name": "assistant-eu-many-3-db5a5", "weight": 1, "priority": 0}, + {"name": "assistant-eu-many-4-bd6f9", "weight": 1, "priority": 0}, + ], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-many": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "many"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + # A max-weight entry beside a tiny one spread over two endpoints scales past + # the per-backendRef limit even though every weight is in bounds, so it + # rescales to the limit rather than composing a route the API server rejects. + Case( + name="an extreme but valid ratio is clamped to the limit", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="big", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "big"}), + priority=0, + weight=1000000, + ), + v1alpha1.Endpoint( + name="small", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "small"}), + priority=0, + weight=1, + ), + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-big": fnv1.Resources( + items=[ + _third_party_endpoint( + name="big-0", + origin="https://big-0.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ) + ] + ), + "endpoints-small": fnv1.Resources( + items=[ + _third_party_endpoint( + name="small-0", + origin="https://small-0.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ), + _third_party_endpoint( + name="small-1", + origin="https://small-1.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ), + ] + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=3, ready_endpoints=3), + resources={ + "backend-big-0": _third_party_backend( + name="assistant-eu-big-0-1e944", hostname="big-0.example.com" + ), + "aibackend-big-0": _third_party_ai_backend(name="assistant-eu-big-0-1e944", schema="OpenAI"), + "backend-small-0": _third_party_backend( + name="assistant-eu-small-0-60d20", hostname="small-0.example.com" + ), + "aibackend-small-0": _third_party_ai_backend(name="assistant-eu-small-0-60d20", schema="OpenAI"), + "backend-small-1": _third_party_backend( + name="assistant-eu-small-1-cad94", hostname="small-1.example.com" + ), + "aibackend-small-1": _third_party_ai_backend(name="assistant-eu-small-1-cad94", schema="OpenAI"), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[ + {"name": "assistant-eu-big-0-1e944", "weight": 1000000, "priority": 0}, + {"name": "assistant-eu-small-0-60d20", "weight": 1, "priority": 0}, + {"name": "assistant-eu-small-1-cad94", "weight": 1, "priority": 0}, + ], + ), }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-big": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "big"}), + ), + "endpoints-small": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "small"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + # The remainder is handed to the first endpoints of a tier, so the order must + # be the endpoints' names rather than the API server's unspecified list order, + # or the composed weights churn. + Case( + name="endpoints are ordered by name for a stable split", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources( + items=[ + _third_party_endpoint( + name="z", + origin="https://z.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ), + _third_party_endpoint( + name="a", + origin="https://a.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ), + _third_party_endpoint( + name="m", + origin="https://m.example.com", + model=None, + schema="OpenAI", + api_key_secret=None, + ready=True, + reason="EndpointUsable", + ), + ] + ), }, ), - CredentialCase( - name="a backend speaking Anthropic's API gets the key in x-api-key", - api=mev1alpha1.Api(schema="Anthropic"), - want={ - "apiVersion": "aigateway.envoyproxy.io/v1beta1", - "kind": "BackendSecurityPolicy", - "metadata": {"name": _be("provider"), "namespace": "mp-ml-team-51733"}, - "spec": { - "type": "AnthropicAPIKey", - "anthropicAPIKey": {"secretRef": {"name": secret}}, - "targetRefs": [target], + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=3, ready_endpoints=3), + resources={ + "backend-a": _third_party_backend(name="assistant-eu-a-18a56", hostname="a.example.com"), + "aibackend-a": _third_party_ai_backend(name="assistant-eu-a-18a56", schema="OpenAI"), + "backend-m": _third_party_backend(name="assistant-eu-m-4d1e4", hostname="m.example.com"), + "aibackend-m": _third_party_ai_backend(name="assistant-eu-m-4d1e4", schema="OpenAI"), + "backend-z": _third_party_backend(name="assistant-eu-z-e5d6f", hostname="z.example.com"), + "aibackend-z": _third_party_ai_backend(name="assistant-eu-z-e5d6f", schema="OpenAI"), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[ + {"name": "assistant-eu-a-18a56", "weight": 1, "priority": 0}, + {"name": "assistant-eu-m-4d1e4", "weight": 1, "priority": 0}, + {"name": "assistant-eu-z-e5d6f", "weight": 1, "priority": 0}, + ], + ), }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + Case( + name="a backend speaking OpenAI's API gets the key as a bearer token", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources( + items=[ + _third_party_endpoint( + name="provider", + origin="https://api.example.com", + model=None, + schema="OpenAI", + api_key_secret="provider-key", + ready=True, + reason="EndpointUsable", + ) + ] + ), + "credential-provider": fnv1.Resources( + items=[_api_key_secret(name="provider-key", data={"apiKey": "c2stcHJvdmlkZXI="})] + ), }, ), - ] - - -@pytest.mark.parametrize("case", _credential_policy_cases(), ids=lambda case: case.name) -def test_credential_policy(case: CredentialCase) -> None: - """A keyed backend's BackendSecurityPolicy sends the key the way its API expects.""" - req = fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_route_xr([_entry("d")])))), - required_resources=_required( - gateway=[_gateway()], - clusters=[_cluster("gw-eu")], - **{ - "endpoints-d": [ - _endpoint( - "provider", - origin="https://api.example.com", - api=case.api, - credential="provider-key", - ) - ], - "credential-provider": [_secret("provider-key", {"apiKey": "sk-provider"})], + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=1), + resources={ + "backend-provider": _third_party_backend( + name="assistant-eu-provider-348cb", hostname="api.example.com" + ), + "aibackend-provider": _third_party_ai_backend(name="assistant-eu-provider-348cb", schema="OpenAI"), + "credential-provider": _credential( + name="assistant-eu-provider-credential-4d66b", api_key="c2stcHJvdmlkZXI=" + ), + "credpolicy-provider": _credential_policy( + name="assistant-eu-provider-348cb", + auth={ + "type": "APIKey", + "apiKey": {"secretRef": {"name": "assistant-eu-provider-credential-4d66b"}}, + }, + ), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[{"name": "assistant-eu-provider-348cb", "weight": 1, "priority": 0}], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + "credential-provider": fnv1.ResourceSelector( + api_version="v1", kind="Secret", namespace="ml-team", match_name="provider-key" + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), + Case( + name="a backend speaking Anthropic's API gets the key in x-api-key", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_route( + endpoints=[ + v1alpha1.Endpoint( + name="d", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "d"}), + ) + ] + ) + ), + required_resources={ + "gateway": fnv1.Resources( + items=[_inference_gateway(tls=False, address="203.0.113.1", client_ca_published=True)] + ), + "clusters": fnv1.Resources(items=[_inference_cluster(gateway_ca_published=True)]), + "endpoints-d": fnv1.Resources( + items=[ + _third_party_endpoint( + name="provider", + origin="https://api.example.com", + model=None, + schema="Anthropic", + api_key_secret="provider-key", + ready=True, + reason="EndpointUsable", + ) + ] + ), + "credential-provider": fnv1.Resources( + items=[_api_key_secret(name="provider-key", data={"apiKey": "c2stcHJvdmlkZXI="})] + ), }, ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert _manifest(got, "credpolicy-provider") == case.want + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_route(address="203.0.113.1", total_endpoints=1, ready_endpoints=1), + resources={ + "backend-provider": _third_party_backend( + name="assistant-eu-provider-348cb", hostname="api.example.com" + ), + "aibackend-provider": _third_party_ai_backend( + name="assistant-eu-provider-348cb", schema="Anthropic" + ), + "credential-provider": _credential( + name="assistant-eu-provider-credential-4d66b", api_key="c2stcHJvdmlkZXI=" + ), + "credpolicy-provider": _credential_policy( + name="assistant-eu-provider-348cb", + auth={ + "type": "AnthropicAPIKey", + "anthropicAPIKey": {"secretRef": {"name": "assistant-eu-provider-credential-4d66b"}}, + }, + ), + "route": _ai_gateway_route( + section_name="http", + backend_refs=[{"name": "assistant-eu-provider-348cb", "weight": 1, "priority": 0}], + ), + }, + ), + results=[ + fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the route on gateway eu to be accepted") + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateway": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", kind="InferenceGateway", match_name="eu" + ), + "clusters": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceCluster"), + "endpoints-d": fnv1.ResourceSelector( + api_version="modelplane.ai/v1alpha1", + kind="ModelEndpoint", + namespace="ml-team", + match_labels=fnv1.MatchLabels(labels={"modelplane.ai/deployment": "d"}), + ), + "credential-provider": fnv1.ResourceSelector( + api_version="v1", kind="Secret", namespace="ml-team", match_name="provider-key" + ), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoute", + message="Waiting for the route on gateway eu to be accepted", + ) + ], + ), + ), +] + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: Case) -> None: + """RunFunction composes a ModelRoute's backends and AIGatewayRoute, or composes nothing and says why.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) diff --git a/functions/compose-model-service/tests/test_fn.py b/functions/compose-model-service/tests/test_fn.py index 4fb2975b9..794dfb0c1 100644 --- a/functions/compose-model-service/tests/test_fn.py +++ b/functions/compose-model-service/tests/test_fn.py @@ -25,13 +25,8 @@ from google.protobuf import duration_pb2 as durationpb from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb -from models.ai.modelplane.inferencegateway import v1alpha1 as igv1alpha1 -from models.ai.modelplane.modelroute import v1alpha1 as mrtv1alpha1 from models.ai.modelplane.modelservice import v1alpha1 - -_NS = "ml-team" -_SVC = "assistant" -_MODEL = f"{_NS}/{_SVC}" +from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 @dataclasses.dataclass @@ -43,375 +38,426 @@ class Case: want: fnv1.RunFunctionResponse -def _entry(deployment: str, *, priority: int | None = None, weight: int | None = None) -> v1alpha1.Endpoint: - kwargs = {} - if priority is not None: - kwargs["priority"] = priority - if weight is not None: - kwargs["weight"] = weight - return v1alpha1.Endpoint( - name=deployment, - selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": deployment}), - **kwargs, - ) - - -def _service(entries: list[v1alpha1.Endpoint], labels: dict[str, str] | None = None) -> dict: - xr = v1alpha1.ModelService( - apiVersion="modelplane.ai/v1alpha1", - kind="ModelService", - metadata={"name": _SVC, "namespace": _NS, **({"labels": labels} if labels else {})}, - # Not the defaults, so a route carrying the defaults fails to match. - spec=v1alpha1.Spec(endpoints=entries, timeouts=v1alpha1.Timeouts(request="600s", idle="0s")), +def _model_service(*, labels: dict[str, str] | None) -> fnv1.Resource: + """The ModelService XR assistant in ml-team, serving kimi-k2.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ModelService( + apiVersion="modelplane.ai/v1alpha1", + kind="ModelService", + metadata=metav1.ObjectMeta(name="assistant", namespace="ml-team", labels=labels), + spec=v1alpha1.Spec( + endpoints=[ + v1alpha1.Endpoint( + name="kimi-k2", + selector=v1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "kimi-k2"}), + ) + ], + timeouts=v1alpha1.Timeouts(request="600s", idle="0s"), + ), + ).model_dump(exclude_none=True, mode="json", by_alias=True) + ) ) - return xr.model_dump(exclude_none=True, mode="json", by_alias=True) -def _gateway(name: str, cluster: str, *, selector: dict[str, str] | None = None, address: str | None = None) -> dict: - gw = igv1alpha1.InferenceGateway( - apiVersion="modelplane.ai/v1alpha1", - kind="InferenceGateway", - metadata={"name": name}, - spec=igv1alpha1.Spec( - clusterName=cluster, - **({"serviceSelector": igv1alpha1.ServiceSelector(matchLabels=selector)} if selector else {}), +def _desired_model_service(*, total_routes: int, ready_routes: int, ready: fnv1.Ready) -> fnv1.Resource: + """The desired ModelService XR, reporting its model and how many of its routes are ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"model": "ml-team/assistant", "routes": {"total": total_routes, "ready": ready_routes}}} ), + ready=ready, ) - d = gw.model_dump(exclude_none=True, mode="json", by_alias=True) - if address: - d["status"] = {"address": address} - return d -def _route(gateway: str, cluster: str) -> dict: - """The ModelRoute the function composes for one gateway, as a plain dict. +def _inference_gateway( + *, name: str, cluster_name: str, service_selector: dict | None, address: str | None +) -> fnv1.Resource: + """An InferenceGateway, as the gateways requirement returns it.""" + spec: dict = {"clusterName": cluster_name} + if service_selector is not None: + spec["serviceSelector"] = service_selector + gateway: dict = { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "InferenceGateway", + "metadata": {"name": name}, + "spec": spec, + } + if address is not None: + gateway["status"] = {"address": address} + return fnv1.Resource(resource=resource.dict_to_struct(gateway)) - The endpoints are spelled out rather than derived from the service's, so a - bug in the copy the function does can't hide in an expectation computed the - same way. Every case here drives the single kimi-k2 entry, which the copy - fills to its priority/weight defaults. The cluster label carries the gateway's - clusterName, which compose-inference-cluster selects routes by. - """ - route = mrtv1alpha1.ModelRoute( - apiVersion="modelplane.ai/v1alpha1", - kind="ModelRoute", - metadata={ - "name": resource.child_name(_SVC, gateway), - "namespace": _NS, - "labels": { - "modelplane.ai/service": _SVC, - "modelplane.ai/gateway": gateway, - "modelplane.ai/cluster": cluster, - }, - }, - spec=mrtv1alpha1.Spec( - gatewayName=gateway, - serviceName=_SVC, - endpoints=[ - mrtv1alpha1.Endpoint( - name="kimi-k2", - selector=mrtv1alpha1.Selector(matchLabels={"modelplane.ai/deployment": "kimi-k2"}), - priority=0, - weight=1, - ) - ], - timeouts=mrtv1alpha1.Timeouts(request="600s", idle="0s"), + +def _model_route(*, name: str, gateway: str, cluster: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed ModelRoute pinning assistant to gateway on cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelRoute", + "metadata": { + "name": name, + "namespace": "ml-team", + "labels": { + "modelplane.ai/service": "assistant", + "modelplane.ai/gateway": gateway, + "modelplane.ai/cluster": cluster, + }, + }, + "spec": { + "gatewayName": gateway, + "serviceName": "assistant", + "endpoints": [ + { + "name": "kimi-k2", + "selector": {"matchLabels": {"modelplane.ai/deployment": "kimi-k2"}}, + "priority": 0, + "weight": 1, + } + ], + "timeouts": {"request": "600s", "idle": "0s"}, + }, + } ), + ready=ready, ) - return route.model_dump(exclude_none=True, mode="json", by_alias=True) - - -def _observed_route(gateway: str, ready: bool) -> fnv1.Resource: - """A composed ModelRoute as observed back, Ready or not.""" - d = { - "apiVersion": "modelplane.ai/v1alpha1", - "kind": "ModelRoute", - "metadata": {"name": resource.child_name(_SVC, gateway), "namespace": _NS}, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True" if ready else "False", - "reason": "Available" if ready else "Creating", - "lastTransitionTime": "2026-06-08T00:00:00Z", - } - ] - }, - } - return fnv1.Resource(resource=resource.dict_to_struct(d)) -def _required(**resources) -> dict: # noqa: ANN003 - return { - name: fnv1.Resources(items=[fnv1.Resource(resource=resource.dict_to_struct(r)) for r in items]) - for name, items in resources.items() - } +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) -def _compose_cases() -> list[Case]: - entries = [_entry("kimi-k2")] - return [ - Case( - name="gateways not resolved yet: require them and wait", - req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries)))), +# Every ModelService here sets timeouts that aren't the defaults, so a route +# carrying the defaults fails to match. Each composed ModelRoute's cluster label +# carries its gateway's clusterName, which compose-inference-cluster selects +# routes by. +COMPOSE_CASES = [ + Case( + name="gateways not resolved yet: require them and wait", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_model_service(labels=None)), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_service(total_routes=0, ready_routes=0, ready=fnv1.READY_FALSE) ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 0, "ready": 0}}} - ), - ready=fnv1.READY_FALSE, - ) - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" - ), - } - ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_WAITING_FOR_GATEWAYS, - message="Waiting for the gateways to resolve", - ) - ], - results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the gateways to resolve")], + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for the gateways to resolve")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + } ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateways", + message="Waiting for the gateways to resolve", + ) + ], ), - Case( - name="no gateway selects the service: unreachable, and say so", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_service(entries, labels={"region": "us"})) - ) - ), - required_resources=_required(gateways=[_gateway("eu", "gw-eu", selector={"region": "eu"})]), - ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 0, "ready": 0}}} - ), - ready=fnv1.READY_FALSE, - ) - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" + ), + Case( + name="no gateway selects the service: unreachable, and say so", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_model_service(labels={"region": "us"})), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="eu", + cluster_name="gw-eu", + service_selector={"matchLabels": {"region": "eu"}}, + address=None, ), - } + ] ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_NO_GATEWAY, - message=( - "No InferenceGateway's serviceSelector matches this service's labels, " - "so no caller can reach it" - ), - ) - ], - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message=( - "No InferenceGateway's serviceSelector matches this service's labels, " - "so no caller can reach it" - ), - ) - ], - ), + }, ), - Case( - # A gateway with no address is left out of readiness, but with no - # other gateway there is nowhere a caller could reach the service. - name="its only gateway has no address yet: not RoutingReady", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - ), - required_resources=_required(gateways=[_gateway("eu", "gw-eu")]), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_service(total_routes=0, ready_routes=0, ready=fnv1.READY_FALSE) ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 1, "ready": 0}}} - ), - ready=fnv1.READY_FALSE, + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message=( + "No InferenceGateway's serviceSelector matches this service's labels, so no caller can reach it" ), - resources={ - "route-eu": fnv1.Resource(resource=resource.dict_to_struct(_route("eu", "gw-eu"))), - }, - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" - ), - } - ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_WAITING_FOR_GATEWAYS, - message="Waiting for gateways to come up: eu", - ) - ], - results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for gateways to come up: eu")], + ) + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + } ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="NoGatewayServesThisService", + message=( + "No InferenceGateway's serviceSelector matches this service's labels, so no caller can reach it" + ), + ) + ], ), - Case( - name="two gateways serve it: a ModelRoute each, waiting for both routes", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - ), - required_resources=_required( - gateways=[ - _gateway("eu", "gw-eu", address="203.0.113.1"), - _gateway("us", "gw-us", address="203.0.113.2"), - ], + ), + # A gateway with no address is left out of readiness, but with no other + # gateway there is nowhere a caller could reach the service. + Case( + name="its only gateway has no address yet: not RoutingReady", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_model_service(labels=None)), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway(name="eu", cluster_name="gw-eu", service_selector=None, address=None), + ] ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_service(total_routes=1, ready_routes=0, ready=fnv1.READY_FALSE), + resources={ + "route-eu": _model_route( + name="assistant-eu-b0e0c", gateway="eu", cluster="gw-eu", ready=fnv1.READY_UNSPECIFIED + ), + }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 2, "ready": 0}}} + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for gateways to come up: eu")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForGateways", + message="Waiting for gateways to come up: eu", + ) + ], + ), + ), + Case( + name="two gateways serve it: a ModelRoute each, waiting for both routes", + req=fnv1.RunFunctionRequest( + observed=fnv1.State(composite=_model_service(labels=None)), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="eu", cluster_name="gw-eu", service_selector=None, address="203.0.113.1" ), - ready=fnv1.READY_FALSE, - ), - resources={ - "route-eu": fnv1.Resource(resource=resource.dict_to_struct(_route("eu", "gw-eu"))), - "route-us": fnv1.Resource(resource=resource.dict_to_struct(_route("us", "gw-us"))), - }, - ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" + _inference_gateway( + name="us", cluster_name="gw-us", service_selector=None, address="203.0.113.2" ), - } + ] ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_FALSE, - reason=fn.CONDITION_REASON_WAITING_FOR_ROUTES, - message="Waiting for routes on gateways: eu, us", - ) - ], - results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for routes on gateways: eu, us")], - ), + }, ), - Case( - name="both routes accepted: ModelRoutes ready, service RoutingReady", - req=fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - resources={ - "route-eu": _observed_route("eu", ready=True), - "route-us": _observed_route("us", ready=True), - }, - ), - required_resources=_required( - gateways=[ - _gateway("eu", "gw-eu", address="203.0.113.1"), - _gateway("us", "gw-us", address="203.0.113.2"), - ], - ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_service(total_routes=2, ready_routes=0, ready=fnv1.READY_FALSE), + resources={ + "route-eu": _model_route( + name="assistant-eu-b0e0c", gateway="eu", cluster="gw-eu", ready=fnv1.READY_UNSPECIFIED + ), + "route-us": _model_route( + name="assistant-us-04e03", gateway="us", cluster="gw-us", ready=fnv1.READY_UNSPECIFIED + ), + }, + ), + results=[fnv1.Result(severity=fnv1.SEVERITY_NORMAL, message="Waiting for routes on gateways: eu, us")], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + } ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_FALSE, + reason="WaitingForRoutes", + message="Waiting for routes on gateways: eu, us", + ) + ], + ), + ), + Case( + name="both routes accepted: ModelRoutes ready, service RoutingReady", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_service(labels=None), + resources={ + "route-eu": fnv1.Resource( resource=resource.dict_to_struct( - {"status": {"model": _MODEL, "routes": {"total": 2, "ready": 2}}} - ), - ready=fnv1.READY_TRUE, + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelRoute", + "metadata": {"name": "assistant-eu-b0e0c", "namespace": "ml-team"}, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + } + ] + }, + } + ) ), - resources={ - "route-eu": fnv1.Resource( - resource=resource.dict_to_struct(_route("eu", "gw-eu")), ready=fnv1.READY_TRUE + "route-us": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelRoute", + "metadata": {"name": "assistant-us-04e03", "namespace": "ml-team"}, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + } + ] + }, + } + ) + ), + }, + ), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="eu", cluster_name="gw-eu", service_selector=None, address="203.0.113.1" ), - "route-us": fnv1.Resource( - resource=resource.dict_to_struct(_route("us", "gw-us")), ready=fnv1.READY_TRUE + _inference_gateway( + name="us", cluster_name="gw-us", service_selector=None, address="203.0.113.2" ), - }, + ] ), - context=structpb.Struct(), - requirements=fnv1.Requirements( - resources={ - "gateways": fnv1.ResourceSelector( - api_version="modelplane.ai/v1alpha1", kind="InferenceGateway" + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_service(total_routes=2, ready_routes=2, ready=fnv1.READY_TRUE), + resources={ + "route-eu": _model_route( + name="assistant-eu-b0e0c", gateway="eu", cluster="gw-eu", ready=fnv1.READY_TRUE + ), + "route-us": _model_route( + name="assistant-us-04e03", gateway="us", cluster="gw-us", ready=fnv1.READY_TRUE + ), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="RoutesAccepted", + ) + ], + ), + ), + # Neither gateway has a serviceSelector, so both serve the service. The us + # gateway is still coming up (no address), so it's excluded from readiness + # rather than failing it: both RoutingReady and the service's own Ready ignore + # its unready ModelRoute. + Case( + name="a gateway with no address doesn't block readiness, nor does its unready ModelRoute", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_model_service(labels=None), + resources={ + "route-eu": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "modelplane.ai/v1alpha1", + "kind": "ModelRoute", + "metadata": {"name": "assistant-eu-b0e0c", "namespace": "ml-team"}, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2026-06-08T00:00:00Z", + } + ] + }, + } + ) + ), + }, + ), + required_resources={ + "gateways": fnv1.Resources( + items=[ + _inference_gateway( + name="eu", cluster_name="gw-eu", service_selector=None, address="203.0.113.1" ), - } + _inference_gateway(name="us", cluster_name="gw-us", service_selector=None, address=None), + ] ), - conditions=[ - fnv1.Condition( - type=fn.CONDITION_TYPE_ROUTING_READY, - status=fnv1.STATUS_CONDITION_TRUE, - reason=fn.CONDITION_REASON_ROUTES_ACCEPTED, - ) - ], + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_model_service(total_routes=2, ready_routes=1, ready=fnv1.READY_TRUE), + resources={ + "route-eu": _model_route( + name="assistant-eu-b0e0c", gateway="eu", cluster="gw-eu", ready=fnv1.READY_TRUE + ), + "route-us": _model_route( + name="assistant-us-04e03", gateway="us", cluster="gw-us", ready=fnv1.READY_UNSPECIFIED + ), + }, ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "gateways": fnv1.ResourceSelector(api_version="modelplane.ai/v1alpha1", kind="InferenceGateway"), + } + ), + conditions=[ + fnv1.Condition( + type="RoutingReady", + status=fnv1.STATUS_CONDITION_TRUE, + reason="RoutesAccepted", + ) + ], ), - ] - - -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + ), +] -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction composes a ModelRoute per serving gateway and reports routing readiness.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) assert _to_dict(got) == _to_dict(case.want) - - -def test_absent_selector_serves_every_service() -> None: - """A gateway with no serviceSelector serves the service, and one with no address doesn't block readiness.""" - # A gateway with no serviceSelector serves the service, and one still - # coming up (no address) is excluded from readiness rather than failing - # it: both RoutingReady and the service's own Ready ignore its unready - # ModelRoute. - entries = [_entry("kimi-k2")] - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_service(entries))), - resources={"route-eu": _observed_route("eu", ready=True)}, - ), - required_resources=_required( - gateways=[ - _gateway("eu", "gw-eu", address="203.0.113.1"), - _gateway("us", "gw-us"), # no address: still coming up - ], - ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - assert "route-eu" in got.desired.resources - assert "route-us" in got.desired.resources - cond = next(c for c in got.conditions if c.type == fn.CONDITION_TYPE_ROUTING_READY) - assert cond.status == fnv1.STATUS_CONDITION_TRUE, "us has no address, so it doesn't block" - assert got.desired.composite.ready == fnv1.READY_TRUE, "nor does its unready ModelRoute" diff --git a/functions/compose-nebius-cluster/tests/test_fn.py b/functions/compose-nebius-cluster/tests/test_fn.py index 74b131657..2c5e67cab 100644 --- a/functions/compose-nebius-cluster/tests/test_fn.py +++ b/functions/compose-nebius-cluster/tests/test_fn.py @@ -17,6 +17,7 @@ import asyncio import dataclasses import json +from typing import Any import pytest from crossplane.function import resource @@ -38,271 +39,259 @@ class Case: want: fnv1.RunFunctionResponse -# The Nebius ClusterProviderConfig the function reads the credentials -# Secret off. -_NEBIUS_PROVIDER_CONFIG = { - "apiVersion": "nebius.m.upbound.io/v1beta1", - "kind": "ClusterProviderConfig", - "metadata": {"name": "default"}, - "spec": { - "identity": {"type": "ServiceAccount"}, - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "crossplane-system", +def _nebius_cluster( + *, + max_node_count: int | None, + node_count: int, + fabric: v1alpha1.Fabric | None, + credentials: v1alpha1.Credentials | None, +) -> fnv1.Resource: + """The test-cluster NebiusCluster XR, with one gpu-h100 node pool.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.NebiusCluster( + metadata=metav1.ObjectMeta(name="test-cluster", namespace="modelplane-system"), + spec=v1alpha1.Spec( + nodePools=[ + v1alpha1.NodePool( + name="gpu-h100", + role="GPU", + platform="gpu-h100-sxm", + preset="8gpu-128vcpu-1600gb", + diskSizeGb=200, + maxNodeCount=max_node_count, + nodeCount=node_count, + fabric=fabric, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), + ), + ], + credentials=credentials, + ), + ).model_dump(exclude_none=True, mode="json", by_alias=True) + ), + ) + + +def _desired_nebius_cluster(*, credentials_secret: bool) -> fnv1.Resource: + """The desired NebiusCluster XR's status, which names the credentials Secret only if credentials_secret.""" + secrets = [{"type": "Kubeconfig", "name": "test-cluster-kubeconfig-55b57", "key": "kubeconfig"}] + if credentials_secret: + secrets.append( + { + "type": "NebiusServiceAccountCredentials", "name": "nebius-credentials", "key": "credentials.json", + "namespace": "crossplane-system", }, - }, - "projectID": "project-e00test", - }, -} - -_PROVIDER_CONFIG_SELECTOR = fnv1.ResourceSelector( - api_version="nebius.m.upbound.io/v1beta1", - kind="ClusterProviderConfig", - match_name="default", -) - -# Name of the composed cloud-init Secret. Derived like the function derives -# it - the hash suffix depends only on the parent and child names. -_CLOUD_INIT_SECRET_NAME = resource.child_name("test-cluster", "cloud-init") - -# The cloud-init user data mounting the cache filesystem on every node. -_CLOUD_INIT = ( - "#cloud-config\n" - "runcmd:\n" - " - mkdir -p /mnt/data\n" - " - mount -t virtiofs modelplane-cache /mnt/data\n" - ' - printf "modelplane-cache /mnt/data virtiofs defaults,nofail 0 2\\n" >> /etc/fstab\n' -) - -# The cache filesystem attachment and cloud-init reference every node group -# template carries. -_TEMPLATE_CACHE_MOUNT = { - "filesystems": [ - { - "attachMode": "READ_WRITE", - "mountTag": "modelplane-cache", - "existingFilesystem": {"idSelector": {"matchControllerRef": True}}, - }, - ], - "cloudInitUserDataSecretRef": {"name": _CLOUD_INIT_SECRET_NAME, "key": "userData"}, -} - - -def _xr( - pools: list[v1alpha1.NodePool], - credentials: v1alpha1.Credentials | None = None, -) -> dict: - """A NebiusCluster XR with the given node pools, as a request dict.""" - return v1alpha1.NebiusCluster( - metadata=metav1.ObjectMeta( - name="test-cluster", - namespace="modelplane-system", - ), - spec=v1alpha1.Spec( - nodePools=pools, - credentials=credentials, + ) + return fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"secrets": secrets, "cache": {"storageClassName": "modelplane-rwx-fs"}}}, ), - ).model_dump(exclude_none=True, mode="json") + ) -def _req( - pools: list[v1alpha1.NodePool], - observed_resources: dict[str, fnv1.Resource] | None = None, - credentials: v1alpha1.Credentials | None = None, - *, - with_provider_config: bool = True, - provider_config_resource: dict | None = None, -) -> fnv1.RunFunctionRequest: - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(pools, credentials))), - resources=observed_resources or {}, +def _nebius_provider_config(*, kind: str, name: str, namespace: str | None) -> fnv1.Resource: + """The Nebius provider config the XR's credentials name, sourcing them from a Secret.""" + metadata = {"name": name} + if namespace is not None: + metadata["namespace"] = namespace + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "nebius.m.upbound.io/v1beta1", + "kind": kind, + "metadata": metadata, + "spec": { + "identity": {"type": "ServiceAccount"}, + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "crossplane-system", + "name": "nebius-credentials", + "key": "credentials.json", + }, + }, + "projectID": "project-e00test", + }, + } ), ) - if with_provider_config: - pc = provider_config_resource if provider_config_resource is not None else _NEBIUS_PROVIDER_CONFIG - req.required_resources["nebius-provider-config"].items.append( - fnv1.Resource(resource=resource.dict_to_struct(pc)), - ) - return req - - -def _network(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "vpc.nebius.m.upbound.io/v1beta1", - "kind": "Network", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": {"name": "test-cluster"}, - }, - } -def _subnet(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "vpc.nebius.m.upbound.io/v1beta1", - "kind": "Subnet", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "name": "test-cluster", - "networkIdSelector": {"matchControllerRef": True}, - "ipv4PrivatePools": {"useNetworkPools": True}, - }, - }, - } - - -def _cluster(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", - "kind": "Cluster", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "name": "test-cluster", - "controlPlane": { - "version": "1.34", - "subnetIdSelector": {"matchControllerRef": True}, - "endpoints": {"publicEndpoint": {}}, +def _network(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The test-cluster VPC network.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vpc.nebius.m.upbound.io/v1beta1", + "kind": "Network", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": {"name": "test-cluster"}, }, - }, - "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, - }, - } + } + ), + ready=ready, + ) -def _filesystem(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "compute.nebius.m.upbound.io/v1beta1", - "kind": "Filesystem", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "name": "test-cluster-cache", - "type": "NETWORK_SSD", - "sizeGibibytes": 1024, - }, - }, - } +def _subnet(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The test-cluster subnet.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vpc.nebius.m.upbound.io/v1beta1", + "kind": "Subnet", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "name": "test-cluster", + "networkIdSelector": {"matchControllerRef": True}, + "ipv4PrivatePools": {"useNetworkPools": True}, + }, + }, + } + ), + ready=ready, + ) -def _cloud_init_secret() -> dict: - return { - "apiVersion": "v1", - "kind": "Secret", - "metadata": { - "name": _CLOUD_INIT_SECRET_NAME, - "namespace": "modelplane-system", - }, - "type": "Opaque", - "stringData": {"userData": _CLOUD_INIT}, - } +def _cluster(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The test-cluster mk8s cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", + "kind": "Cluster", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "name": "test-cluster", + "controlPlane": { + "version": "1.34", + "subnetIdSelector": {"matchControllerRef": True}, + "endpoints": {"publicEndpoint": {}}, + }, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, + }, + } + ), + ready=ready, + ) -def _csi_release() -> dict: - return { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "Release", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { - "kind": "ProviderConfig", - "name": "test-cluster-kubeconfig-55b57", - }, - "forProvider": { - "chart": { - "name": "csi-mounted-fs-path", - "repository": "oci://cr.eu-north1.nebius.cloud/mk8s/helm", - "version": "0.1.6", - }, - "namespace": "kube-system", - "values": { - "dataDir": "/mnt/data/csi-mounted-fs-path-data/", - # The node plugin must tolerate the GPU taint so engine - # pods on GPU nodes can mount cache PVCs. - "tolerations": [ - { - "key": "nvidia.com/gpu", - "operator": "Exists", - "effect": "NoSchedule", - }, - ], +def _filesystem(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The test-cluster cache filesystem.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.nebius.m.upbound.io/v1beta1", + "kind": "Filesystem", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "name": "test-cluster-cache", + "type": "NETWORK_SSD", + "sizeGibibytes": 1024, + }, }, - }, - }, - } + } + ), + ready=ready, + ) -def _storage_class() -> dict: - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe", "Create", "Update"], - "providerConfigRef": { - "kind": "ProviderConfig", - "name": "test-cluster-kubeconfig-55b57", - }, - "readiness": {"policy": "SuccessfulCreate"}, - "forProvider": { - "manifest": { - "apiVersion": "storage.k8s.io/v1", - "kind": "StorageClass", - "metadata": {"name": "modelplane-rwx-fs"}, - "provisioner": "mounted-fs-path.csi.nebius.ai", - "volumeBindingMode": "WaitForFirstConsumer", +def _cloud_init_secret() -> fnv1.Resource: + """The Ready Secret holding the cloud-init user data that mounts the cache filesystem on every node.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "v1", + "kind": "Secret", + "metadata": { + "name": "test-cluster-cloud-init-fd2f2", + "namespace": "modelplane-system", }, - }, - }, - } + "type": "Opaque", + "stringData": { + "userData": ( + "#cloud-config\n" + "runcmd:\n" + " - mkdir -p /mnt/data\n" + " - mount -t virtiofs modelplane-cache /mnt/data\n" + ' - printf "modelplane-cache /mnt/data virtiofs defaults,nofail 0 2\\n" >> /etc/fstab\n' + ), + }, + } + ), + ready=fnv1.READY_TRUE, + ) -def _nodegroup_system(cred_kind: str = "ClusterProviderConfig", cred_name: str = "default") -> dict: - return { - "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", - "kind": "NodeGroup", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "name": "test-cluster-system", - "parentIdSelector": {"matchControllerRef": True}, - "version": "1.34", - "autoscaling": {"minNodeCount": 1, "maxNodeCount": 2}, - "template": { - "resources": {"platform": "cpu-d3", "preset": "4vcpu-16gb"}, - "bootDisk": {"sizeGibibytes": 100, "type": "NETWORK_SSD"}, - "networkInterfaces": [ - {"subnetIdSelector": {"matchControllerRef": True}}, - ], - **_TEMPLATE_CACHE_MOUNT, - "metadata": {"labels": {"modelplane.ai/pool": "system"}}, +def _nodegroup_system(*, cred_kind: str, cred_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The system node group.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": { + "name": "test-cluster-system", + "parentIdSelector": {"matchControllerRef": True}, + "version": "1.34", + "autoscaling": {"minNodeCount": 1, "maxNodeCount": 2}, + "template": { + "resources": {"platform": "cpu-d3", "preset": "4vcpu-16gb"}, + "bootDisk": {"sizeGibibytes": 100, "type": "NETWORK_SSD"}, + "networkInterfaces": [ + {"subnetIdSelector": {"matchControllerRef": True}}, + ], + "filesystems": [ + { + "attachMode": "READ_WRITE", + "mountTag": "modelplane-cache", + "existingFilesystem": {"idSelector": {"matchControllerRef": True}}, + }, + ], + "cloudInitUserDataSecretRef": {"name": "test-cluster-cloud-init-fd2f2", "key": "userData"}, + "metadata": {"labels": {"modelplane.ai/pool": "system"}}, + }, + }, }, - }, - }, - } + } + ), + ready=ready, + ) def _nodegroup_gpu( - template_extra: dict, - cred_kind: str = "ClusterProviderConfig", - cred_name: str = "default", - **for_provider_extra: object, -) -> dict: - """A GPU node group golden with the standard template, merged with the - given scaling config and extra template fields.""" - template = { + *, + cred_kind: str, + cred_name: str, + autoscaling: dict | None, + fixed_node_count: int | None, + fabric: str | None, + ready: fnv1.Ready, +) -> fnv1.Resource: + """The gpu-h100 pool's node group, on fabric's GPU cluster if any.""" + template: dict[str, Any] = { "resources": {"platform": "gpu-h100-sxm", "preset": "8gpu-128vcpu-1600gb"}, "bootDisk": {"sizeGibibytes": 200, "type": "NETWORK_SSD"}, "networkInterfaces": [ {"subnetIdSelector": {"matchControllerRef": True}}, ], - **_TEMPLATE_CACHE_MOUNT, + "filesystems": [ + { + "attachMode": "READ_WRITE", + "mountTag": "modelplane-cache", + "existingFilesystem": {"idSelector": {"matchControllerRef": True}}, + }, + ], + "cloudInitUserDataSecretRef": {"name": "test-cluster-cloud-init-fd2f2", "key": "userData"}, "metadata": { "labels": { "modelplane.ai/pool": "gpu-h100", @@ -314,499 +303,808 @@ def _nodegroup_gpu( {"key": "nvidia.com/gpu", "value": "true", "effect": "NO_SCHEDULE"}, ], } - template.update(template_extra) - return { - "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", - "kind": "NodeGroup", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "name": "test-cluster-gpu-h100", - "parentIdSelector": {"matchControllerRef": True}, - "version": "1.34", - "template": template, - **for_provider_extra, - }, - }, + if fabric is not None: + template["gpuCluster"] = { + "idSelector": {"matchControllerRef": True, "matchLabels": {"modelplane.ai/fabric": fabric}}, + } + for_provider: dict[str, Any] = { + "name": "test-cluster-gpu-h100", + "parentIdSelector": {"matchControllerRef": True}, + "version": "1.34", + "template": template, } + if autoscaling is not None: + for_provider["autoscaling"] = autoscaling + if fixed_node_count is not None: + for_provider["fixedNodeCount"] = fixed_node_count + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": cred_kind, "name": cred_name}, + "forProvider": for_provider, + }, + } + ), + ready=ready, + ) -def _provider_config(api_version: str, kind: str) -> dict: - return { - "apiVersion": api_version, - "kind": kind, - "metadata": {"name": "test-cluster-kubeconfig-55b57"}, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "name": "test-cluster-kubeconfig-55b57", - "namespace": "modelplane-system", - "key": "kubeconfig", - }, - }, - "identity": { - "type": "NebiusServiceAccountCredentials", - "source": "Secret", - "secretRef": { - "name": "nebius-credentials", - "namespace": "crossplane-system", - "key": "credentials.json", +def _provider_config(*, api_version: str) -> fnv1.Resource: + """A Ready ProviderConfig targeting the cluster as the Nebius service account.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": api_version, + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster-kubeconfig-55b57"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", + "key": "kubeconfig", + }, + }, + "identity": { + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": { + "name": "nebius-credentials", + "namespace": "crossplane-system", + "key": "credentials.json", + }, + }, }, - }, - }, - } + } + ), + ready=fnv1.READY_TRUE, + ) -def _status(*, with_credentials: bool = True) -> dict: - secrets: list[dict] = [ - { - "type": "Kubeconfig", - "name": "test-cluster-kubeconfig-55b57", - "key": "kubeconfig", - }, - ] - if with_credentials: - secrets.append( - { - "type": "NebiusServiceAccountCredentials", - "name": "nebius-credentials", - "key": "credentials.json", - "namespace": "crossplane-system", - }, - ) - return { - "status": { - "secrets": secrets, - "cache": {"storageClassName": "modelplane-rwx-fs"}, - }, - } +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) -def _observed_ready(desired: dict) -> fnv1.Resource: - """An observed variant of a desired resource with a Ready=True condition.""" - observed = { - **desired, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", - }, - ], - }, - } - return fnv1.Resource(resource=resource.dict_to_struct(observed)) - - -_GPU_POOL = v1alpha1.NodePool( - name="gpu-h100", - role="GPU", - platform="gpu-h100-sxm", - preset="8gpu-128vcpu-1600gb", - diskSizeGb=200, - maxNodeCount=4, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), -) - - -def _compose_cases() -> list[Case]: - """The cases for test_compose, with requirements patched onto their wants by position.""" - cases = [ - Case( - name="first pass composes infra resources; autoscaling from maxNodeCount", - req=_req([_GPU_POOL]), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - # The CSI driver release and StorageClass aren't - # composed yet: the cluster isn't observed, so the - # ProviderConfigs can't reach it. - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - # nodeCount defaults to 1 and minNodeCount is - # unset, so autoscaling starts at the node count. - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, +# Every case is a NebiusCluster named test-cluster in modelplane-system. Every +# response requires the Nebius provider config the XR's credentials name, which +# is the ClusterProviderConfig named default unless the XR says otherwise. The +# function reads the credentials Secret off it. +COMPOSE_CASES = [ + # The pool sets maxNodeCount but not minNodeCount, so autoscaling starts at + # nodeCount. The CSI driver release and StorageClass aren't composed yet: the + # cluster isn't observed, so the ProviderConfigs can't reach it. + Case( + name="first pass composes infra resources; autoscaling from maxNodeCount", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_nebius_cluster(max_node_count=4, node_count=1, fabric=None, credentials=None), + ), + required_resources={ + "nebius-provider-config": fnv1.Resources( + items=[_nebius_provider_config(kind="ClusterProviderConfig", name="default", namespace=None)], ), - context=structpb.Struct(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_nebius_cluster(credentials_secret=True), + resources={ + "network": _network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "filesystem": _filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cloud-init": _cloud_init_secret(), + "nodegroup-system": _nodegroup_system( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-gpu-h100": _nodegroup_gpu( + cred_kind="ClusterProviderConfig", + cred_name="default", + autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, + fixed_node_count=None, + fabric=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "nebius-provider-config": fnv1.ResourceSelector( + api_version="nebius.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", + ), + }, ), ), - Case( - name="provider config not yet fetched gates provider configs, not infra", - req=_req([_GPU_POOL], with_provider_config=False), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct(_status(with_credentials=False)), - ), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), - ), - }, + ), + Case( + name="provider config not yet fetched gates provider configs, not infra", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_nebius_cluster(max_node_count=4, node_count=1, fabric=None, credentials=None), + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_nebius_cluster(credentials_secret=False), + resources={ + "network": _network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "filesystem": _filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cloud-init": _cloud_init_secret(), + "nodegroup-system": _nodegroup_system( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-gpu-h100": _nodegroup_gpu( + cred_kind="ClusterProviderConfig", + cred_name="default", + autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, + fixed_node_count=None, + fabric=None, + ready=fnv1.READY_UNSPECIFIED, + ), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Waiting for Nebius ClusterProviderConfig default", ), - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Waiting for Nebius ClusterProviderConfig default", + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "nebius-provider-config": fnv1.ResourceSelector( + api_version="nebius.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", ), - ], - context=structpb.Struct(), + }, ), ), - Case( - name="deleted provider config keeps credentials from the observed ProviderConfig", - req=_req( - [_GPU_POOL], - observed_resources={ + ), + # The Nebius provider config hasn't been fetched, but a composed + # ProviderConfig is observed, so the function keeps the credentials that + # ProviderConfig carries rather than tear the ProviderConfigs out of desired + # state. It falls back the same way when the config is gone. + Case( + name="provider config not yet fetched keeps credentials from the observed ProviderConfig", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_nebius_cluster(max_node_count=4, node_count=1, fabric=None, credentials=None), + resources={ "provider-config-kubernetes": fnv1.Resource( resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster-kubeconfig-55b57"}, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", + "key": "kubeconfig", + }, + }, + "identity": { + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": { + "name": "nebius-credentials", + "namespace": "crossplane-system", + "key": "credentials.json", + }, + }, + }, + } ), ), }, - with_provider_config=False, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_nebius_cluster(credentials_secret=True), + resources={ + "network": _network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "filesystem": _filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cloud-init": _cloud_init_secret(), + "nodegroup-system": _nodegroup_system( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-gpu-h100": _nodegroup_gpu( + cred_kind="ClusterProviderConfig", + cred_name="default", + autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, + fixed_node_count=None, + fabric=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_NORMAL, + message="Nebius ClusterProviderConfig default not found; keeping the " + "credentials the composed ProviderConfig already carries", ), - results=[ - fnv1.Result( - severity=fnv1.SEVERITY_NORMAL, - message="Nebius ClusterProviderConfig default not found; keeping the " - "credentials the composed ProviderConfig already carries", - ), - ], - context=structpb.Struct(), + ], + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "nebius-provider-config": fnv1.ResourceSelector( + api_version="nebius.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", + ), + }, ), ), - Case( - name="fixed-size fabric pool composes a GPU cluster and fixedNodeCount", - req=_req( - [ - v1alpha1.NodePool( - name="gpu-h100", - role="GPU", - platform="gpu-h100-sxm", - preset="8gpu-128vcpu-1600gb", - diskSizeGb=200, - nodeCount=2, - fabric=v1alpha1.Fabric( - type="InfiniBand", - infiniband=v1alpha1.Infiniband(fabric="fabric-2"), + ), + Case( + name="fixed-size fabric pool composes a GPU cluster and fixedNodeCount", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_nebius_cluster( + max_node_count=None, + node_count=2, + fabric=v1alpha1.Fabric(type="InfiniBand", infiniband=v1alpha1.Infiniband(fabric="fabric-2")), + credentials=None, + ), + ), + required_resources={ + "nebius-provider-config": fnv1.Resources( + items=[_nebius_provider_config(kind="ClusterProviderConfig", name="default", namespace=None)], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_nebius_cluster(credentials_secret=True), + resources={ + "network": _network( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "subnet": _subnet( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "filesystem": _filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "cloud-init": _cloud_init_secret(), + "gpu-cluster-fabric-2": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.nebius.m.upbound.io/v1beta1", + "kind": "GpuCluster", + "metadata": {"labels": {"modelplane.ai/fabric": "fabric-2"}}, + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "name": "test-cluster-fabric-2", + "infinibandFabric": "fabric-2", + }, + }, + } ), - gpu=v1alpha1.Gpu(acceleratorType="nvidia-h100"), ), - ] + "nodegroup-system": _nodegroup_system( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-gpu-h100": _nodegroup_gpu( + cred_kind="ClusterProviderConfig", + cred_name="default", + autoscaling=None, + fixed_node_count=2, + fabric="fabric-2", + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "nebius-provider-config": fnv1.ResourceSelector( + api_version="nebius.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", + ), + }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource(resource=resource.dict_to_struct(_network())), - "subnet": fnv1.Resource(resource=resource.dict_to_struct(_subnet())), - "cluster": fnv1.Resource(resource=resource.dict_to_struct(_cluster())), - "filesystem": fnv1.Resource(resource=resource.dict_to_struct(_filesystem())), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, + ), + ), + # The cluster is observed, so the CSI driver release and StorageClass are + # composed too. The release tolerates the GPU taint so its node plugin runs + # on the GPU nodes, where engine pods mount cache PVCs. + Case( + name="marks managed resources ready from observed conditions", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_nebius_cluster(max_node_count=4, node_count=1, fabric=None, credentials=None), + resources={ + "network": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vpc.nebius.m.upbound.io/v1beta1", + "kind": "Network", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": {"name": "test-cluster"}, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } ), - "gpu-cluster-fabric-2": fnv1.Resource( - resource=resource.dict_to_struct( - { - "apiVersion": "compute.nebius.m.upbound.io/v1beta1", - "kind": "GpuCluster", - "metadata": {"labels": {"modelplane.ai/fabric": "fabric-2"}}, - "spec": { - "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, - "forProvider": { - "name": "test-cluster-fabric-2", - "infinibandFabric": "fabric-2", + ), + "subnet": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vpc.nebius.m.upbound.io/v1beta1", + "kind": "Subnet", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "name": "test-cluster", + "networkIdSelector": {"matchControllerRef": True}, + "ipv4PrivatePools": {"useNetworkPools": True}, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ), + "cluster": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", + "kind": "Cluster", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "name": "test-cluster", + "controlPlane": { + "version": "1.34", + "subnetIdSelector": {"matchControllerRef": True}, + "endpoints": {"publicEndpoint": {}}, }, }, - } - ), + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } ), - "nodegroup-system": fnv1.Resource(resource=resource.dict_to_struct(_nodegroup_system())), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu( - { - "gpuCluster": { - "idSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/fabric": "fabric-2"}, - }, + ), + "filesystem": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "compute.nebius.m.upbound.io/v1beta1", + "kind": "Filesystem", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "name": "test-cluster-cache", + "type": "NETWORK_SSD", + "sizeGibibytes": 1024, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", }, + ], + }, + } + ), + ), + "release-csi-mounted-fs-path": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", }, - fixedNodeCount=2, - ), - ), + "forProvider": { + "chart": { + "name": "csi-mounted-fs-path", + "repository": "oci://cr.eu-north1.nebius.cloud/mk8s/helm", + "version": "0.1.6", + }, + "namespace": "kube-system", + "values": { + "dataDir": "/mnt/data/csi-mounted-fs-path-data/", + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + }, + ], + }, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ), + "nodegroup-system": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "name": "test-cluster-system", + "parentIdSelector": {"matchControllerRef": True}, + "version": "1.34", + "autoscaling": {"minNodeCount": 1, "maxNodeCount": 2}, + "template": { + "resources": {"platform": "cpu-d3", "preset": "4vcpu-16gb"}, + "bootDisk": {"sizeGibibytes": 100, "type": "NETWORK_SSD"}, + "networkInterfaces": [ + {"subnetIdSelector": {"matchControllerRef": True}}, + ], + "filesystems": [ + { + "attachMode": "READ_WRITE", + "mountTag": "modelplane-cache", + "existingFilesystem": {"idSelector": {"matchControllerRef": True}}, + }, + ], + "cloudInitUserDataSecretRef": { + "name": "test-cluster-cloud-init-fd2f2", + "key": "userData", + }, + "metadata": {"labels": {"modelplane.ai/pool": "system"}}, + }, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ), + "nodegroup-gpu-h100": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "mk8s.nebius.m.upbound.io/v1beta1", + "kind": "NodeGroup", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "name": "test-cluster-gpu-h100", + "parentIdSelector": {"matchControllerRef": True}, + "version": "1.34", + "template": { + "resources": {"platform": "gpu-h100-sxm", "preset": "8gpu-128vcpu-1600gb"}, + "bootDisk": {"sizeGibibytes": 200, "type": "NETWORK_SSD"}, + "networkInterfaces": [ + {"subnetIdSelector": {"matchControllerRef": True}}, + ], + "filesystems": [ + { + "attachMode": "READ_WRITE", + "mountTag": "modelplane-cache", + "existingFilesystem": {"idSelector": {"matchControllerRef": True}}, + }, + ], + "cloudInitUserDataSecretRef": { + "name": "test-cluster-cloud-init-fd2f2", + "key": "userData", + }, + "metadata": { + "labels": { + "modelplane.ai/pool": "gpu-h100", + "modelplane.ai/gpu": "nvidia-h100", + }, + }, + "gpuSettings": {"driversPreset": "cuda13.0"}, + "taints": [ + {"key": "nvidia.com/gpu", "value": "true", "effect": "NO_SCHEDULE"}, + ], + }, + "autoscaling": {"minNodeCount": 1, "maxNodeCount": 4}, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } ), - }, - ), - context=structpb.Struct(), - ), - ), - Case( - name="marks managed resources ready from observed conditions", - req=_req( - [_GPU_POOL], - observed_resources={ - "network": _observed_ready(_network()), - "subnet": _observed_ready(_subnet()), - "cluster": _observed_ready(_cluster()), - "filesystem": _observed_ready(_filesystem()), - "release-csi-mounted-fs-path": _observed_ready(_csi_release()), - "nodegroup-system": _observed_ready(_nodegroup_system()), - "nodegroup-gpu-h100": _observed_ready( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), ), }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network()), - ready=fnv1.READY_TRUE, - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet()), - ready=fnv1.READY_TRUE, - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "filesystem": fnv1.Resource( - resource=resource.dict_to_struct(_filesystem()), - ready=fnv1.READY_TRUE, - ), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - # The cluster is observed, so the CSI driver - # release and StorageClass are composed too. - "release-csi-mounted-fs-path": fnv1.Resource( - resource=resource.dict_to_struct(_csi_release()), - ready=fnv1.READY_TRUE, - ), - "storage-class-rwx-fs": fnv1.Resource( - resource=resource.dict_to_struct(_storage_class()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource( - resource=resource.dict_to_struct(_nodegroup_system()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu({}, autoscaling={"minNodeCount": 1, "maxNodeCount": 4}), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + required_resources={ + "nebius-provider-config": fnv1.Resources( + items=[_nebius_provider_config(kind="ClusterProviderConfig", name="default", namespace=None)], + ), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_nebius_cluster(credentials_secret=True), + resources={ + "network": _network(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "subnet": _subnet(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "cluster": _cluster(cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE), + "filesystem": _filesystem( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "cloud-init": _cloud_init_secret(), + "release-csi-mounted-fs-path": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "forProvider": { + "chart": { + "name": "csi-mounted-fs-path", + "repository": "oci://cr.eu-north1.nebius.cloud/mk8s/helm", + "version": "0.1.6", + }, + "namespace": "kube-system", + "values": { + "dataDir": "/mnt/data/csi-mounted-fs-path-data/", + "tolerations": [ + { + "key": "nvidia.com/gpu", + "operator": "Exists", + "effect": "NoSchedule", + }, + ], + }, + }, + }, + } ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ready=fnv1.READY_TRUE, + ), + "storage-class-rwx-fs": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe", "Create", "Update"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "readiness": {"policy": "SuccessfulCreate"}, + "forProvider": { + "manifest": { + "apiVersion": "storage.k8s.io/v1", + "kind": "StorageClass", + "metadata": {"name": "modelplane-rwx-fs"}, + "provisioner": "mounted-fs-path.csi.nebius.ai", + "volumeBindingMode": "WaitForFirstConsumer", + }, + }, + }, + } ), - }, - ), - context=structpb.Struct(), + ready=fnv1.READY_TRUE, + ), + "nodegroup-system": _nodegroup_system( + cred_kind="ClusterProviderConfig", cred_name="default", ready=fnv1.READY_TRUE + ), + "nodegroup-gpu-h100": _nodegroup_gpu( + cred_kind="ClusterProviderConfig", + cred_name="default", + autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, + fixed_node_count=None, + fabric=None, + ready=fnv1.READY_TRUE, + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, ), - ), - Case( - name="custom credentials flow through to all cloud MRs", - req=_req( - [_GPU_POOL], - credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-nebius-account"), - provider_config_resource={ - "apiVersion": "nebius.m.upbound.io/v1beta1", - "kind": "ProviderConfig", - "metadata": {"name": "my-nebius-account", "namespace": "crossplane-system"}, - "spec": { - "identity": {"type": "ServiceAccount"}, - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "crossplane-system", - "name": "nebius-credentials", - "key": "credentials.json", - }, - }, - "projectID": "project-e00test", - }, + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "nebius-provider-config": fnv1.ResourceSelector( + api_version="nebius.m.upbound.io/v1beta1", + kind="ClusterProviderConfig", + match_name="default", + ), }, ), - want=fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), - resources={ - "network": fnv1.Resource( - resource=resource.dict_to_struct(_network("ProviderConfig", "my-nebius-account")), - ), - "subnet": fnv1.Resource( - resource=resource.dict_to_struct(_subnet("ProviderConfig", "my-nebius-account")), - ), - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster("ProviderConfig", "my-nebius-account")), - ), - "filesystem": fnv1.Resource( - resource=resource.dict_to_struct(_filesystem("ProviderConfig", "my-nebius-account")), - ), - "cloud-init": fnv1.Resource( - resource=resource.dict_to_struct(_cloud_init_secret()), - ready=fnv1.READY_TRUE, - ), - "nodegroup-system": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_system("ProviderConfig", "my-nebius-account"), - ), - ), - "nodegroup-gpu-h100": fnv1.Resource( - resource=resource.dict_to_struct( - _nodegroup_gpu( - {}, - "ProviderConfig", - "my-nebius-account", - autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, - ), - ), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("kubernetes.m.crossplane.io/v1alpha1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct( - _provider_config("helm.m.crossplane.io/v1beta1", "ProviderConfig"), - ), - ready=fnv1.READY_TRUE, + ), + ), + # The XR names a namespaced ProviderConfig, so the function requires it from + # the XR's own namespace. The ProviderConfig returned here sits in + # crossplane-system, which Crossplane wouldn't return for that selector. The + # function looks for the credentials Secret in the ProviderConfig's + # namespace, so that's the namespace status names. + Case( + name="custom credentials flow through to all cloud MRs", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_nebius_cluster( + max_node_count=4, + node_count=1, + fabric=None, + credentials=v1alpha1.Credentials(type="ProviderConfig", name="my-nebius-account"), + ), + ), + required_resources={ + "nebius-provider-config": fnv1.Resources( + items=[ + _nebius_provider_config( + kind="ProviderConfig", name="my-nebius-account", namespace="crossplane-system" ), - }, + ], ), - context=structpb.Struct(), + }, + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_nebius_cluster(credentials_secret=True), + resources={ + "network": _network( + cred_kind="ProviderConfig", cred_name="my-nebius-account", ready=fnv1.READY_UNSPECIFIED + ), + "subnet": _subnet( + cred_kind="ProviderConfig", cred_name="my-nebius-account", ready=fnv1.READY_UNSPECIFIED + ), + "cluster": _cluster( + cred_kind="ProviderConfig", cred_name="my-nebius-account", ready=fnv1.READY_UNSPECIFIED + ), + "filesystem": _filesystem( + cred_kind="ProviderConfig", cred_name="my-nebius-account", ready=fnv1.READY_UNSPECIFIED + ), + "cloud-init": _cloud_init_secret(), + "nodegroup-system": _nodegroup_system( + cred_kind="ProviderConfig", cred_name="my-nebius-account", ready=fnv1.READY_UNSPECIFIED + ), + "nodegroup-gpu-h100": _nodegroup_gpu( + cred_kind="ProviderConfig", + cred_name="my-nebius-account", + autoscaling={"minNodeCount": 1, "maxNodeCount": 4}, + fixed_node_count=None, + fabric=None, + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-kubernetes": _provider_config(api_version="kubernetes.m.crossplane.io/v1alpha1"), + "provider-config-helm": _provider_config(api_version="helm.m.crossplane.io/v1beta1"), + }, + ), + context=structpb.Struct(), + requirements=fnv1.Requirements( + resources={ + "nebius-provider-config": fnv1.ResourceSelector( + api_version="nebius.m.upbound.io/v1beta1", + kind="ProviderConfig", + match_name="my-nebius-account", + namespace="modelplane-system", + ), + }, ), ), - ] - - # Every compose path declares the provider config requirement; the - # selector kind and name vary by credentials. - custom_creds_selector = fnv1.ResourceSelector( - api_version="nebius.m.upbound.io/v1beta1", - kind="ProviderConfig", - match_name="my-nebius-account", - namespace="modelplane-system", - ) - for case in cases[:-1]: - case.want.requirements.resources["nebius-provider-config"].CopyFrom(_PROVIDER_CONFIG_SELECTOR) - cases[-1].want.requirements.resources["nebius-provider-config"].CopyFrom(custom_creds_selector) - return cases - - -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + ), +] -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction composes a NebiusCluster's network, mk8s cluster, node groups and ProviderConfigs.""" got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) diff --git a/functions/compose-serving-stack/tests/test_fn.py b/functions/compose-serving-stack/tests/test_fn.py index 9213ba2fb..40d8d633c 100644 --- a/functions/compose-serving-stack/tests/test_fn.py +++ b/functions/compose-serving-stack/tests/test_fn.py @@ -14,17 +14,17 @@ """Tests for the compose-serving-stack function. -Two layers. The Case table compares whole RunFunctionResponses for the -Existing/Dynamo stack across the reconcile passes; its expectations are -built from the provider models with literal arguments typed here, never -from the stacks package, so a stack-data change shows up as a test diff. -The golden inventory then pins the composed-resource key set - the -identity contract; renaming a key deletes and recreates the remote -resource - for every cloud and stack, as frozen literals. +Two tables. COMPOSE_CASES compares whole RunFunctionResponses: the +Existing/Dynamo stack across the reconcile passes, a non-GCP identity secret, +and the Existing/Standard stack's gateway with and without a client CA. Its +expectations are literals typed here, never read from the stacks package, so a +stack-data change shows up as a test diff. Only the vendored CRD bundles are +read from their files. COMPOSED_RESOURCE_KEYS_CASES then pins the +composed-resource key set - the identity contract; renaming a key deletes and +recreates the remote resource - for every cloud and stack. """ import asyncio -import copy import dataclasses import json import pathlib @@ -33,1126 +33,2646 @@ import yaml from crossplane.function import resource from crossplane.function.proto.v1 import run_function_pb2 as fnv1 -from function import fn +from function import fn, stacks from google.protobuf import duration_pb2 as durationpb from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb from models.ai.modelplane.infrastructure.servingstack import v1alpha1 -from models.io.crossplane.m.helm.providerconfig import v1beta1 as helmpcv1beta1 -from models.io.crossplane.m.helm.release import v1beta1 as helmv1beta1 -from models.io.crossplane.m.kubernetes.object import v1alpha1 as k8sobjv1alpha1 -from models.io.crossplane.m.kubernetes.providerconfig import ( - v1alpha1 as k8spcv1alpha1, -) -from models.io.crossplane.protection.usage import v1beta1 as usagev1beta1 from models.io.k8s.apimachinery.pkg.apis.meta import v1 as metav1 -# Precomputed child_name value for test-backend. -_PC_NAME = "test-backend-cluster-63fde" - -_RELEASE_REF = ("helm.m.crossplane.io/v1beta1", "Release") -_OBJECT_REF = ("kubernetes.m.crossplane.io/v1alpha1", "Object") - -_GATEWAY_READY_CEL = "has(object.status.addresses) && object.status.addresses.size() > 0" -_CERTIFICATE_READY_CEL = ( - "has(object.status) && has(object.status.conditions) && " - "object.status.conditions.exists(c, c.type == 'Ready' && c.status == 'True')" -) -_BUNDLE_SYNCED_CEL = ( - "has(object.status) && has(object.status.conditions) && " - "object.status.conditions.exists(c, c.type == 'Synced' && c.status == 'True')" -) -_POLICY_ACCEPTED_CEL = ( - "has(object.status) && has(object.status.ancestors) && " - "object.status.ancestors.exists(a, has(a.conditions) && " - "a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))" -) -_MODELEXPRESS_READY_CEL = ( - 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")' -) - -# The name InferenceGateways reach the test stack's gateway by, and one -# InferenceGateway's client CA for it to trust. With both, the gateway serves. -_GATEWAY_HOSTNAME = "test-backend.gateways.example.com" -_CLIENT_CA = "-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n" - -# Resolve the vendored CRD bundles via the installed function package: -# the sandboxed test check runs against the venv's copy, not the tree. -_CRDS_DIR = pathlib.Path(fn.__file__).parent / "stacks" / "crds" - - -def _crds(filename: str) -> list[dict]: - """The CRDs a vendored bundle carries, content straight from the file.""" - return [ + +@dataclasses.dataclass +class ComposeCase: + """A test case for RunFunction's whole response.""" + + name: str + req: fnv1.RunFunctionRequest + want: fnv1.RunFunctionResponse + + +@dataclasses.dataclass +class ComposedResourceKeysCase: + """A test case for the composed-resource keys RunFunction renders.""" + + name: str + req: fnv1.RunFunctionRequest + want: set[str] + + +def _crd(*, filename: str, name: str) -> dict: + """The CRD named name in the vendored bundle filename, as the file has it.""" + # The vendored CRD bundles are upstream release artifacts, a thousand lines + # of schema, so the expectations read them rather than restating them. They + # resolve via the installed function package, because the sandboxed test + # check runs against the venv's copy, not the tree. + bundle = pathlib.Path(fn.__file__).parent / "stacks" / "crds" / filename + return next( doc - for doc in yaml.safe_load_all((_CRDS_DIR / filename).read_text()) - if doc and doc.get("kind") == "CustomResourceDefinition" - ] - - -def _request(cloud: str, stack: str, observed: dict | None = None) -> fnv1.RunFunctionRequest: - """Build a RunFunctionRequest for a test-backend ServingStack.""" - return fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud=cloud, # ty: ignore[invalid-argument-type] # cases pass values of the literal - stack=stack, # ty: ignore[invalid-argument-type] - secrets=[ - v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), - v1alpha1.Secret( - type="GoogleApplicationCredentials", name="sa-secret", key="private_key" - ), - ], - gateway=v1alpha1.Gateway( - hostname=_GATEWAY_HOSTNAME, - clientCAs=[v1alpha1.ClientCA(name="eu", certificate=_CLIENT_CA)], - ), - ), - ).model_dump(exclude_none=True, mode="json") - ), - ), - resources=observed or {}, - ), + for doc in yaml.safe_load_all(bundle.read_text()) + if doc and doc["kind"] == "CustomResourceDefinition" and doc["metadata"]["name"] == name ) -def _release( - key: str, - release: str, - namespace: str, - chart: str, - repository: str, - version: str, - values: dict | None = None, - *, - wait: bool = False, -) -> fnv1.Resource: - """The expected Release for a Chart entry, built from literal arguments.""" - model = helmv1beta1.Release( - metadata=metav1.ObjectMeta( - annotations={"crossplane.io/external-name": release}, - labels={"modelplane.ai/resource": key}, - ), - spec=helmv1beta1.Spec( - providerConfigRef=helmv1beta1.ProviderConfigRef(kind="ProviderConfig", name=_PC_NAME), - forProvider=helmv1beta1.ForProvider( - chart=helmv1beta1.Chart(name=chart, repository=repository, version=version), - namespace=namespace, - ), - ), - ) - if wait: - model.spec.forProvider.wait = True - model.spec.forProvider.waitTimeout = "10m" - if values: - model.spec.forProvider.values = values - res = fnv1.Resource() - resource.update(res, model) - return res - - -def _object( - key: str, - manifest: dict, - cel: str | None = None, - *, - labeled: bool = True, - management_policies: list | None = None, +def _serving_stack( + *, cloud: stacks.Cloud, stack: stacks.Stack, secrets: list[v1alpha1.Secret], gateway: v1alpha1.Gateway ) -> fnv1.Resource: - """The expected Object for one manifest, built from literal arguments. - - The gateway PKI objects carry no resource label (nothing selects them in - a Usage), so labeled=False builds them without one. - """ - model = k8sobjv1alpha1.Object( - # Omit metadata entirely when unlabeled: a null metadata would serialize - # into the composed resource rather than being absent, as it is when the - # function passes none. - **({"metadata": metav1.ObjectMeta(labels={"modelplane.ai/resource": key})} if labeled else {}), - spec=k8sobjv1alpha1.Spec( - providerConfigRef=k8sobjv1alpha1.ProviderConfigRef(kind="ProviderConfig", name=_PC_NAME), - forProvider=k8sobjv1alpha1.ForProvider(manifest=manifest), - ), - ) - if management_policies: - model.spec.managementPolicies = management_policies - if cel is not None: - model.spec.readiness = k8sobjv1alpha1.Readiness(policy="DeriveFromCelQuery", celQuery=cel) - res = fnv1.Resource() - resource.update(res, model) - return res - - -def _usage(of_ref: tuple[str, str], of_key: str, by_ref: tuple[str, str], by_key: str) -> fnv1.Resource: - """The expected teardown Usage for one dependency edge, ready on arrival.""" - res = fnv1.Resource() - resource.update( - res, - usagev1beta1.Usage( - spec=usagev1beta1.Spec( - of=usagev1beta1.Of( - apiVersion=of_ref[0], - kind=of_ref[1], - resourceSelector=usagev1beta1.ResourceSelectorModel( - matchControllerRef=True, - matchLabels={"modelplane.ai/resource": of_key}, - ), - ), - by=usagev1beta1.By( - apiVersion=by_ref[0], - kind=by_ref[1], - resourceSelector=usagev1beta1.ResourceSelector( - matchControllerRef=True, - matchLabels={"modelplane.ai/resource": by_key}, - ), - ), - replayDeletion=True, - ), - ), + """The observed ServingStack, test-backend in namespace test-ns.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + v1alpha1.ServingStack( + metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), + spec=v1alpha1.Spec(cloud=cloud, stack=stack, secrets=secrets, gateway=gateway), + ).model_dump(exclude_none=True, mode="json", by_alias=True) + ) ) - res.ready = fnv1.READY_TRUE - return res - - -def _provider_configs(*, ready: bool = True) -> dict[str, fnv1.Resource]: - """The two expected ProviderConfigs. - - Ready only once observed: on the first pass they and the Usages are - the whole desired state, and ready-on-arrival would let the - composite report Ready before any stack component exists. - """ - k8s = fnv1.Resource() - resource.update( - k8s, - k8spcv1alpha1.ProviderConfig( - metadata=metav1.ObjectMeta(name=_PC_NAME), - spec=k8spcv1alpha1.Spec( - credentials=k8spcv1alpha1.Credentials( - source="Secret", - secretRef=k8spcv1alpha1.SecretRef(name="kube-secret", namespace="test-ns", key="kubeconfig"), - ), - identity=k8spcv1alpha1.Identity( - type="GoogleApplicationCredentials", - source="Secret", - secretRef=k8spcv1alpha1.SecretRef(name="sa-secret", namespace="test-ns", key="private_key"), - ), - ), - ), + + +def _desired_serving_stack(*, gateway: dict | None) -> fnv1.Resource: + """The desired ServingStack, publishing gateway in its status if there is one.""" + status = {} if gateway is None else {"gateway": gateway} + return fnv1.Resource(resource=resource.dict_to_struct({"status": status})) + + +def _observed_ready() -> fnv1.Resource: + """An observed composed resource whose Ready condition is True.""" + return fnv1.Resource( + resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) ) - if ready: - k8s.ready = fnv1.READY_TRUE - helm = fnv1.Resource() - resource.update( - helm, - helmpcv1beta1.ProviderConfig( - metadata=metav1.ObjectMeta(name=_PC_NAME), - spec=helmpcv1beta1.Spec( - credentials=helmpcv1beta1.Credentials( - source="Secret", - secretRef=helmpcv1beta1.SecretRef(name="kube-secret", namespace="test-ns", key="kubeconfig"), - ), - identity=helmpcv1beta1.Identity( - type="GoogleApplicationCredentials", - source="Secret", - secretRef=helmpcv1beta1.SecretRef(name="sa-secret", namespace="test-ns", key="private_key"), - ), - ), - ), + + +def _observed_kubernetes_provider_config() -> fnv1.Resource: + """The observed provider-kubernetes ProviderConfig, which has no conditions.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + {"apiVersion": "kubernetes.m.crossplane.io/v1alpha1", "kind": "ProviderConfig"} + ) ) - if ready: - helm.ready = fnv1.READY_TRUE - return {"provider-config-kubernetes": k8s, "provider-config-helm": helm} - - -def _observed_pcs() -> dict[str, fnv1.Resource]: - """Observed ProviderConfigs, which gate the rest of the stack open.""" - return { - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct( - {"apiVersion": "kubernetes.m.crossplane.io/v1alpha1", "kind": "ProviderConfig"} - ) - ), - "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct({"apiVersion": "helm.m.crossplane.io/v1beta1", "kind": "ProviderConfig"}) - ), - } -# The Usages every Existing/Dynamo pass composes: the two hand-written -# gateway-chain edges, and one derived edge per depends_on in the joined -# stack data. -_EXISTING_DYNAMO_USAGES = { - "usage-gateway-class-by-gateway": _usage(_OBJECT_REF, "gateway-class", _OBJECT_REF, "gateway"), - "usage-envoy-gateway-by-gateway-class": _usage(_RELEASE_REF, "envoy-gateway", _OBJECT_REF, "gateway-class"), - "usage-cert-manager-by-envoy-gateway": _usage(_RELEASE_REF, "cert-manager", _RELEASE_REF, "envoy-gateway"), - "usage-ai-gateway-crds-by-ai-gateway": _usage(_RELEASE_REF, "ai-gateway-crds", _RELEASE_REF, "ai-gateway"), - "usage-gateway-namespace-by-gateway-proxy": _usage(_OBJECT_REF, "gateway-namespace", _OBJECT_REF, "gateway-proxy"), - "usage-cert-manager-by-gateway-selfsigned-issuer": _usage( - _RELEASE_REF, "cert-manager", _OBJECT_REF, "gateway-selfsigned-issuer" - ), - "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage( - _OBJECT_REF, "gateway-namespace", _OBJECT_REF, "gateway-selfsigned-issuer" - ), - "usage-gateway-selfsigned-issuer-by-trust-manager": _usage( - _OBJECT_REF, "gateway-selfsigned-issuer", _RELEASE_REF, "trust-manager" - ), - "usage-kai-scheduler-by-kai-queue-root": _usage(_RELEASE_REF, "kai-scheduler", _OBJECT_REF, "kai-queue-root"), - "usage-kai-scheduler-by-kai-queue": _usage(_RELEASE_REF, "kai-scheduler", _OBJECT_REF, "kai-queue"), - "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _usage( - _OBJECT_REF, "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", _OBJECT_REF, "modelexpress-server" - ), - "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _usage( - _OBJECT_REF, "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", _OBJECT_REF, "modelexpress-server" - ), -} +def _observed_helm_provider_config() -> fnv1.Resource: + """The observed provider-helm ProviderConfig, which has no conditions.""" + return fnv1.Resource( + resource=resource.dict_to_struct({"apiVersion": "helm.m.crossplane.io/v1beta1", "kind": "ProviderConfig"}) + ) -def _kai_queue(name: str, parent: str | None) -> dict: +def _kubernetes_provider_config(*, identity: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The composed provider-kubernetes ProviderConfig, authenticating as identity if there is one.""" spec: dict = { - "resources": { - "cpu": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, - "gpu": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, - "memory": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, + "credentials": { + "source": "Secret", + "secretRef": {"name": "kube-secret", "namespace": "test-ns", "key": "kubeconfig"}, }, } - if parent: - spec["parentQueue"] = parent - return {"apiVersion": "scheduling.run.ai/v2", "kind": "Queue", "metadata": {"name": name}, "spec": spec} - - -_MX_META = {"name": "modelexpress-server", "namespace": "default"} -_MX_SELECT = {"modelplane.ai/modelexpress": "modelexpress-server"} - - -def _existing_dynamo_stack() -> dict[str, fnv1.Resource]: - """Every component the Existing/Dynamo stack renders, as literals.""" - out: dict[str, fnv1.Resource] = {} - - # --- the Existing cloud half (hand-written Modelplane pins) --- - out["cert-manager"] = _release( - key="cert-manager", - release="mp-cert-manager", - namespace="cert-manager", - chart="cert-manager", - repository="https://charts.jetstack.io", - version="v1.20.2", - wait=True, - # clusterResourceNamespace and enableCertificateOwnerRef are forced by - # fn._helm_release for every cloud's cert-manager: the ClusterIssuer CA - # lives in modelplane-system, and a deleted ModelRoute's client - # certificate Secret must go with its Certificate. - values={ - "crds": {"enabled": True}, - "clusterResourceNamespace": "modelplane-system", - "enableCertificateOwnerRef": True, - }, - ) - out["kube-prometheus-stack"] = _release( - key="kube-prometheus-stack", - release="mp-kube-prometheus-stack", - namespace="monitoring", - chart="kube-prometheus-stack", - repository="https://prometheus-community.github.io/helm-charts", - version="84.4.0", - values={ - "fullnameOverride": "prometheus", - "prometheus": { - "prometheusSpec": { - "podMonitorSelectorNilUsesHelmValues": False, - "podMonitorNamespaceSelector": {}, - "additionalScrapeConfigs": [ - { - "job_name": "envoy-gateway-proxy", - "kubernetes_sd_configs": [ - {"role": "pod", "namespaces": {"names": ["envoy-gateway-system"]}}, - ], - "relabel_configs": [ - { - "source_labels": [ - "__meta_kubernetes_pod_label_app_kubernetes_io_component", - ], - "action": "keep", - "regex": "proxy", - }, - { - "source_labels": ["__address__"], - "action": "replace", - "regex": "([^:]+)(?::\\d+)?", - "replacement": "$1:19001", - "target_label": "__address__", - }, - ], - "metrics_path": "/stats/prometheus", - }, - ], - }, - }, - "grafana": {"enabled": False}, - "alertmanager": {"enabled": False}, - }, - ) - out["node-feature-discovery"] = _release( - key="node-feature-discovery", - release="mp-node-feature-discovery", - namespace="node-feature-discovery", - chart="node-feature-discovery", - repository="https://kubernetes-sigs.github.io/node-feature-discovery/charts", - version="0.19.0", - values={ - "worker": { - "tolerations": [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}], - }, - }, + if identity is not None: + spec["identity"] = identity + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ProviderConfig", + "metadata": {"name": "test-backend-cluster-63fde"}, + "spec": spec, + } + ), + ready=ready, ) - out["nvidia-dra-driver-gpu"] = _release( - key="nvidia-dra-driver-gpu", - release="mp-dra-driver-nvidia-gpu", - namespace="nvidia-dra-driver", - chart="dra-driver-nvidia-gpu", - repository="oci://registry.k8s.io/dra-driver-nvidia/charts", - version="0.4.1", - values={ - "gpuResourcesEnabledOverride": True, - "resources": {"computeDomains": {"enabled": False}}, + + +def _helm_provider_config(*, identity: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The composed provider-helm ProviderConfig, authenticating as identity if there is one.""" + spec: dict = { + "credentials": { + "source": "Secret", + "secretRef": {"name": "kube-secret", "namespace": "test-ns", "key": "kubeconfig"}, }, + } + if identity is not None: + spec["identity"] = identity + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "metadata": {"name": "test-backend-cluster-63fde"}, + "spec": spec, + } + ), + ready=ready, ) - # --- the common half --- - out["envoy-gateway"] = _release( - key="envoy-gateway", - release="mp-gateway-helm", - namespace="envoy-gateway-system", - chart="gateway-helm", - repository="oci://docker.io/envoyproxy", - version="v1.8.4", - values={ - "config": { - "envoyGateway": { - "extensionApis": {"enableBackend": True}, - "extensionManager": { - "hooks": { - "xdsTranslator": { - "translation": { - "listener": {"includeAll": True}, - "route": {"includeAll": True}, - "cluster": {"includeAll": True}, - "secret": {"includeAll": True}, - }, - "post": ["Translation", "Cluster", "Route"], - }, + +def _cert_manager(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed cert-manager Release.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-cert-manager"}, + "labels": {"modelplane.ai/resource": "cert-manager"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "cert-manager", + "repository": "https://charts.jetstack.io", + "version": "v1.20.2", }, - "service": { - "fqdn": { - "hostname": "ai-gateway-controller.envoy-ai-gateway-system.svc.cluster.local", - "port": 1063, - }, + "namespace": "cert-manager", + "wait": True, + "waitTimeout": "10m", + # clusterResourceNamespace and enableCertificateOwnerRef are forced by + # fn._helm_release for every cloud's cert-manager: the ClusterIssuer CA + # lives in modelplane-system, and a deleted ModelRoute's client + # certificate Secret must go with its Certificate. + "values": { + "crds": {"enabled": True}, + "clusterResourceNamespace": "modelplane-system", + "enableCertificateOwnerRef": True, }, - "backendResources": [ - {"group": "inference.networking.k8s.io", "kind": "InferencePool", "version": "v1"}, - ], }, }, - }, - }, - ) - out["ai-gateway-crds"] = _release( - key="ai-gateway-crds", - release="mp-ai-gateway-crds-helm", - namespace="envoy-ai-gateway-system", - chart="ai-gateway-crds-helm", - repository="oci://docker.io/envoyproxy", - version="v1.1.0", - wait=True, - ) - out["ai-gateway"] = _release( - key="ai-gateway", - release="mp-ai-gateway-helm", - namespace="envoy-ai-gateway-system", - chart="ai-gateway-helm", - repository="oci://docker.io/envoyproxy", - version="v1.1.0", - values={"controller": {"logRequestHeaderAttributes": "x-modelplane-caller:caller"}}, - ) - for doc in _crds("gaie.yaml"): - key = f"gaie-crds-{doc['metadata']['name']}" - out[key] = _object(key, doc) - out["gateway-namespace"] = _object( - "gateway-namespace", - { - "apiVersion": "v1", - "kind": "Namespace", - "metadata": {"name": "modelplane-system", "labels": {"modelplane.ai/namespace": "modelplane-system"}}, - }, - ) - out["gateway-proxy"] = _object( - "gateway-proxy", - { - "apiVersion": "gateway.envoyproxy.io/v1alpha1", - "kind": "EnvoyProxy", - "metadata": {"name": "cluster-gateway", "namespace": "modelplane-system"}, - "spec": { - "provider": { - "type": "Kubernetes", - "kubernetes": {"envoyService": {"externalTrafficPolicy": "Cluster"}}, - }, - }, - }, - ) - out["dra-driver-critical-pods-quota"] = _object( - "dra-driver-critical-pods-quota", - { - "apiVersion": "v1", - "kind": "ResourceQuota", - "metadata": {"name": "allow-critical-pods", "namespace": "nvidia-dra-driver"}, - "spec": { - "hard": {"pods": "1000"}, - "scopeSelector": { - "matchExpressions": [ - { - "operator": "In", - "scopeName": "PriorityClass", - "values": ["system-node-critical", "system-cluster-critical"], - }, - ], - }, - }, - }, + } + ), + ready=ready, ) - # --- the Dynamo half --- - out["grove"] = _release( - key="grove", - release="mp-grove-charts", - namespace="grove-system", - chart="grove-charts", - repository="oci://ghcr.io/ai-dynamo/grove", - version="v0.1.0-alpha.12-rc2", - ) - out["kai-scheduler"] = _release( - key="kai-scheduler", - release="mp-kai-scheduler", - namespace="kai-scheduler", - chart="kai-scheduler", - repository="oci://ghcr.io/kai-scheduler/kai-scheduler", - version="v0.16.8", - wait=True, - ) - out["kai-queue-root"] = _object("kai-queue-root", _kai_queue("modelplane-root", None)) - out["kai-queue"] = _object("kai-queue", _kai_queue("modelplane", "modelplane-root")) - for doc in _crds("modelexpress.yaml"): - key = f"modelexpress-crds-{doc['metadata']['name']}" - out[key] = _object(key, doc) - out["modelexpress-server-sa"] = _object( - "modelexpress-server-sa", - {"apiVersion": "v1", "kind": "ServiceAccount", "metadata": _MX_META}, - ) - out["modelexpress-server-role"] = _object( - "modelexpress-server-role", - { - "apiVersion": "rbac.authorization.k8s.io/v1", - "kind": "Role", - "metadata": _MX_META, - "rules": [ - { - "apiGroups": ["modelexpress.nvidia.com"], - "resources": ["modelmetadatas", "modelmetadatas/status"], - "verbs": ["get", "list", "create", "update", "patch", "delete"], - }, - { - "apiGroups": [""], - "resources": ["configmaps"], - "verbs": ["get", "list", "create", "update", "patch", "delete"], - }, - { - "apiGroups": ["modelexpress.nvidia.com"], - "resources": ["modelcacheentries", "modelcacheentries/status"], - "verbs": ["get", "list", "create", "update", "patch", "delete"], + +def _kube_prometheus_stack(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed kube-prometheus-stack Release.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-kube-prometheus-stack"}, + "labels": {"modelplane.ai/resource": "kube-prometheus-stack"}, }, - ], - }, - ) - out["modelexpress-server-rolebinding"] = _object( - "modelexpress-server-rolebinding", - { - "apiVersion": "rbac.authorization.k8s.io/v1", - "kind": "RoleBinding", - "metadata": _MX_META, - "subjects": [{"kind": "ServiceAccount", "name": "modelexpress-server", "namespace": "default"}], - "roleRef": {"apiGroup": "rbac.authorization.k8s.io", "kind": "Role", "name": "modelexpress-server"}, - }, - ) - out["modelexpress-server-svc"] = _object( - "modelexpress-server-svc", - { - "apiVersion": "v1", - "kind": "Service", - "metadata": _MX_META, - "spec": { - "selector": _MX_SELECT, - "ports": [{"name": "grpc", "port": 8001, "targetPort": 8001}], - }, - }, - ) - out["modelexpress-server"] = _object( - "modelexpress-server", - { - "apiVersion": "apps/v1", - "kind": "Deployment", - "metadata": _MX_META, - "spec": { - "replicas": 1, - "selector": {"matchLabels": _MX_SELECT}, - "template": { - "metadata": {"labels": _MX_SELECT}, - "spec": { - "serviceAccountName": "modelexpress-server", - "containers": [ - { - "name": "modelexpress-server", - "image": "nvcr.io/nvidia/ai-dynamo/modelexpress-server:0.4.1", - "ports": [{"containerPort": 8001}], - "env": [ - {"name": "MODEL_EXPRESS_CACHE_DIRECTORY", "value": "/mnt/models"}, - {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, - {"name": "MX_METADATA_BACKEND", "value": "kubernetes"}, - { - "name": "POD_NAMESPACE", - "valueFrom": {"fieldRef": {"fieldPath": "metadata.namespace"}}, - }, - ], - "volumeMounts": [{"name": "cache", "mountPath": "/mnt/models"}], - "readinessProbe": {"tcpSocket": {"port": 8001}, "periodSeconds": 10}, - "livenessProbe": {"tcpSocket": {"port": 8001}, "periodSeconds": 20}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "kube-prometheus-stack", + "repository": "https://prometheus-community.github.io/helm-charts", + "version": "84.4.0", + }, + "namespace": "monitoring", + "values": { + "fullnameOverride": "prometheus", + "prometheus": { + "prometheusSpec": { + "podMonitorSelectorNilUsesHelmValues": False, + "podMonitorNamespaceSelector": {}, + "additionalScrapeConfigs": [ + { + "job_name": "envoy-gateway-proxy", + "kubernetes_sd_configs": [ + { + "role": "pod", + "namespaces": {"names": ["envoy-gateway-system"]}, + } + ], + "relabel_configs": [ + { + "source_labels": [ + "__meta_kubernetes_pod_label_app_kubernetes_io_component" + ], + "action": "keep", + "regex": "proxy", + }, + { + "source_labels": ["__address__"], + "action": "replace", + "regex": "([^:]+)(?::\\d+)?", + "replacement": "$1:19001", + "target_label": "__address__", + }, + ], + "metrics_path": "/stats/prometheus", + } + ], + } }, - ], - "volumes": [{"name": "cache", "emptyDir": {}}], + "grafana": {"enabled": False}, + "alertmanager": {"enabled": False}, + }, }, }, - }, - }, - cel=_MODELEXPRESS_READY_CEL, + } + ), + ready=ready, ) - # --- the hand-rendered gateway pair --- - out["gateway-class"] = _object( - "gateway-class", - { - "apiVersion": "gateway.networking.k8s.io/v1", - "kind": "GatewayClass", - "metadata": {"name": "envoy"}, - "spec": { - "controllerName": "gateway.envoyproxy.io/gatewayclass-controller", - "parametersRef": { - "group": "gateway.envoyproxy.io", - "kind": "EnvoyProxy", - "name": "cluster-gateway", - "namespace": "modelplane-system", - }, - }, - }, - ) - out["gateway"] = _object( - "gateway", - { - "apiVersion": "gateway.networking.k8s.io/v1", - "kind": "Gateway", - "metadata": {"name": "cluster-gateway", "namespace": "modelplane-system"}, - "spec": { - "gatewayClassName": "envoy", - "listeners": [ - { - "name": "https", - "protocol": "HTTPS", - "port": 443, - "hostname": _GATEWAY_HOSTNAME, - "tls": {"mode": "Terminate", "certificateRefs": [{"name": "cluster-gateway-serving"}]}, - "allowedRoutes": { - "namespaces": { - "from": "Selector", - "selector": { - "matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}] - }, + +def _node_feature_discovery(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed node-feature-discovery Release.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-node-feature-discovery"}, + "labels": {"modelplane.ai/resource": "node-feature-discovery"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "node-feature-discovery", + "repository": "https://kubernetes-sigs.github.io/node-feature-discovery/charts", + "version": "0.19.0", + }, + "namespace": "node-feature-discovery", + "values": { + "worker": { + "tolerations": [{"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"}] } }, }, - ], - }, - }, - cel=_GATEWAY_READY_CEL, + }, + } + ), + ready=ready, ) - # --- the cluster gateway's PKI, issued for its hostname --- - out["gateway-ca-certificate"] = _object( - "gateway-ca-certificate", - { - "apiVersion": "cert-manager.io/v1", - "kind": "Certificate", - "metadata": {"name": "modelplane-cluster-ca", "namespace": "modelplane-system"}, - "spec": { - "isCA": True, - "commonName": f"modelplane cluster CA {_GATEWAY_HOSTNAME}", - "secretName": "modelplane-cluster-ca", - "duration": "87600h", - "renewBefore": "8760h", - "privateKey": {"algorithm": "ECDSA", "size": 256}, - "issuerRef": {"name": "modelplane-selfsigned", "kind": "Issuer", "group": "cert-manager.io"}, - }, - }, - cel=_CERTIFICATE_READY_CEL, - labeled=False, - ) - out["gateway-ca-issuer"] = _object( - "gateway-ca-issuer", - { - "apiVersion": "cert-manager.io/v1", - "kind": "Issuer", - "metadata": {"name": "modelplane-cluster-ca", "namespace": "modelplane-system"}, - "spec": {"ca": {"secretName": "modelplane-cluster-ca"}}, - }, - labeled=False, - ) - out["gateway-serving-certificate"] = _object( - "gateway-serving-certificate", - { - "apiVersion": "cert-manager.io/v1", - "kind": "Certificate", - "metadata": {"name": "cluster-gateway-serving", "namespace": "modelplane-system"}, - "spec": { - "secretName": "cluster-gateway-serving", - "dnsNames": [_GATEWAY_HOSTNAME], - "duration": "2160h", - "renewBefore": "720h", - "privateKey": {"algorithm": "ECDSA", "size": 256, "rotationPolicy": "Always"}, - "issuerRef": {"name": "modelplane-cluster-ca", "kind": "Issuer", "group": "cert-manager.io"}, - }, - }, - cel=_CERTIFICATE_READY_CEL, - labeled=False, - ) - out["gateway-ca-bundle"] = _object( - "gateway-ca-bundle", - { - "apiVersion": "trust.cert-manager.io/v1alpha1", - "kind": "Bundle", - "metadata": {"name": "modelplane-cluster-ca"}, - "spec": { - "sources": [{"secret": {"name": "modelplane-cluster-ca", "key": "ca.crt"}}], - "target": { - "configMap": {"key": "ca.crt"}, - "namespaceSelector": {"matchLabels": {"kubernetes.io/metadata.name": "modelplane-system"}}, - }, - }, - }, - cel=_BUNDLE_SYNCED_CEL, - labeled=False, - ) - out["gateway-ca-configmap"] = _object( - "gateway-ca-configmap", - { - "apiVersion": "v1", - "kind": "ConfigMap", - "metadata": {"name": "modelplane-cluster-ca", "namespace": "modelplane-system"}, - }, - labeled=False, - management_policies=["Observe"], - ) - out["gateway-client-ca-bundle"] = _object( - "gateway-client-ca-bundle", - { - "apiVersion": "v1", - "kind": "ConfigMap", - "metadata": {"name": "modelplane-inference-gateway-cas", "namespace": "modelplane-system"}, - "data": {"ca.crt": _CLIENT_CA}, - }, - labeled=False, - ) - out["gateway-client-auth"] = _object( - "gateway-client-auth", - { - "apiVersion": "gateway.envoyproxy.io/v1alpha1", - "kind": "ClientTrafficPolicy", - "metadata": {"name": "cluster-gateway-client-auth", "namespace": "modelplane-system"}, - "spec": { - "targetRefs": [ - { - "group": "gateway.networking.k8s.io", - "kind": "Gateway", - "name": "cluster-gateway", - "sectionName": "https", - } - ], - "tls": { - "clientValidation": { - "caCertificateRefs": [ - {"kind": "ConfigMap", "group": "", "name": "modelplane-inference-gateway-cas"} - ] - } - }, - }, - }, - cel=_POLICY_ACCEPTED_CEL, - labeled=False, - ) - # --- the gateway PKI trust anchor, common components on every cluster --- - out["gateway-selfsigned-issuer"] = _object( - "gateway-selfsigned-issuer", - { - "apiVersion": "cert-manager.io/v1", - "kind": "Issuer", - "metadata": {"name": "modelplane-selfsigned", "namespace": "modelplane-system"}, - "spec": {"selfSigned": {}}, - }, - ) - out["trust-manager"] = _release( - key="trust-manager", - release="mp-trust-manager", - namespace="modelplane-system", - chart="trust-manager", - repository="oci://quay.io/jetstack/charts", - version="v0.25.0", - values={ - "crds": {"enabled": True, "keep": True}, - "app": {"trust": {"namespace": "modelplane-system"}}, - "defaultPackage": {"enabled": False}, - }, +def _nvidia_dra_driver_gpu(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed NVIDIA GPU DRA driver Release.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-dra-driver-nvidia-gpu"}, + "labels": {"modelplane.ai/resource": "nvidia-dra-driver-gpu"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "dra-driver-nvidia-gpu", + "repository": "oci://registry.k8s.io/dra-driver-nvidia/charts", + "version": "0.4.1", + }, + "namespace": "nvidia-dra-driver", + "values": { + "gpuResourcesEnabledOverride": True, + "resources": {"computeDomains": {"enabled": False}}, + }, + }, + }, + } + ), + ready=ready, ) - return out - -def _response(resources: dict[str, fnv1.Resource], status: dict | None = None) -> fnv1.RunFunctionResponse: - """A whole expected response: 60s TTL, empty context, the XR status.""" - return fnv1.RunFunctionResponse( - meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), - desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct({"status": status if status is not None else {}})), - resources=resources, +def _ai_gateway_crds(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Envoy AI Gateway CRDs Release.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-ai-gateway-crds-helm"}, + "labels": {"modelplane.ai/resource": "ai-gateway-crds"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "ai-gateway-crds-helm", + "repository": "oci://docker.io/envoyproxy", + "version": "v1.1.0", + }, + "namespace": "envoy-ai-gateway-system", + "wait": True, + "waitTimeout": "10m", + }, + }, + } ), - context=structpb.Struct(), + ready=ready, ) -@dataclasses.dataclass -class Case: - name: str - req: fnv1.RunFunctionRequest - want: fnv1.RunFunctionResponse +def _gaie_crds_inferenceobjectives_x_k8s(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed x-k8s.io InferenceObjective CRD.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "labels": {"modelplane.ai/resource": "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io"} + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": _crd(filename="gaie.yaml", name="inferenceobjectives.inference.networking.x-k8s.io") + }, + }, + } + ), + ready=ready, + ) -def _compose_cases() -> list[Case]: - """The test_compose cases, built from the Existing/Dynamo stack's resources.""" - full = _provider_configs() | _EXISTING_DYNAMO_USAGES | _existing_dynamo_stack() - - # Second pass: PCs observed. depends_on gates first creation, so - # only the dependency-free wave renders; each dependent waits for - # its dependency's Ready before it is first created. - dep_gated = { - "envoy-gateway", # -> cert-manager - "ai-gateway", # -> ai-gateway-crds - "gateway-proxy", # -> gateway-namespace - "kai-queue-root", # -> kai-scheduler - "kai-queue", # -> kai-scheduler - "modelexpress-server", # -> modelexpress-crds - "gateway-selfsigned-issuer", # -> cert-manager, gateway-namespace - "trust-manager", # -> gateway-selfsigned-issuer - } - first_wave = {k: v for k, v in full.items() if k not in dep_gated} - - # Third pass: every rendered resource observed Ready (the gateway - # with its address assigned), so everything is marked ready and - # the address lands in the XR status. - rendered = [k for k in _existing_dynamo_stack() if k != "gateway"] - observed_ready = _observed_pcs() - for key in rendered: - observed_ready[key] = fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ) - observed_ready["gateway"] = fnv1.Resource( +def _gaie_crds_inferencepools_k8s(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed k8s.io InferencePool CRD.""" + return fnv1.Resource( resource=resource.dict_to_struct( { - "status": { - "conditions": [{"type": "Ready", "status": "True"}], - "atProvider": { - "manifest": {"status": {"addresses": [{"type": "IPAddress", "value": "203.0.113.7"}]}}, + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "labels": {"modelplane.ai/resource": "gaie-crds-inferencepools.inference.networking.k8s.io"} + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": _crd(filename="gaie.yaml", name="inferencepools.inference.networking.k8s.io") }, }, } - ) - ) - # Every component observed Ready; PCs and Usages are ready on arrival. - all_ready = copy.deepcopy(full) - for res in all_ready.values(): - res.ready = fnv1.READY_TRUE - - return [ - Case( - name="first pass composes only the provider configs and usages", - req=_request("Existing", "Dynamo"), - # Everything targeting the remote cluster is gated on the - # ProviderConfigs having been observed; Usages reference - # nothing remote and compose immediately. The unready - # ProviderConfigs keep the composite unready until the - # stack actually renders. - want=_response(_provider_configs(ready=False) | _EXISTING_DYNAMO_USAGES), - ), - Case( - name="second pass renders the dependency-free wave", - req=_request("Existing", "Dynamo", observed=_observed_pcs()), - want=_response(first_wave), - ), - Case( - name="all dependencies ready renders the whole stack, marks it ready, and writes the gateway address", - req=_request("Existing", "Dynamo", observed=observed_ready), - want=_response(all_ready, status={"gateway": {"address": "203.0.113.7"}}), ), - ] + ready=ready, + ) -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) +def _gaie_crds_inferencepools_x_k8s(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed x-k8s.io InferencePool CRD.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "labels": {"modelplane.ai/resource": "gaie-crds-inferencepools.inference.networking.x-k8s.io"} + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": _crd(filename="gaie.yaml", name="inferencepools.inference.networking.x-k8s.io") + }, + }, + } + ), + ready=ready, + ) -@pytest.mark.parametrize("case", _compose_cases(), ids=lambda case: case.name) -def test_compose(case: Case) -> None: - """RunFunction composes the Existing/Dynamo stack across the reconcile passes.""" - got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) - assert _to_dict(got) == _to_dict(case.want) +def _gateway_namespace(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed modelplane-system Namespace.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "gateway-namespace"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Namespace", + "metadata": { + "name": "modelplane-system", + "labels": {"modelplane.ai/namespace": "modelplane-system"}, + }, + } + }, + }, + } + ), + ready=ready, + ) -def test_identity_secret_type_flows_to_provider_configs() -> None: - """A non-GCP identity secret's type and namespace reach both ProviderConfigs verbatim.""" - # The type is stamped as is rather than being forced to - # GoogleApplicationCredentials, and the secret's own namespace wins over - # the XR's. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Nebius", - secrets=[ - v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), - v1alpha1.Secret( - type="NebiusServiceAccountCredentials", - name="nebius-secret", - key="credentials.json", - namespace="other-ns", - ), - ], - gateway=v1alpha1.Gateway(hostname=_GATEWAY_HOSTNAME), - ), - ).model_dump(exclude_none=True, mode="json") - ), - ), +def _gateway_selfsigned_issuer() -> fnv1.Resource: + """The composed self-signed Issuer the cluster CA roots in, marked ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "gateway-selfsigned-issuer"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "Issuer", + "metadata": {"name": "modelplane-selfsigned", "namespace": "modelplane-system"}, + "spec": {"selfSigned": {}}, + } + }, + }, + } ), + ready=fnv1.READY_TRUE, ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - pc = resource.struct_to_dict(got.desired.resources["provider-config-kubernetes"].resource) - assert pc["spec"]["identity"]["type"] == "NebiusServiceAccountCredentials" - assert pc["spec"]["identity"]["secretRef"]["namespace"] == "other-ns" - helm_pc = resource.struct_to_dict(got.desired.resources["provider-config-helm"].resource) - assert helm_pc["spec"]["identity"]["type"] == "NebiusServiceAccountCredentials" - - -def test_cluster_gateway_composes_mtls_with_ca() -> None: - """A cluster with an InferenceGateway CA composes its own PKI and serves mTLS.""" - # It issues its own PKI, republishes the CA without its key, demands a - # client certificate on its HTTPS listener, and publishes the CA in status. - # - # The hostname is a full Service FQDN, so the CA certificate's commonName - # overflows the 64-byte X.509 limit and is truncated. - hostname = "gateway-test-backend-12345.modelplane-system.svc.cluster.local" - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Existing", - stack="Standard", - secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], - gateway=v1alpha1.Gateway( - hostname=hostname, - # Deliberately out of name order, to prove the - # bundle sorts before concatenating. - clientCAs=[ - v1alpha1.ClientCA(name="fleet-b", certificate="BBB"), - v1alpha1.ClientCA(name="fleet-a", certificate="AAA"), - ], - ), - ), - ).model_dump(exclude_none=True, mode="json") - ), - ), - # PCs observed, the self-signed Issuer Ready (so trust-manager and - # the CA chain proceed), and the CA ConfigMap trust-manager syncs - # carrying the certificate back for status. - resources=_observed_pcs() - | { - "gateway-selfsigned-issuer": fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ), - "gateway-ca-configmap": fnv1.Resource( - resource=resource.dict_to_struct( - {"status": {"atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}}} - ) - ), - }, + + +def _trust_manager(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed trust-manager Release.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-trust-manager"}, + "labels": {"modelplane.ai/resource": "trust-manager"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": { + "name": "trust-manager", + "repository": "oci://quay.io/jetstack/charts", + "version": "v0.25.0", + }, + "namespace": "modelplane-system", + "values": { + "crds": {"enabled": True, "keep": True}, + "app": {"trust": {"namespace": "modelplane-system"}}, + "defaultPackage": {"enabled": False}, + }, + }, + }, + } ), - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - - def manifest(key: str) -> dict: - return resource.struct_to_dict(got.desired.resources[key].resource)["spec"]["forProvider"]["manifest"] - - ca_cert = manifest("gateway-ca-certificate") - assert ca_cert["spec"]["commonName"] == "modelplane cluster CA gateway-test-backend-12345.modelplane-syst" - assert len(ca_cert["spec"]["commonName"]) <= 64 - assert ca_cert["spec"]["isCA"] - assert ca_cert["spec"]["issuerRef"]["name"] == "modelplane-selfsigned" - - serving = manifest("gateway-serving-certificate") - assert serving["spec"]["dnsNames"] == [hostname] - assert serving["spec"]["issuerRef"]["name"] == "modelplane-cluster-ca" - - bundle = manifest("gateway-ca-bundle") - assert bundle["apiVersion"] == "trust.cert-manager.io/v1alpha1" - assert bundle["spec"]["sources"] == [{"secret": {"name": "modelplane-cluster-ca", "key": "ca.crt"}}] - - # Observed only, never managed: trust-manager owns the ConfigMap. - ca_cm = got.desired.resources["gateway-ca-configmap"] - assert resource.struct_to_dict(ca_cm.resource)["spec"]["managementPolicies"] == ["Observe"] - - # Every InferenceGateway's CA, sorted by name and concatenated. - client_bundle = manifest("gateway-client-ca-bundle") - assert client_bundle["data"]["ca.crt"] == "AAA\nBBB\n" - - client_auth = manifest("gateway-client-auth") - assert client_auth["kind"] == "ClientTrafficPolicy" - assert client_auth["spec"]["targetRefs"][0]["sectionName"] == "https" - assert ( - client_auth["spec"]["tls"]["clientValidation"]["caCertificateRefs"][0]["name"] - == "modelplane-inference-gateway-cas" + ready=ready, ) - # One HTTPS listener, terminating TLS with the serving certificate. - gateway = manifest("gateway") - assert gateway["spec"]["listeners"] == [ - { - "name": "https", - "protocol": "HTTPS", - "port": 443, - "hostname": hostname, - "tls": {"mode": "Terminate", "certificateRefs": [{"name": "cluster-gateway-serving"}]}, - "allowedRoutes": { - "namespaces": { - "from": "Selector", - "selector": {"matchExpressions": [{"key": "modelplane.ai/namespace", "operator": "Exists"}]}, - } - }, - } - ] - - status = resource.struct_to_dict(got.desired.composite.resource)["status"] - assert status["gateway"]["caCertificate"] == "CLUSTERCA" - - # Every PKI resource must be tracked for readiness: - # compose_gateway_pki marks only the keys it returns, so one composed - # but not returned would silently hold the cluster un-Ready. Observe - # each Ready and assert it's marked ready, which fails if the key was - # dropped from the rendered list. (The self-signed Issuer and - # trust-manager are stack components, covered by the golden test.) - pki_keys = [ - "gateway-ca-certificate", - "gateway-ca-issuer", - "gateway-serving-certificate", - "gateway-ca-bundle", - "gateway-ca-configmap", - "gateway-client-ca-bundle", - "gateway-client-auth", - ] - for key in pki_keys: - req.observed.resources[key].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ) - ) - # Preserve the CA ConfigMap's data alongside its Ready condition. - req.observed.resources["gateway-ca-configmap"].CopyFrom( - fnv1.Resource( - resource=resource.dict_to_struct( - { - "status": { - "conditions": [{"type": "Ready", "status": "True"}], - "atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}, - } - } - ) - ) - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - for key in pki_keys: - assert got.desired.resources[key].ready == fnv1.READY_TRUE, f"{key} not marked ready" - - -def test_cluster_gateway_without_ca_serves_nothing() -> None: - """A cluster with no InferenceGateway CA withholds its Gateway, and warns.""" - # Withholding the Gateway entirely, rather than serving the engines - # unauthenticated. - req = fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource( - resource=resource.dict_to_struct( - v1alpha1.ServingStack( - metadata=metav1.ObjectMeta(name="test-backend", namespace="test-ns"), - spec=v1alpha1.Spec( - cloud="Existing", - stack="Standard", - secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], - gateway=v1alpha1.Gateway(hostname="gw.clusters.example.com"), - ), - ).model_dump(exclude_none=True, mode="json") - ), - ), - resources=_observed_pcs(), + +def _dra_driver_critical_pods_quota(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed ResourceQuota admitting the DRA driver's critical pods.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "dra-driver-critical-pods-quota"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ResourceQuota", + "metadata": {"name": "allow-critical-pods", "namespace": "nvidia-dra-driver"}, + "spec": { + "hard": {"pods": "1000"}, + "scopeSelector": { + "matchExpressions": [ + { + "operator": "In", + "scopeName": "PriorityClass", + "values": ["system-node-critical", "system-cluster-critical"], + } + ] + }, + }, + } + }, + }, + } ), + ready=ready, ) - got = asyncio.run(fn.FunctionRunner().RunFunction(req, None)) - # The GatewayClass and the cluster's own PKI are composed, so the CA is - # ready to publish when the first InferenceGateway's CA arrives. The - # Gateway, the client CA bundle and the policy demanding a client - # certificate aren't, and nor is the Usage protecting the Gateway. - gateway_keys = {k for k in got.desired.resources if k.startswith(("gateway", "usage-gateway"))} - assert gateway_keys == { - "gateway-class", - "gateway-ca-certificate", - "gateway-ca-issuer", - "gateway-serving-certificate", - "gateway-ca-bundle", - "gateway-ca-configmap", - "gateway-namespace", - "usage-gateway-namespace-by-gateway-proxy", - "usage-gateway-namespace-by-gateway-selfsigned-issuer", - "usage-gateway-selfsigned-issuer-by-trust-manager", - } - assert list(got.results) == [ - fnv1.Result( - severity=fnv1.SEVERITY_WARNING, - message=( - "Gateway gw.clusters.example.com not served: no InferenceGateway has published a client " - "CA for this cluster to trust, and serving without one would accept unauthenticated callers" - ), - ) - ] + + +def _leader_worker_set() -> fnv1.Resource: + """The composed LeaderWorkerSet Release, not yet marked ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-lws"}, + "labels": {"modelplane.ai/resource": "leader-worker-set"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "chart": {"name": "lws", "repository": "oci://registry.k8s.io/lws/charts", "version": "v0.8.0"}, + "namespace": "lws-system", + }, + }, + } + ), + ready=fnv1.READY_UNSPECIFIED, + ) + + +def _gateway_class(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed envoy GatewayClass, parameterised by the gateway's EnvoyProxy.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "gateway-class"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "GatewayClass", + "metadata": {"name": "envoy"}, + "spec": { + "controllerName": "gateway.envoyproxy.io/gatewayclass-controller", + "parametersRef": { + "group": "gateway.envoyproxy.io", + "kind": "EnvoyProxy", + "name": "cluster-gateway", + "namespace": "modelplane-system", + }, + }, + } + }, + }, + } + ), + ready=ready, + ) + + +def _gateway(*, hostname: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Gateway: one HTTPS listener for hostname, terminating TLS with the serving certificate.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "gateway"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.networking.k8s.io/v1", + "kind": "Gateway", + "metadata": {"name": "cluster-gateway", "namespace": "modelplane-system"}, + "spec": { + "gatewayClassName": "envoy", + "listeners": [ + { + "name": "https", + "protocol": "HTTPS", + "port": 443, + "hostname": hostname, + "tls": { + "mode": "Terminate", + "certificateRefs": [{"name": "cluster-gateway-serving"}], + }, + "allowedRoutes": { + "namespaces": { + "from": "Selector", + "selector": { + "matchExpressions": [ + {"key": "modelplane.ai/namespace", "operator": "Exists"} + ] + }, + } + }, + } + ], + }, + } + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status.addresses) && object.status.addresses.size() > 0", + }, + }, + } + ), + ready=ready, + ) + + +def _gateway_ca_certificate(*, common_name: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed cluster CA Certificate, issued by the self-signed Issuer.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "Certificate", + "metadata": {"name": "modelplane-cluster-ca", "namespace": "modelplane-system"}, + "spec": { + "isCA": True, + "commonName": common_name, + "secretName": "modelplane-cluster-ca", + "duration": "87600h", + "renewBefore": "8760h", + "privateKey": {"algorithm": "ECDSA", "size": 256}, + "issuerRef": { + "name": "modelplane-selfsigned", + "kind": "Issuer", + "group": "cert-manager.io", + }, + }, + } + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status) && has(object.status.conditions) && object.status.conditions.exists(c, c.type == 'Ready' && c.status == 'True')", + }, + }, + } + ), + ready=ready, + ) + + +def _gateway_ca_issuer(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Issuer that signs with the cluster CA.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "Issuer", + "metadata": {"name": "modelplane-cluster-ca", "namespace": "modelplane-system"}, + "spec": {"ca": {"secretName": "modelplane-cluster-ca"}}, + } + }, + }, + } + ), + ready=ready, + ) + + +def _gateway_serving_certificate(*, hostname: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Certificate the gateway serves for hostname, issued by the cluster CA.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "cert-manager.io/v1", + "kind": "Certificate", + "metadata": {"name": "cluster-gateway-serving", "namespace": "modelplane-system"}, + "spec": { + "secretName": "cluster-gateway-serving", + "dnsNames": [hostname], + "duration": "2160h", + "renewBefore": "720h", + "privateKey": {"algorithm": "ECDSA", "size": 256, "rotationPolicy": "Always"}, + "issuerRef": { + "name": "modelplane-cluster-ca", + "kind": "Issuer", + "group": "cert-manager.io", + }, + }, + } + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status) && has(object.status.conditions) && object.status.conditions.exists(c, c.type == 'Ready' && c.status == 'True')", + }, + }, + } + ), + ready=ready, + ) + + +def _gateway_ca_bundle(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed trust-manager Bundle republishing the cluster CA's certificate without its key.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "trust.cert-manager.io/v1alpha1", + "kind": "Bundle", + "metadata": {"name": "modelplane-cluster-ca"}, + "spec": { + "sources": [{"secret": {"name": "modelplane-cluster-ca", "key": "ca.crt"}}], + "target": { + "configMap": {"key": "ca.crt"}, + "namespaceSelector": { + "matchLabels": {"kubernetes.io/metadata.name": "modelplane-system"} + }, + }, + }, + } + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status) && has(object.status.conditions) && object.status.conditions.exists(c, c.type == 'Synced' && c.status == 'True')", + }, + }, + } + ), + ready=ready, + ) + + +def _gateway_ca_configmap(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Object observing, never managing, the CA ConfigMap trust-manager owns.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": {"name": "modelplane-cluster-ca", "namespace": "modelplane-system"}, + } + }, + "managementPolicies": ["Observe"], + }, + } + ), + ready=ready, + ) + + +def _gateway_client_ca_bundle(*, ca_crt: str, ready: fnv1.Ready) -> fnv1.Resource: + """The composed ConfigMap holding ca_crt, the InferenceGateway CAs the gateway trusts.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ConfigMap", + "metadata": { + "name": "modelplane-inference-gateway-cas", + "namespace": "modelplane-system", + }, + "data": {"ca.crt": ca_crt}, + } + }, + }, + } + ), + ready=ready, + ) + + +def _gateway_client_auth(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed ClientTrafficPolicy demanding a client certificate on the HTTPS listener.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-backend-cluster-63fde"}, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "ClientTrafficPolicy", + "metadata": { + "name": "cluster-gateway-client-auth", + "namespace": "modelplane-system", + }, + "spec": { + "targetRefs": [ + { + "group": "gateway.networking.k8s.io", + "kind": "Gateway", + "name": "cluster-gateway", + "sectionName": "https", + } + ], + "tls": { + "clientValidation": { + "caCertificateRefs": [ + { + "kind": "ConfigMap", + "group": "", + "name": "modelplane-inference-gateway-cas", + } + ] + } + }, + }, + } + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": "has(object.status) && has(object.status.ancestors) && object.status.ancestors.exists(a, has(a.conditions) && a.conditions.exists(c, c.type == 'Accepted' && c.status == 'True'))", + }, + }, + } + ), + ready=ready, + ) + + +def _usage_cert_manager_by_envoy_gateway() -> fnv1.Resource: + """The composed Usage holding the cert-manager Release until the envoy-gateway Release is gone.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "spec": { + "of": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "cert-manager"}, + }, + }, + "by": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "envoy-gateway"}, + }, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _usage_ai_gateway_crds_by_ai_gateway() -> fnv1.Resource: + """The composed Usage holding the ai-gateway-crds Release until the ai-gateway Release is gone.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "spec": { + "of": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "ai-gateway-crds"}, + }, + }, + "by": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "ai-gateway"}, + }, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _usage_gateway_namespace_by_gateway_proxy() -> fnv1.Resource: + """The composed Usage holding the gateway-namespace Object until the gateway-proxy Object is gone.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "spec": { + "of": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "gateway-namespace"}, + }, + }, + "by": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "gateway-proxy"}, + }, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _usage_cert_manager_by_gateway_selfsigned_issuer() -> fnv1.Resource: + """The composed Usage holding the cert-manager Release until the gateway-selfsigned-issuer Object is gone.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "spec": { + "of": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "cert-manager"}, + }, + }, + "by": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "gateway-selfsigned-issuer"}, + }, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _usage_gateway_namespace_by_gateway_selfsigned_issuer() -> fnv1.Resource: + """The composed Usage holding the gateway-namespace Object until the gateway-selfsigned-issuer Object is gone.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "spec": { + "of": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "gateway-namespace"}, + }, + }, + "by": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "gateway-selfsigned-issuer"}, + }, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _usage_gateway_selfsigned_issuer_by_trust_manager() -> fnv1.Resource: + """The composed Usage holding the gateway-selfsigned-issuer Object until the trust-manager Release is gone.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "spec": { + "of": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "gateway-selfsigned-issuer"}, + }, + }, + "by": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "trust-manager"}, + }, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _usage_kai_scheduler_by_kai_queue_root() -> fnv1.Resource: + """The composed Usage holding the kai-scheduler Release until the kai-queue-root Object is gone.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "spec": { + "of": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "kai-scheduler"}, + }, + }, + "by": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "kai-queue-root"}, + }, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _usage_kai_scheduler_by_kai_queue() -> fnv1.Resource: + """The composed Usage holding the kai-scheduler Release until the kai-queue Object is gone.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "spec": { + "of": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "kai-scheduler"}, + }, + }, + "by": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "kai-queue"}, + }, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _usage_modelexpress_crds_modelmetadatas_by_modelexpress_server() -> fnv1.Resource: + """The composed Usage holding the ModelMetadata CRD Object until the modelexpress-server Object is gone.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "spec": { + "of": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": { + "modelplane.ai/resource": "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com" + }, + }, + }, + "by": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "modelexpress-server"}, + }, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _usage_modelexpress_crds_modelcacheentries_by_modelexpress_server() -> fnv1.Resource: + """The composed Usage holding the ModelCacheEntry CRD Object until the modelexpress-server Object is gone.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "spec": { + "of": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": { + "modelplane.ai/resource": "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com" + }, + }, + }, + "by": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "modelexpress-server"}, + }, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _usage_gateway_class_by_gateway() -> fnv1.Resource: + """The composed Usage holding the gateway-class Object until the gateway Object is gone.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "spec": { + "of": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "gateway-class"}, + }, + }, + "by": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "gateway"}, + }, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _usage_envoy_gateway_by_gateway_class() -> fnv1.Resource: + """The composed Usage holding the envoy-gateway Release until the gateway-class Object is gone.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "spec": { + "of": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "envoy-gateway"}, + }, + }, + "by": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/resource": "gateway-class"}, + }, + }, + "replayDeletion": True, + }, + } + ), + ready=fnv1.READY_TRUE, + ) + + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) + + +COMPOSE_CASES = [ + # Everything targeting the remote cluster is gated on the ProviderConfigs + # having been observed; Usages reference nothing remote and compose + # immediately. The unready ProviderConfigs keep the composite unready until + # the stack actually renders. + ComposeCase( + name="first pass composes only the provider configs and usages", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-helm": _helm_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_UNSPECIFIED, + ), + # One Usage per depends_on edge in the joined stack data, then the + # two hand-written edges of the gateway chain. + "usage-cert-manager-by-envoy-gateway": _usage_cert_manager_by_envoy_gateway(), + "usage-ai-gateway-crds-by-ai-gateway": _usage_ai_gateway_crds_by_ai_gateway(), + "usage-gateway-namespace-by-gateway-proxy": _usage_gateway_namespace_by_gateway_proxy(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage_cert_manager_by_gateway_selfsigned_issuer(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage_gateway_namespace_by_gateway_selfsigned_issuer(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage_gateway_selfsigned_issuer_by_trust_manager(), + "usage-kai-scheduler-by-kai-queue-root": _usage_kai_scheduler_by_kai_queue_root(), + "usage-kai-scheduler-by-kai-queue": _usage_kai_scheduler_by_kai_queue(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _usage_modelexpress_crds_modelmetadatas_by_modelexpress_server(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _usage_modelexpress_crds_modelcacheentries_by_modelexpress_server(), + "usage-gateway-class-by-gateway": _usage_gateway_class_by_gateway(), + "usage-envoy-gateway-by-gateway-class": _usage_envoy_gateway_by_gateway_class(), + }, + ), + context=structpb.Struct(), + ), + ), + # The ProviderConfigs are observed, which opens the gate on the rest of the + # stack. depends_on gates first creation too, so only the dependency-free + # wave renders. Each dependent waits for its dependencies' Ready before it's + # first created: envoy-gateway on cert-manager, ai-gateway on + # ai-gateway-crds, gateway-proxy on gateway-namespace, kai-queue-root and + # kai-queue on kai-scheduler, modelexpress-server on modelexpress-crds, + # gateway-selfsigned-issuer on cert-manager and gateway-namespace, and + # trust-manager on gateway-selfsigned-issuer. + ComposeCase( + name="second pass renders the dependency-free wave", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": _helm_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "cert-manager": _cert_manager(ready=fnv1.READY_UNSPECIFIED), + "kube-prometheus-stack": _kube_prometheus_stack(ready=fnv1.READY_UNSPECIFIED), + "node-feature-discovery": _node_feature_discovery(ready=fnv1.READY_UNSPECIFIED), + "nvidia-dra-driver-gpu": _nvidia_dra_driver_gpu(ready=fnv1.READY_UNSPECIFIED), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crds_inferenceobjectives_x_k8s( + ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crds_inferencepools_k8s( + ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crds_inferencepools_x_k8s( + ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "grove": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-grove-charts"}, + "labels": {"modelplane.ai/resource": "grove"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "grove-charts", + "repository": "oci://ghcr.io/ai-dynamo/grove", + "version": "v0.1.0-alpha.12-rc2", + }, + "namespace": "grove-system", + }, + }, + } + ) + ), + "kai-scheduler": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-kai-scheduler"}, + "labels": {"modelplane.ai/resource": "kai-scheduler"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "kai-scheduler", + "repository": "oci://ghcr.io/kai-scheduler/kai-scheduler", + "version": "v0.16.8", + }, + "namespace": "kai-scheduler", + "wait": True, + "waitTimeout": "10m", + }, + }, + } + ) + ), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "labels": { + "modelplane.ai/resource": "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com" + } + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": _crd( + filename="modelexpress.yaml", name="modelmetadatas.modelexpress.nvidia.com" + ) + }, + }, + } + ) + ), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "labels": { + "modelplane.ai/resource": "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com" + } + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": _crd( + filename="modelexpress.yaml", + name="modelcacheentries.modelexpress.nvidia.com", + ) + }, + }, + } + ) + ), + "modelexpress-server-sa": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-sa"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ServiceAccount", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + } + }, + }, + } + ) + ), + "modelexpress-server-role": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-role"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "Role", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + "rules": [ + { + "apiGroups": ["modelexpress.nvidia.com"], + "resources": ["modelmetadatas", "modelmetadatas/status"], + "verbs": ["get", "list", "create", "update", "patch", "delete"], + }, + { + "apiGroups": [""], + "resources": ["configmaps"], + "verbs": ["get", "list", "create", "update", "patch", "delete"], + }, + { + "apiGroups": ["modelexpress.nvidia.com"], + "resources": ["modelcacheentries", "modelcacheentries/status"], + "verbs": ["get", "list", "create", "update", "patch", "delete"], + }, + ], + } + }, + }, + } + ) + ), + "modelexpress-server-rolebinding": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-rolebinding"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "RoleBinding", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + "subjects": [ + { + "kind": "ServiceAccount", + "name": "modelexpress-server", + "namespace": "default", + } + ], + "roleRef": { + "apiGroup": "rbac.authorization.k8s.io", + "kind": "Role", + "name": "modelexpress-server", + }, + } + }, + }, + } + ) + ), + "modelexpress-server-svc": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-svc"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Service", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + "spec": { + "selector": {"modelplane.ai/modelexpress": "modelexpress-server"}, + "ports": [{"name": "grpc", "port": 8001, "targetPort": 8001}], + }, + } + }, + }, + } + ) + ), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway": _gateway(hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA test-backend.gateways.example.com", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_UNSPECIFIED), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="test-backend.gateways.example.com", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_UNSPECIFIED), + "gateway-client-ca-bundle": _gateway_client_ca_bundle( + ca_crt="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-client-auth": _gateway_client_auth(ready=fnv1.READY_UNSPECIFIED), + "usage-cert-manager-by-envoy-gateway": _usage_cert_manager_by_envoy_gateway(), + "usage-ai-gateway-crds-by-ai-gateway": _usage_ai_gateway_crds_by_ai_gateway(), + "usage-gateway-namespace-by-gateway-proxy": _usage_gateway_namespace_by_gateway_proxy(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage_cert_manager_by_gateway_selfsigned_issuer(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage_gateway_namespace_by_gateway_selfsigned_issuer(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage_gateway_selfsigned_issuer_by_trust_manager(), + "usage-kai-scheduler-by-kai-queue-root": _usage_kai_scheduler_by_kai_queue_root(), + "usage-kai-scheduler-by-kai-queue": _usage_kai_scheduler_by_kai_queue(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _usage_modelexpress_crds_modelmetadatas_by_modelexpress_server(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _usage_modelexpress_crds_modelcacheentries_by_modelexpress_server(), + "usage-gateway-class-by-gateway": _usage_gateway_class_by_gateway(), + "usage-envoy-gateway-by-gateway-class": _usage_envoy_gateway_by_gateway_class(), + }, + ), + context=structpb.Struct(), + ), + ), + # Every rendered resource is observed Ready, and the gateway has its address + # assigned. So everything is marked ready, and the address lands in the XR + # status. The ProviderConfigs are ready because they're observed, and the + # Usages on arrival. + ComposeCase( + name="all dependencies ready renders the whole stack, marks it ready, and writes the gateway address", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "envoy-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "ai-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "grove": _observed_ready(), + "kai-scheduler": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-queue": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "modelexpress-server": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway": fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "conditions": [{"type": "Ready", "status": "True"}], + "atProvider": { + "manifest": { + "status": {"addresses": [{"type": "IPAddress", "value": "203.0.113.7"}]} + } + }, + } + } + ) + ), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway={"address": "203.0.113.7"}), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "provider-config-helm": _helm_provider_config( + identity={ + "type": "GoogleApplicationCredentials", + "source": "Secret", + "secretRef": {"name": "sa-secret", "namespace": "test-ns", "key": "private_key"}, + }, + ready=fnv1.READY_TRUE, + ), + "cert-manager": _cert_manager(ready=fnv1.READY_TRUE), + "kube-prometheus-stack": _kube_prometheus_stack(ready=fnv1.READY_TRUE), + "node-feature-discovery": _node_feature_discovery(ready=fnv1.READY_TRUE), + "nvidia-dra-driver-gpu": _nvidia_dra_driver_gpu(ready=fnv1.READY_TRUE), + "envoy-gateway": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-gateway-helm"}, + "labels": {"modelplane.ai/resource": "envoy-gateway"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "gateway-helm", + "repository": "oci://docker.io/envoyproxy", + "version": "v1.8.4", + }, + "namespace": "envoy-gateway-system", + "values": { + "config": { + "envoyGateway": { + "extensionApis": {"enableBackend": True}, + "extensionManager": { + "hooks": { + "xdsTranslator": { + "translation": { + "listener": {"includeAll": True}, + "route": {"includeAll": True}, + "cluster": {"includeAll": True}, + "secret": {"includeAll": True}, + }, + "post": ["Translation", "Cluster", "Route"], + } + }, + "service": { + "fqdn": { + "hostname": "ai-gateway-controller.envoy-ai-gateway-system.svc.cluster.local", + "port": 1063, + } + }, + "backendResources": [ + { + "group": "inference.networking.k8s.io", + "kind": "InferencePool", + "version": "v1", + } + ], + }, + } + } + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_TRUE), + "ai-gateway": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-ai-gateway-helm"}, + "labels": {"modelplane.ai/resource": "ai-gateway"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "ai-gateway-helm", + "repository": "oci://docker.io/envoyproxy", + "version": "v1.1.0", + }, + "namespace": "envoy-ai-gateway-system", + "values": { + "controller": {"logRequestHeaderAttributes": "x-modelplane-caller:caller"} + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crds_inferenceobjectives_x_k8s( + ready=fnv1.READY_TRUE + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crds_inferencepools_k8s( + ready=fnv1.READY_TRUE + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crds_inferencepools_x_k8s( + ready=fnv1.READY_TRUE + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_TRUE), + "gateway-proxy": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "gateway-proxy"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "gateway.envoyproxy.io/v1alpha1", + "kind": "EnvoyProxy", + "metadata": {"name": "cluster-gateway", "namespace": "modelplane-system"}, + "spec": { + "provider": { + "type": "Kubernetes", + "kubernetes": { + "envoyService": {"externalTrafficPolicy": "Cluster"} + }, + } + }, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "gateway-selfsigned-issuer": _gateway_selfsigned_issuer(), + "trust-manager": _trust_manager(ready=fnv1.READY_TRUE), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_TRUE), + "grove": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-grove-charts"}, + "labels": {"modelplane.ai/resource": "grove"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "grove-charts", + "repository": "oci://ghcr.io/ai-dynamo/grove", + "version": "v0.1.0-alpha.12-rc2", + }, + "namespace": "grove-system", + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "kai-scheduler": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "annotations": {"crossplane.io/external-name": "mp-kai-scheduler"}, + "labels": {"modelplane.ai/resource": "kai-scheduler"}, + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "chart": { + "name": "kai-scheduler", + "repository": "oci://ghcr.io/kai-scheduler/kai-scheduler", + "version": "v0.16.8", + }, + "namespace": "kai-scheduler", + "wait": True, + "waitTimeout": "10m", + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "kai-queue-root": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "kai-queue-root"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "scheduling.run.ai/v2", + "kind": "Queue", + "metadata": {"name": "modelplane-root"}, + "spec": { + "resources": { + "cpu": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, + "gpu": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, + "memory": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, + } + }, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "kai-queue": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "kai-queue"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "scheduling.run.ai/v2", + "kind": "Queue", + "metadata": {"name": "modelplane"}, + "spec": { + "resources": { + "cpu": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, + "gpu": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, + "memory": {"quota": -1, "limit": -1, "overQuotaWeight": 1}, + }, + "parentQueue": "modelplane-root", + }, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "labels": { + "modelplane.ai/resource": "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com" + } + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": _crd( + filename="modelexpress.yaml", name="modelmetadatas.modelexpress.nvidia.com" + ) + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "labels": { + "modelplane.ai/resource": "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com" + } + }, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": _crd( + filename="modelexpress.yaml", + name="modelcacheentries.modelexpress.nvidia.com", + ) + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "modelexpress-server-sa": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-sa"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "ServiceAccount", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "modelexpress-server-role": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-role"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "Role", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + "rules": [ + { + "apiGroups": ["modelexpress.nvidia.com"], + "resources": ["modelmetadatas", "modelmetadatas/status"], + "verbs": ["get", "list", "create", "update", "patch", "delete"], + }, + { + "apiGroups": [""], + "resources": ["configmaps"], + "verbs": ["get", "list", "create", "update", "patch", "delete"], + }, + { + "apiGroups": ["modelexpress.nvidia.com"], + "resources": ["modelcacheentries", "modelcacheentries/status"], + "verbs": ["get", "list", "create", "update", "patch", "delete"], + }, + ], + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "modelexpress-server-rolebinding": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-rolebinding"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "rbac.authorization.k8s.io/v1", + "kind": "RoleBinding", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + "subjects": [ + { + "kind": "ServiceAccount", + "name": "modelexpress-server", + "namespace": "default", + } + ], + "roleRef": { + "apiGroup": "rbac.authorization.k8s.io", + "kind": "Role", + "name": "modelexpress-server", + }, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "modelexpress-server-svc": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server-svc"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "v1", + "kind": "Service", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + "spec": { + "selector": {"modelplane.ai/modelexpress": "modelexpress-server"}, + "ports": [{"name": "grpc", "port": 8001, "targetPort": 8001}], + }, + } + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "modelexpress-server": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"labels": {"modelplane.ai/resource": "modelexpress-server"}}, + "spec": { + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-backend-cluster-63fde", + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "Deployment", + "metadata": {"name": "modelexpress-server", "namespace": "default"}, + "spec": { + "replicas": 1, + "selector": { + "matchLabels": {"modelplane.ai/modelexpress": "modelexpress-server"} + }, + "template": { + "metadata": { + "labels": {"modelplane.ai/modelexpress": "modelexpress-server"} + }, + "spec": { + "serviceAccountName": "modelexpress-server", + "containers": [ + { + "name": "modelexpress-server", + "image": "nvcr.io/nvidia/ai-dynamo/modelexpress-server:0.4.1", + "ports": [{"containerPort": 8001}], + "env": [ + { + "name": "MODEL_EXPRESS_CACHE_DIRECTORY", + "value": "/mnt/models", + }, + {"name": "HF_HUB_CACHE", "value": "/mnt/models"}, + { + "name": "MX_METADATA_BACKEND", + "value": "kubernetes", + }, + { + "name": "POD_NAMESPACE", + "valueFrom": { + "fieldRef": { + "fieldPath": "metadata.namespace" + } + }, + }, + ], + "volumeMounts": [ + {"name": "cache", "mountPath": "/mnt/models"} + ], + "readinessProbe": { + "tcpSocket": {"port": 8001}, + "periodSeconds": 10, + }, + "livenessProbe": { + "tcpSocket": {"port": 8001}, + "periodSeconds": 20, + }, + } + ], + "volumes": [{"name": "cache", "emptyDir": {}}], + }, + }, + }, + } + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": 'has(object.status.conditions) && object.status.conditions.exists(c, c.type == "Available" && c.status == "True")', + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ), + "gateway-class": _gateway_class(ready=fnv1.READY_TRUE), + "gateway": _gateway(hostname="test-backend.gateways.example.com", ready=fnv1.READY_TRUE), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA test-backend.gateways.example.com", ready=fnv1.READY_TRUE + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_TRUE), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="test-backend.gateways.example.com", ready=fnv1.READY_TRUE + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_TRUE), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_TRUE), + "gateway-client-ca-bundle": _gateway_client_ca_bundle( + ca_crt="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", ready=fnv1.READY_TRUE + ), + "gateway-client-auth": _gateway_client_auth(ready=fnv1.READY_TRUE), + "usage-cert-manager-by-envoy-gateway": _usage_cert_manager_by_envoy_gateway(), + "usage-ai-gateway-crds-by-ai-gateway": _usage_ai_gateway_crds_by_ai_gateway(), + "usage-gateway-namespace-by-gateway-proxy": _usage_gateway_namespace_by_gateway_proxy(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage_cert_manager_by_gateway_selfsigned_issuer(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage_gateway_namespace_by_gateway_selfsigned_issuer(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage_gateway_selfsigned_issuer_by_trust_manager(), + "usage-kai-scheduler-by-kai-queue-root": _usage_kai_scheduler_by_kai_queue_root(), + "usage-kai-scheduler-by-kai-queue": _usage_kai_scheduler_by_kai_queue(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _usage_modelexpress_crds_modelmetadatas_by_modelexpress_server(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _usage_modelexpress_crds_modelcacheentries_by_modelexpress_server(), + "usage-gateway-class-by-gateway": _usage_gateway_class_by_gateway(), + "usage-envoy-gateway-by-gateway-class": _usage_envoy_gateway_by_gateway_class(), + }, + ), + context=structpb.Struct(), + ), + ), + # Both ProviderConfigs stamp the identity type as is rather than forcing + # GoogleApplicationCredentials, and the secret's own namespace wins over the + # XR's. + ComposeCase( + name="a non-GCP identity secret's type and namespace reach both ProviderConfigs verbatim", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Nebius", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret( + type="NebiusServiceAccountCredentials", + name="nebius-secret", + key="credentials.json", + namespace="other-ns", + ), + ], + gateway=v1alpha1.Gateway(hostname="test-backend.gateways.example.com"), + ), + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config( + identity={ + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": {"name": "nebius-secret", "namespace": "other-ns", "key": "credentials.json"}, + }, + ready=fnv1.READY_UNSPECIFIED, + ), + "provider-config-helm": _helm_provider_config( + identity={ + "type": "NebiusServiceAccountCredentials", + "source": "Secret", + "secretRef": {"name": "nebius-secret", "namespace": "other-ns", "key": "credentials.json"}, + }, + ready=fnv1.READY_UNSPECIFIED, + ), + "usage-cert-manager-by-envoy-gateway": _usage_cert_manager_by_envoy_gateway(), + "usage-ai-gateway-crds-by-ai-gateway": _usage_ai_gateway_crds_by_ai_gateway(), + "usage-gateway-namespace-by-gateway-proxy": _usage_gateway_namespace_by_gateway_proxy(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage_cert_manager_by_gateway_selfsigned_issuer(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage_gateway_namespace_by_gateway_selfsigned_issuer(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage_gateway_selfsigned_issuer_by_trust_manager(), + "usage-envoy-gateway-by-gateway-class": _usage_envoy_gateway_by_gateway_class(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="Gateway test-backend.gateways.example.com not served: no InferenceGateway has published a client CA for this cluster to trust, and serving without one would accept unauthenticated callers", + ), + ], + context=structpb.Struct(), + ), + ), + # The ProviderConfigs are observed, the self-signed Issuer trust-manager + # depends on is Ready, and the CA ConfigMap trust-manager syncs carries the + # certificate back for status. + ComposeCase( + name="a cluster with an InferenceGateway CA composes its own PKI and serves mTLS", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Standard", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway( + # A full Service FQDN, so the CA certificate's commonName overflows + # the 64-byte X.509 limit. + hostname="gateway-test-backend-12345.modelplane-system.svc.cluster.local", + # Deliberately out of name order, to prove the bundle sorts before concatenating. + clientCAs=[ + v1alpha1.ClientCA(name="fleet-b", certificate="BBB"), + v1alpha1.ClientCA(name="fleet-a", certificate="AAA"), + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + "gateway-selfsigned-issuer": _observed_ready(), + "gateway-ca-configmap": fnv1.Resource( + resource=resource.dict_to_struct( + {"status": {"atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}}} + ) + ), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + # The cluster's CA, published for InferenceGateways to trust. + composite=_desired_serving_stack(gateway={"caCertificate": "CLUSTERCA"}), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config(identity=None, ready=fnv1.READY_TRUE), + "provider-config-helm": _helm_provider_config(identity=None, ready=fnv1.READY_TRUE), + "cert-manager": _cert_manager(ready=fnv1.READY_UNSPECIFIED), + "kube-prometheus-stack": _kube_prometheus_stack(ready=fnv1.READY_UNSPECIFIED), + "node-feature-discovery": _node_feature_discovery(ready=fnv1.READY_UNSPECIFIED), + "nvidia-dra-driver-gpu": _nvidia_dra_driver_gpu(ready=fnv1.READY_UNSPECIFIED), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crds_inferenceobjectives_x_k8s( + ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crds_inferencepools_k8s( + ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crds_inferencepools_x_k8s( + ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "gateway-selfsigned-issuer": _gateway_selfsigned_issuer(), + "trust-manager": _trust_manager(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "leader-worker-set": _leader_worker_set(), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway": _gateway( + hostname="gateway-test-backend-12345.modelplane-system.svc.cluster.local", + ready=fnv1.READY_UNSPECIFIED, + ), + # The commonName is truncated to the 64-byte X.509 limit. + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA gateway-test-backend-12345.modelplane-syst", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_UNSPECIFIED), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="gateway-test-backend-12345.modelplane-system.svc.cluster.local", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_UNSPECIFIED), + # Every InferenceGateway's CA, sorted by name and concatenated. + "gateway-client-ca-bundle": _gateway_client_ca_bundle( + ca_crt="AAA\nBBB\n", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-client-auth": _gateway_client_auth(ready=fnv1.READY_UNSPECIFIED), + "usage-cert-manager-by-envoy-gateway": _usage_cert_manager_by_envoy_gateway(), + "usage-ai-gateway-crds-by-ai-gateway": _usage_ai_gateway_crds_by_ai_gateway(), + "usage-gateway-namespace-by-gateway-proxy": _usage_gateway_namespace_by_gateway_proxy(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage_cert_manager_by_gateway_selfsigned_issuer(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage_gateway_namespace_by_gateway_selfsigned_issuer(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage_gateway_selfsigned_issuer_by_trust_manager(), + "usage-gateway-class-by-gateway": _usage_gateway_class_by_gateway(), + "usage-envoy-gateway-by-gateway-class": _usage_envoy_gateway_by_gateway_class(), + }, + ), + context=structpb.Struct(), + ), + ), + # Every PKI resource must be tracked for readiness: mark_readiness marks only + # the keys compose_gateway_pki returns, so one composed but not returned + # would silently hold the cluster un-Ready. Each is observed Ready here, so a + # key dropped from the rendered list fails this case. + ComposeCase( + name="every gateway PKI resource observed Ready is marked ready", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Standard", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway( + hostname="gateway-test-backend-12345.modelplane-system.svc.cluster.local", + clientCAs=[ + v1alpha1.ClientCA(name="fleet-b", certificate="BBB"), + v1alpha1.ClientCA(name="fleet-a", certificate="AAA"), + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + "gateway-selfsigned-issuer": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + # The CA ConfigMap's data, alongside its Ready condition. + "gateway-ca-configmap": fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "conditions": [{"type": "Ready", "status": "True"}], + "atProvider": {"manifest": {"data": {"ca.crt": "CLUSTERCA"}}}, + } + } + ) + ), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway={"caCertificate": "CLUSTERCA"}), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config(identity=None, ready=fnv1.READY_TRUE), + "provider-config-helm": _helm_provider_config(identity=None, ready=fnv1.READY_TRUE), + "cert-manager": _cert_manager(ready=fnv1.READY_UNSPECIFIED), + "kube-prometheus-stack": _kube_prometheus_stack(ready=fnv1.READY_UNSPECIFIED), + "node-feature-discovery": _node_feature_discovery(ready=fnv1.READY_UNSPECIFIED), + "nvidia-dra-driver-gpu": _nvidia_dra_driver_gpu(ready=fnv1.READY_UNSPECIFIED), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crds_inferenceobjectives_x_k8s( + ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crds_inferencepools_k8s( + ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crds_inferencepools_x_k8s( + ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "gateway-selfsigned-issuer": _gateway_selfsigned_issuer(), + "trust-manager": _trust_manager(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "leader-worker-set": _leader_worker_set(), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway": _gateway( + hostname="gateway-test-backend-12345.modelplane-system.svc.cluster.local", + ready=fnv1.READY_UNSPECIFIED, + ), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA gateway-test-backend-12345.modelplane-syst", + ready=fnv1.READY_TRUE, + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_TRUE), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="gateway-test-backend-12345.modelplane-system.svc.cluster.local", ready=fnv1.READY_TRUE + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_TRUE), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_TRUE), + "gateway-client-ca-bundle": _gateway_client_ca_bundle(ca_crt="AAA\nBBB\n", ready=fnv1.READY_TRUE), + "gateway-client-auth": _gateway_client_auth(ready=fnv1.READY_TRUE), + "usage-cert-manager-by-envoy-gateway": _usage_cert_manager_by_envoy_gateway(), + "usage-ai-gateway-crds-by-ai-gateway": _usage_ai_gateway_crds_by_ai_gateway(), + "usage-gateway-namespace-by-gateway-proxy": _usage_gateway_namespace_by_gateway_proxy(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage_cert_manager_by_gateway_selfsigned_issuer(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage_gateway_namespace_by_gateway_selfsigned_issuer(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage_gateway_selfsigned_issuer_by_trust_manager(), + "usage-gateway-class-by-gateway": _usage_gateway_class_by_gateway(), + "usage-envoy-gateway-by-gateway-class": _usage_envoy_gateway_by_gateway_class(), + }, + ), + context=structpb.Struct(), + ), + ), + # The GatewayClass and the cluster's own PKI are composed, so the CA is ready + # to publish when the first InferenceGateway's CA arrives. The Gateway, the + # client CA bundle and the policy demanding a client certificate aren't, and + # nor is the Usage holding the GatewayClass for the Gateway. + ComposeCase( + name="a cluster with no InferenceGateway CA withholds its Gateway and warns", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Standard", + secrets=[v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig")], + gateway=v1alpha1.Gateway(hostname="gw.clusters.example.com"), + ), + resources={ + "provider-config-kubernetes": _observed_kubernetes_provider_config(), + "provider-config-helm": _observed_helm_provider_config(), + }, + ), + ), + want=fnv1.RunFunctionResponse( + meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), + desired=fnv1.State( + composite=_desired_serving_stack(gateway=None), + resources={ + "provider-config-kubernetes": _kubernetes_provider_config(identity=None, ready=fnv1.READY_TRUE), + "provider-config-helm": _helm_provider_config(identity=None, ready=fnv1.READY_TRUE), + "cert-manager": _cert_manager(ready=fnv1.READY_UNSPECIFIED), + "kube-prometheus-stack": _kube_prometheus_stack(ready=fnv1.READY_UNSPECIFIED), + "node-feature-discovery": _node_feature_discovery(ready=fnv1.READY_UNSPECIFIED), + "nvidia-dra-driver-gpu": _nvidia_dra_driver_gpu(ready=fnv1.READY_UNSPECIFIED), + "ai-gateway-crds": _ai_gateway_crds(ready=fnv1.READY_UNSPECIFIED), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _gaie_crds_inferenceobjectives_x_k8s( + ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.k8s.io": _gaie_crds_inferencepools_k8s( + ready=fnv1.READY_UNSPECIFIED + ), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _gaie_crds_inferencepools_x_k8s( + ready=fnv1.READY_UNSPECIFIED + ), + "gateway-namespace": _gateway_namespace(ready=fnv1.READY_UNSPECIFIED), + "dra-driver-critical-pods-quota": _dra_driver_critical_pods_quota(ready=fnv1.READY_UNSPECIFIED), + "leader-worker-set": _leader_worker_set(), + "gateway-class": _gateway_class(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-certificate": _gateway_ca_certificate( + common_name="modelplane cluster CA gw.clusters.example.com", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-ca-issuer": _gateway_ca_issuer(ready=fnv1.READY_UNSPECIFIED), + "gateway-serving-certificate": _gateway_serving_certificate( + hostname="gw.clusters.example.com", ready=fnv1.READY_UNSPECIFIED + ), + "gateway-ca-bundle": _gateway_ca_bundle(ready=fnv1.READY_UNSPECIFIED), + "gateway-ca-configmap": _gateway_ca_configmap(ready=fnv1.READY_UNSPECIFIED), + "usage-cert-manager-by-envoy-gateway": _usage_cert_manager_by_envoy_gateway(), + "usage-ai-gateway-crds-by-ai-gateway": _usage_ai_gateway_crds_by_ai_gateway(), + "usage-gateway-namespace-by-gateway-proxy": _usage_gateway_namespace_by_gateway_proxy(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _usage_cert_manager_by_gateway_selfsigned_issuer(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _usage_gateway_namespace_by_gateway_selfsigned_issuer(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _usage_gateway_selfsigned_issuer_by_trust_manager(), + "usage-envoy-gateway-by-gateway-class": _usage_envoy_gateway_by_gateway_class(), + }, + ), + results=[ + fnv1.Result( + severity=fnv1.SEVERITY_WARNING, + message="Gateway gw.clusters.example.com not served: no InferenceGateway has published a client CA for this cluster to trust, and serving without one would accept unauthenticated callers", + ), + ], + context=structpb.Struct(), + ), + ), +] + + +@pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) +def test_compose(case: ComposeCase) -> None: + """RunFunction composes the serving stack, gated on what's observed.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert _to_dict(got) == _to_dict(case.want) # The composed-resource key a component renders under is its identity: @@ -1162,159 +2682,1646 @@ def test_cluster_gateway_without_ca_serves_nothing() -> None: # reviewed literals. A failure here means the stack data changed a key - # make sure that's intended, then update the inventory and the release # notes. +# +# Each case's XR has an InferenceGateway CA to trust, so the Gateway and its +# client-auth policy are included. Every key the case expects is observed +# Ready, so the depends_on install gate opens and the full stack renders; a +# key the function doesn't render still fails the comparison. The cases +# compare keys rather than whole responses because every cloud's rendered +# half would restate its stack data, much of it generated, where +# COMPOSE_CASES already covers how components render. +COMPOSED_RESOURCE_KEYS_CASES = [ + ComposedResourceKeysCase( + name="EKS Standard", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="EKS", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "gpu-operator": _observed_ready(), + "k8s-ephemeral-storage-metrics": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nodewright-operator": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "nvsentinel": _observed_ready(), + "prometheus-adapter": _observed_ready(), + "prometheus-operator-crds": _observed_ready(), + "usage-cert-manager-by-gpu-operator": _observed_ready(), + "usage-cert-manager-by-nvsentinel": _observed_ready(), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _observed_ready(), + "usage-gpu-operator-by-nvsentinel": _observed_ready(), + "usage-kube-prometheus-stack-by-gpu-operator": _observed_ready(), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-kube-prometheus-stack-by-prometheus-adapter": _observed_ready(), + "usage-node-feature-discovery-by-gpu-operator": _observed_ready(), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _observed_ready(), + "usage-prometheus-operator-crds-by-nvsentinel": _observed_ready(), + "leader-worker-set": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The EKS half, generated from aicr. + "cert-manager", + "gpu-operator", + "k8s-ephemeral-storage-metrics", + "kube-prometheus-stack", + "node-feature-discovery", + "nodewright-operator", + "nvidia-dra-driver-gpu", + "nvsentinel", + "prometheus-adapter", + "prometheus-operator-crds", + "usage-cert-manager-by-gpu-operator", + "usage-cert-manager-by-nvsentinel", + "usage-gpu-operator-by-nvidia-dra-driver-gpu", + "usage-gpu-operator-by-nvsentinel", + "usage-kube-prometheus-stack-by-gpu-operator", + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics", + "usage-kube-prometheus-stack-by-prometheus-adapter", + "usage-node-feature-discovery-by-gpu-operator", + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics", + "usage-prometheus-operator-crds-by-kube-prometheus-stack", + "usage-prometheus-operator-crds-by-nvsentinel", + # The Standard stack. + "leader-worker-set", + }, + ), + ComposedResourceKeysCase( + name="EKS Dynamo", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="EKS", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "gpu-operator": _observed_ready(), + "k8s-ephemeral-storage-metrics": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nodewright-operator": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "nvsentinel": _observed_ready(), + "prometheus-adapter": _observed_ready(), + "prometheus-operator-crds": _observed_ready(), + "usage-cert-manager-by-gpu-operator": _observed_ready(), + "usage-cert-manager-by-nvsentinel": _observed_ready(), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _observed_ready(), + "usage-gpu-operator-by-nvsentinel": _observed_ready(), + "usage-kube-prometheus-stack-by-gpu-operator": _observed_ready(), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-kube-prometheus-stack-by-prometheus-adapter": _observed_ready(), + "usage-node-feature-discovery-by-gpu-operator": _observed_ready(), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _observed_ready(), + "usage-prometheus-operator-crds-by-nvsentinel": _observed_ready(), + "grove": _observed_ready(), + "kai-queue": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-scheduler": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "usage-kai-scheduler-by-kai-queue": _observed_ready(), + "usage-kai-scheduler-by-kai-queue-root": _observed_ready(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The EKS half, generated from aicr. + "cert-manager", + "gpu-operator", + "k8s-ephemeral-storage-metrics", + "kube-prometheus-stack", + "node-feature-discovery", + "nodewright-operator", + "nvidia-dra-driver-gpu", + "nvsentinel", + "prometheus-adapter", + "prometheus-operator-crds", + "usage-cert-manager-by-gpu-operator", + "usage-cert-manager-by-nvsentinel", + "usage-gpu-operator-by-nvidia-dra-driver-gpu", + "usage-gpu-operator-by-nvsentinel", + "usage-kube-prometheus-stack-by-gpu-operator", + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics", + "usage-kube-prometheus-stack-by-prometheus-adapter", + "usage-node-feature-discovery-by-gpu-operator", + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics", + "usage-prometheus-operator-crds-by-kube-prometheus-stack", + "usage-prometheus-operator-crds-by-nvsentinel", + # The Dynamo stack. + "grove", + "kai-queue", + "kai-queue-root", + "kai-scheduler", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-server", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-sa", + "modelexpress-server-svc", + "usage-kai-scheduler-by-kai-queue", + "usage-kai-scheduler-by-kai-queue-root", + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", + }, + ), + ComposedResourceKeysCase( + name="AKS Standard", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="AKS", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "gpu-operator": _observed_ready(), + "k8s-ephemeral-storage-metrics": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nodewright-operator": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "nvsentinel": _observed_ready(), + "prometheus-adapter": _observed_ready(), + "prometheus-operator-crds": _observed_ready(), + "usage-cert-manager-by-gpu-operator": _observed_ready(), + "usage-cert-manager-by-nvsentinel": _observed_ready(), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _observed_ready(), + "usage-gpu-operator-by-nvsentinel": _observed_ready(), + "usage-kube-prometheus-stack-by-gpu-operator": _observed_ready(), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-kube-prometheus-stack-by-prometheus-adapter": _observed_ready(), + "usage-node-feature-discovery-by-gpu-operator": _observed_ready(), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _observed_ready(), + "usage-prometheus-operator-crds-by-nvsentinel": _observed_ready(), + "gpu-operator-manifests": _observed_ready(), + "usage-gpu-operator-by-gpu-operator-manifests": _observed_ready(), + "leader-worker-set": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The AKS half, generated from aicr. + "cert-manager", + "gpu-operator", + "k8s-ephemeral-storage-metrics", + "kube-prometheus-stack", + "node-feature-discovery", + "nodewright-operator", + "nvidia-dra-driver-gpu", + "nvsentinel", + "prometheus-adapter", + "prometheus-operator-crds", + "usage-cert-manager-by-gpu-operator", + "usage-cert-manager-by-nvsentinel", + "usage-gpu-operator-by-nvidia-dra-driver-gpu", + "usage-gpu-operator-by-nvsentinel", + "usage-kube-prometheus-stack-by-gpu-operator", + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics", + "usage-kube-prometheus-stack-by-prometheus-adapter", + "usage-node-feature-discovery-by-gpu-operator", + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics", + "usage-prometheus-operator-crds-by-kube-prometheus-stack", + "usage-prometheus-operator-crds-by-nvsentinel", + # AKS additionally carries the gpu-operator's toolkit-hardening manifest. + "gpu-operator-manifests", + "usage-gpu-operator-by-gpu-operator-manifests", + # The Standard stack. + "leader-worker-set", + }, + ), + ComposedResourceKeysCase( + name="AKS Dynamo", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="AKS", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "gpu-operator": _observed_ready(), + "k8s-ephemeral-storage-metrics": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nodewright-operator": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "nvsentinel": _observed_ready(), + "prometheus-adapter": _observed_ready(), + "prometheus-operator-crds": _observed_ready(), + "usage-cert-manager-by-gpu-operator": _observed_ready(), + "usage-cert-manager-by-nvsentinel": _observed_ready(), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _observed_ready(), + "usage-gpu-operator-by-nvsentinel": _observed_ready(), + "usage-kube-prometheus-stack-by-gpu-operator": _observed_ready(), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-kube-prometheus-stack-by-prometheus-adapter": _observed_ready(), + "usage-node-feature-discovery-by-gpu-operator": _observed_ready(), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _observed_ready(), + "usage-prometheus-operator-crds-by-nvsentinel": _observed_ready(), + "gpu-operator-manifests": _observed_ready(), + "usage-gpu-operator-by-gpu-operator-manifests": _observed_ready(), + "grove": _observed_ready(), + "kai-queue": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-scheduler": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "usage-kai-scheduler-by-kai-queue": _observed_ready(), + "usage-kai-scheduler-by-kai-queue-root": _observed_ready(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The AKS half, generated from aicr. + "cert-manager", + "gpu-operator", + "k8s-ephemeral-storage-metrics", + "kube-prometheus-stack", + "node-feature-discovery", + "nodewright-operator", + "nvidia-dra-driver-gpu", + "nvsentinel", + "prometheus-adapter", + "prometheus-operator-crds", + "usage-cert-manager-by-gpu-operator", + "usage-cert-manager-by-nvsentinel", + "usage-gpu-operator-by-nvidia-dra-driver-gpu", + "usage-gpu-operator-by-nvsentinel", + "usage-kube-prometheus-stack-by-gpu-operator", + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics", + "usage-kube-prometheus-stack-by-prometheus-adapter", + "usage-node-feature-discovery-by-gpu-operator", + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics", + "usage-prometheus-operator-crds-by-kube-prometheus-stack", + "usage-prometheus-operator-crds-by-nvsentinel", + # AKS additionally carries the gpu-operator's toolkit-hardening manifest. + "gpu-operator-manifests", + "usage-gpu-operator-by-gpu-operator-manifests", + # The Dynamo stack. + "grove", + "kai-queue", + "kai-queue-root", + "kai-scheduler", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-server", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-sa", + "modelexpress-server-svc", + "usage-kai-scheduler-by-kai-queue", + "usage-kai-scheduler-by-kai-queue-root", + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", + }, + ), + ComposedResourceKeysCase( + name="GKE Standard", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="GKE", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "gpu-operator": _observed_ready(), + "k8s-ephemeral-storage-metrics": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nodewright-operator": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "nvsentinel": _observed_ready(), + "prometheus-adapter": _observed_ready(), + "prometheus-operator-crds": _observed_ready(), + "usage-cert-manager-by-gpu-operator": _observed_ready(), + "usage-cert-manager-by-nvsentinel": _observed_ready(), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _observed_ready(), + "usage-gpu-operator-by-nvsentinel": _observed_ready(), + "usage-kube-prometheus-stack-by-gpu-operator": _observed_ready(), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-kube-prometheus-stack-by-prometheus-adapter": _observed_ready(), + "usage-node-feature-discovery-by-gpu-operator": _observed_ready(), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _observed_ready(), + "usage-prometheus-operator-crds-by-nvsentinel": _observed_ready(), + "gpu-operator-pre-manifests-aicr-gke-critical-pods": _observed_ready(), + "gpu-operator-pre-manifests-gpu-operator": _observed_ready(), + "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator": _observed_ready(), + "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator": _observed_ready(), + "leader-worker-set": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The GKE half, generated from aicr. + "cert-manager", + "gpu-operator", + "k8s-ephemeral-storage-metrics", + "kube-prometheus-stack", + "node-feature-discovery", + "nodewright-operator", + "nvidia-dra-driver-gpu", + "nvsentinel", + "prometheus-adapter", + "prometheus-operator-crds", + "usage-cert-manager-by-gpu-operator", + "usage-cert-manager-by-nvsentinel", + "usage-gpu-operator-by-nvidia-dra-driver-gpu", + "usage-gpu-operator-by-nvsentinel", + "usage-kube-prometheus-stack-by-gpu-operator", + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics", + "usage-kube-prometheus-stack-by-prometheus-adapter", + "usage-node-feature-discovery-by-gpu-operator", + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics", + "usage-prometheus-operator-crds-by-kube-prometheus-stack", + "usage-prometheus-operator-crds-by-nvsentinel", + # GKE additionally carries the critical-pods ResourceQuota aicr's + # bundler synthesizes as a gpu-operator pre-manifest (GKE rejects + # system-node-critical pods in a namespace without one; aicr#915). + "gpu-operator-pre-manifests-aicr-gke-critical-pods", + "gpu-operator-pre-manifests-gpu-operator", + "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator", + "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator", + # The Standard stack. + "leader-worker-set", + }, + ), + ComposedResourceKeysCase( + name="GKE Dynamo", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="GKE", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "gpu-operator": _observed_ready(), + "k8s-ephemeral-storage-metrics": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nodewright-operator": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "nvsentinel": _observed_ready(), + "prometheus-adapter": _observed_ready(), + "prometheus-operator-crds": _observed_ready(), + "usage-cert-manager-by-gpu-operator": _observed_ready(), + "usage-cert-manager-by-nvsentinel": _observed_ready(), + "usage-gpu-operator-by-nvidia-dra-driver-gpu": _observed_ready(), + "usage-gpu-operator-by-nvsentinel": _observed_ready(), + "usage-kube-prometheus-stack-by-gpu-operator": _observed_ready(), + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-kube-prometheus-stack-by-prometheus-adapter": _observed_ready(), + "usage-node-feature-discovery-by-gpu-operator": _observed_ready(), + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics": _observed_ready(), + "usage-prometheus-operator-crds-by-kube-prometheus-stack": _observed_ready(), + "usage-prometheus-operator-crds-by-nvsentinel": _observed_ready(), + "gpu-operator-pre-manifests-aicr-gke-critical-pods": _observed_ready(), + "gpu-operator-pre-manifests-gpu-operator": _observed_ready(), + "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator": _observed_ready(), + "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator": _observed_ready(), + "grove": _observed_ready(), + "kai-queue": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-scheduler": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "usage-kai-scheduler-by-kai-queue": _observed_ready(), + "usage-kai-scheduler-by-kai-queue-root": _observed_ready(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The GKE half, generated from aicr. + "cert-manager", + "gpu-operator", + "k8s-ephemeral-storage-metrics", + "kube-prometheus-stack", + "node-feature-discovery", + "nodewright-operator", + "nvidia-dra-driver-gpu", + "nvsentinel", + "prometheus-adapter", + "prometheus-operator-crds", + "usage-cert-manager-by-gpu-operator", + "usage-cert-manager-by-nvsentinel", + "usage-gpu-operator-by-nvidia-dra-driver-gpu", + "usage-gpu-operator-by-nvsentinel", + "usage-kube-prometheus-stack-by-gpu-operator", + "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics", + "usage-kube-prometheus-stack-by-prometheus-adapter", + "usage-node-feature-discovery-by-gpu-operator", + "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics", + "usage-prometheus-operator-crds-by-kube-prometheus-stack", + "usage-prometheus-operator-crds-by-nvsentinel", + # GKE additionally carries the critical-pods ResourceQuota aicr's + # bundler synthesizes as a gpu-operator pre-manifest (GKE rejects + # system-node-critical pods in a namespace without one; aicr#915). + "gpu-operator-pre-manifests-aicr-gke-critical-pods", + "gpu-operator-pre-manifests-gpu-operator", + "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator", + "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator", + # The Dynamo stack. + "grove", + "kai-queue", + "kai-queue-root", + "kai-scheduler", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-server", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-sa", + "modelexpress-server-svc", + "usage-kai-scheduler-by-kai-queue", + "usage-kai-scheduler-by-kai-queue-root", + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", + }, + ), + ComposedResourceKeysCase( + name="Nebius Standard", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Nebius", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "leader-worker-set": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The hand-written Nebius half. + "cert-manager", + "kube-prometheus-stack", + "node-feature-discovery", + "nvidia-dra-driver-gpu", + # The Standard stack. + "leader-worker-set", + }, + ), + ComposedResourceKeysCase( + name="Nebius Dynamo", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Nebius", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "grove": _observed_ready(), + "kai-queue": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-scheduler": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "usage-kai-scheduler-by-kai-queue": _observed_ready(), + "usage-kai-scheduler-by-kai-queue-root": _observed_ready(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The hand-written Nebius half. + "cert-manager", + "kube-prometheus-stack", + "node-feature-discovery", + "nvidia-dra-driver-gpu", + # The Dynamo stack. + "grove", + "kai-queue", + "kai-queue-root", + "kai-scheduler", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-server", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-sa", + "modelexpress-server-svc", + "usage-kai-scheduler-by-kai-queue", + "usage-kai-scheduler-by-kai-queue-root", + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", + }, + ), + ComposedResourceKeysCase( + name="Vultr Standard", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Vultr", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "leader-worker-set": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The hand-written Vultr half. VKE pre-installs NFD via its managed GPU + # Operator add-on, so it carries no node-feature-discovery of its own. + "cert-manager", + "kube-prometheus-stack", + "nvidia-dra-driver-gpu", + # The Standard stack. + "leader-worker-set", + }, + ), + ComposedResourceKeysCase( + name="Vultr Dynamo", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Vultr", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "grove": _observed_ready(), + "kai-queue": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-scheduler": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "usage-kai-scheduler-by-kai-queue": _observed_ready(), + "usage-kai-scheduler-by-kai-queue-root": _observed_ready(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The hand-written Vultr half. VKE pre-installs NFD via its managed GPU + # Operator add-on, so it carries no node-feature-discovery of its own. + "cert-manager", + "kube-prometheus-stack", + "nvidia-dra-driver-gpu", + # The Dynamo stack. + "grove", + "kai-queue", + "kai-queue-root", + "kai-scheduler", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-server", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-sa", + "modelexpress-server-svc", + "usage-kai-scheduler-by-kai-queue", + "usage-kai-scheduler-by-kai-queue-root", + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", + }, + ), + ComposedResourceKeysCase( + name="Existing Standard", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Standard", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "leader-worker-set": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The hand-written Existing half. + "cert-manager", + "kube-prometheus-stack", + "node-feature-discovery", + "nvidia-dra-driver-gpu", + # The Standard stack. + "leader-worker-set", + }, + ), + ComposedResourceKeysCase( + name="Existing Dynamo", + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_serving_stack( + cloud="Existing", + stack="Dynamo", + secrets=[ + v1alpha1.Secret(type="Kubeconfig", name="kube-secret", key="kubeconfig"), + v1alpha1.Secret(type="GoogleApplicationCredentials", name="sa-secret", key="private_key"), + ], + gateway=v1alpha1.Gateway( + hostname="test-backend.gateways.example.com", + clientCAs=[ + v1alpha1.ClientCA( + name="eu", + certificate="-----BEGIN CERTIFICATE-----\nfleet\n-----END CERTIFICATE-----\n", + ) + ], + ), + ), + resources={ + "provider-config-kubernetes": _observed_ready(), + "provider-config-helm": _observed_ready(), + "gateway": _observed_ready(), + "gateway-class": _observed_ready(), + "gateway-ca-certificate": _observed_ready(), + "gateway-ca-issuer": _observed_ready(), + "gateway-serving-certificate": _observed_ready(), + "gateway-ca-bundle": _observed_ready(), + "gateway-ca-configmap": _observed_ready(), + "gateway-client-ca-bundle": _observed_ready(), + "gateway-client-auth": _observed_ready(), + "usage-gateway-class-by-gateway": _observed_ready(), + "usage-envoy-gateway-by-gateway-class": _observed_ready(), + "ai-gateway": _observed_ready(), + "ai-gateway-crds": _observed_ready(), + "dra-driver-critical-pods-quota": _observed_ready(), + "envoy-gateway": _observed_ready(), + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.k8s.io": _observed_ready(), + "gaie-crds-inferencepools.inference.networking.x-k8s.io": _observed_ready(), + "gateway-namespace": _observed_ready(), + "gateway-proxy": _observed_ready(), + "gateway-selfsigned-issuer": _observed_ready(), + "trust-manager": _observed_ready(), + "usage-ai-gateway-crds-by-ai-gateway": _observed_ready(), + "usage-cert-manager-by-envoy-gateway": _observed_ready(), + "usage-cert-manager-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-namespace-by-gateway-proxy": _observed_ready(), + "usage-gateway-namespace-by-gateway-selfsigned-issuer": _observed_ready(), + "usage-gateway-selfsigned-issuer-by-trust-manager": _observed_ready(), + "cert-manager": _observed_ready(), + "kube-prometheus-stack": _observed_ready(), + "node-feature-discovery": _observed_ready(), + "nvidia-dra-driver-gpu": _observed_ready(), + "grove": _observed_ready(), + "kai-queue": _observed_ready(), + "kai-queue-root": _observed_ready(), + "kai-scheduler": _observed_ready(), + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com": _observed_ready(), + "modelexpress-server": _observed_ready(), + "modelexpress-server-role": _observed_ready(), + "modelexpress-server-rolebinding": _observed_ready(), + "modelexpress-server-sa": _observed_ready(), + "modelexpress-server-svc": _observed_ready(), + "usage-kai-scheduler-by-kai-queue": _observed_ready(), + "usage-kai-scheduler-by-kai-queue-root": _observed_ready(), + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server": _observed_ready(), + }, + ), + ), + want={ + # Every cloud and stack. + "provider-config-kubernetes", + "provider-config-helm", + "gateway", + "gateway-class", + "gateway-ca-certificate", + "gateway-ca-issuer", + "gateway-serving-certificate", + "gateway-ca-bundle", + "gateway-ca-configmap", + "gateway-client-ca-bundle", + "gateway-client-auth", + "usage-gateway-class-by-gateway", + "usage-envoy-gateway-by-gateway-class", + # The common components. + "ai-gateway", + "ai-gateway-crds", + "dra-driver-critical-pods-quota", + "envoy-gateway", + "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", + "gaie-crds-inferencepools.inference.networking.k8s.io", + "gaie-crds-inferencepools.inference.networking.x-k8s.io", + "gateway-namespace", + "gateway-proxy", + "gateway-selfsigned-issuer", + "trust-manager", + "usage-ai-gateway-crds-by-ai-gateway", + "usage-cert-manager-by-envoy-gateway", + "usage-cert-manager-by-gateway-selfsigned-issuer", + "usage-gateway-namespace-by-gateway-proxy", + "usage-gateway-namespace-by-gateway-selfsigned-issuer", + "usage-gateway-selfsigned-issuer-by-trust-manager", + # The hand-written Existing half. + "cert-manager", + "kube-prometheus-stack", + "node-feature-discovery", + "nvidia-dra-driver-gpu", + # The Dynamo stack. + "grove", + "kai-queue", + "kai-queue-root", + "kai-scheduler", + "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", + "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", + "modelexpress-server", + "modelexpress-server-role", + "modelexpress-server-rolebinding", + "modelexpress-server-sa", + "modelexpress-server-svc", + "usage-kai-scheduler-by-kai-queue", + "usage-kai-scheduler-by-kai-queue-root", + "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", + "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", + }, + ), +] -# Every cloud and stack, for a stack with an InferenceGateway CA to trust (as -# _request builds), so the Gateway and its client-auth policy are included. -_ALWAYS = frozenset( - { - "provider-config-kubernetes", - "provider-config-helm", - "gateway", - "gateway-class", - "gateway-ca-certificate", - "gateway-ca-issuer", - "gateway-serving-certificate", - "gateway-ca-bundle", - "gateway-ca-configmap", - "gateway-client-ca-bundle", - "gateway-client-auth", - "usage-gateway-class-by-gateway", - "usage-envoy-gateway-by-gateway-class", - } -) - -_COMMON = frozenset( - { - "ai-gateway", - "ai-gateway-crds", - "dra-driver-critical-pods-quota", - "envoy-gateway", - "gaie-crds-inferenceobjectives.inference.networking.x-k8s.io", - "gaie-crds-inferencepools.inference.networking.k8s.io", - "gaie-crds-inferencepools.inference.networking.x-k8s.io", - "gateway-namespace", - "gateway-proxy", - "gateway-selfsigned-issuer", - "trust-manager", - "usage-ai-gateway-crds-by-ai-gateway", - "usage-cert-manager-by-envoy-gateway", - "usage-cert-manager-by-gateway-selfsigned-issuer", - "usage-gateway-namespace-by-gateway-proxy", - "usage-gateway-namespace-by-gateway-selfsigned-issuer", - "usage-gateway-selfsigned-issuer-by-trust-manager", - } -) - -_STANDARD = frozenset( - { - "leader-worker-set", - } -) - -_DYNAMO = frozenset( - { - "grove", - "kai-queue", - "kai-queue-root", - "kai-scheduler", - "modelexpress-crds-modelcacheentries.modelexpress.nvidia.com", - "modelexpress-crds-modelmetadatas.modelexpress.nvidia.com", - "modelexpress-server", - "modelexpress-server-role", - "modelexpress-server-rolebinding", - "modelexpress-server-sa", - "modelexpress-server-svc", - "usage-kai-scheduler-by-kai-queue", - "usage-kai-scheduler-by-kai-queue-root", - "usage-modelexpress-crds-modelcacheentries.modelexpress.nvidia.com-by-modelexpress-server", - "usage-modelexpress-crds-modelmetadatas.modelexpress.nvidia.com-by-modelexpress-server", - } -) - -_EKS = frozenset( - { - "cert-manager", - "gpu-operator", - "k8s-ephemeral-storage-metrics", - "kube-prometheus-stack", - "node-feature-discovery", - "nodewright-operator", - "nvidia-dra-driver-gpu", - "nvsentinel", - "prometheus-adapter", - "prometheus-operator-crds", - "usage-cert-manager-by-gpu-operator", - "usage-cert-manager-by-nvsentinel", - "usage-gpu-operator-by-nvidia-dra-driver-gpu", - "usage-gpu-operator-by-nvsentinel", - "usage-kube-prometheus-stack-by-gpu-operator", - "usage-kube-prometheus-stack-by-k8s-ephemeral-storage-metrics", - "usage-kube-prometheus-stack-by-prometheus-adapter", - "usage-node-feature-discovery-by-gpu-operator", - "usage-prometheus-operator-crds-by-k8s-ephemeral-storage-metrics", - "usage-prometheus-operator-crds-by-kube-prometheus-stack", - "usage-prometheus-operator-crds-by-nvsentinel", - } -) -# AKS additionally carries the gpu-operator's toolkit-hardening manifest. -_AKS = _EKS | frozenset( - { - "gpu-operator-manifests", - "usage-gpu-operator-by-gpu-operator-manifests", - } -) - -# GKE additionally carries the critical-pods ResourceQuota aicr's -# bundler synthesizes as a gpu-operator pre-manifest (GKE rejects -# system-node-critical pods in a namespace without one; aicr#915). -_GKE = _EKS | frozenset( - { - "gpu-operator-pre-manifests-gpu-operator", - "gpu-operator-pre-manifests-aicr-gke-critical-pods", - "usage-gpu-operator-pre-manifests-gpu-operator-by-gpu-operator", - "usage-gpu-operator-pre-manifests-aicr-gke-critical-pods-by-gpu-operator", - } -) - -_HAND_WRITTEN = frozenset( - { - "cert-manager", - "kube-prometheus-stack", - "node-feature-discovery", - "nvidia-dra-driver-gpu", - } -) - -# VKE pre-installs NFD via its managed GPU Operator add-on, so the -# Vultr half carries no node-feature-discovery of its own. -_VULTR = _HAND_WRITTEN - frozenset({"node-feature-discovery"}) - -_INVENTORY = { - "EKS": _EKS, - "AKS": _AKS, - "GKE": _GKE, - "Nebius": _HAND_WRITTEN, - "Vultr": _VULTR, - "Existing": _HAND_WRITTEN, -} - - -@pytest.mark.parametrize( - ("stack", "stack_keys"), [("Standard", _STANDARD), ("Dynamo", _DYNAMO)], ids=["Standard", "Dynamo"] -) -@pytest.mark.parametrize(("cloud", "cloud_keys"), list(_INVENTORY.items()), ids=list(_INVENTORY)) -def test_composed_resource_keys(cloud: str, cloud_keys: frozenset[str], stack: str, stack_keys: frozenset[str]) -> None: - """Every cloud and stack composes exactly its inventoried resource keys.""" - expected = _ALWAYS | _COMMON | cloud_keys | stack_keys - # Observe every expected key Ready so the depends_on - # install gate opens and the full stack renders; a - # key the function doesn't render still fails the - # comparison. - observed = _observed_pcs() - for key in expected: - observed[key] = fnv1.Resource( - resource=resource.dict_to_struct({"status": {"conditions": [{"type": "Ready", "status": "True"}]}}) - ) - got = asyncio.run(fn.FunctionRunner().RunFunction(_request(cloud, stack, observed=observed), None)) - assert set(got.desired.resources.keys()) == expected +@pytest.mark.parametrize("case", COMPOSED_RESOURCE_KEYS_CASES, ids=lambda case: case.name) +def test_composed_resource_keys(case: ComposedResourceKeysCase) -> None: + """RunFunction composes exactly the inventoried resource keys for a cloud and stack.""" + got = asyncio.run(fn.FunctionRunner().RunFunction(case.req, None)) + assert set(got.desired.resources) == case.want diff --git a/functions/compose-serving-stack/tests/test_stacks.py b/functions/compose-serving-stack/tests/test_stacks.py index 2017a2264..677368835 100644 --- a/functions/compose-serving-stack/tests/test_stacks.py +++ b/functions/compose-serving-stack/tests/test_stacks.py @@ -14,56 +14,72 @@ """Tests for the serving stack component lists. -The join itself is the assertion: components() fails closed on -duplicate keys and on depends_on edges the join didn't produce, so -iterating every cloud and stack pair gates every list - including the -generated ones, once mapped - without involving fn.py. +The join itself is the assertion: join() fails closed on duplicate +keys and on depends_on edges the join didn't produce, so iterating +every cloud and stack pair gates every list - including the generated +ones, once mapped - without involving fn.py. The rest are properties +every joined stack must hold, each checked over the whole stack at once +so a failure lists every component that breaks it. """ +import dataclasses + import pytest from function import stacks +@dataclasses.dataclass +class Case: + """A test case for stacks.join.""" + + name: str + cloud: str + stack: str + want: str + + @pytest.mark.parametrize("stack", stacks.stacks()) @pytest.mark.parametrize("cloud", stacks.clouds()) def test_every_cloud_and_stack_joins(cloud: stacks.Cloud, stack: stacks.Stack) -> None: """Every cloud and stack pair joins into a non-empty stack.""" - got = stacks.join(cloud, stack) - assert got, "a joined stack can't be empty" + assert stacks.join(cloud, stack), "a joined stack can't be empty" @pytest.mark.parametrize("stack", stacks.stacks()) @pytest.mark.parametrize("cloud", stacks.clouds()) def test_charts_have_reserved_release_names(cloud: stacks.Cloud, stack: stacks.Stack) -> None: """Every Chart's Helm release is named mp-.""" - for c in stacks.join(cloud, stack): - if isinstance(c, stacks.Chart): - assert c.release == f"mp-{c.chart}", ( - f"{c.key}: release names are mp-: stable across upgrades, reserved to Modelplane" - ) + charts = [c for c in stacks.join(cloud, stack) if isinstance(c, stacks.Chart)] + assert {c.key: c.release for c in charts} == {c.key: f"mp-{c.chart}" for c in charts}, ( + "release names are mp-: stable across upgrades, reserved to Modelplane" + ) @pytest.mark.parametrize("stack", stacks.stacks()) @pytest.mark.parametrize("cloud", stacks.clouds()) def test_manifests_are_populated(cloud: stacks.Cloud, stack: stacks.Stack) -> None: """Every Manifests entry carries at least one manifest.""" - for c in stacks.join(cloud, stack): - if isinstance(c, stacks.Manifests): - assert c.manifests, f"{c.key}: a Manifests entry can't be empty" + empty = [c.key for c in stacks.join(cloud, stack) if isinstance(c, stacks.Manifests) and not c.manifests] + assert empty == [], "a Manifests entry can't be empty" @pytest.mark.parametrize("stack", stacks.stacks()) @pytest.mark.parametrize("cloud", stacks.clouds()) def test_multi_doc_manifests_derive_per_doc_keys(cloud: stacks.Cloud, stack: stacks.Stack) -> None: - """A multi-doc Manifests entry renders a key per doc, and anything else renders its own key.""" - for c in stacks.join(cloud, stack): - keys = stacks.components.doc_keys(c) + """doc_keys agrees with a restatement of its own rule.""" + # This restates doc_keys line for line, so it only catches one copy + # changing without the other. COMPOSED_RESOURCE_KEYS_CASES in test_fn.py + # pins the real keys, as literals. + joined = stacks.join(cloud, stack) + want = {} + for c in joined: if isinstance(c, stacks.Chart) or len(c.manifests) == 1: - assert keys == [c.key] - continue - assert keys == [f"{c.key}-{doc['metadata']['name']}" for doc in c.manifests], ( - f"{c.key}: a multi-doc bundle renders one Object per doc, keyed -" - ) + want[c.key] = [c.key] + else: + want[c.key] = [f"{c.key}-{doc['metadata']['name']}" for doc in c.manifests] + assert {c.key: stacks.components.doc_keys(c) for c in joined} == want, ( + "a multi-doc bundle renders one Object per doc, keyed -" + ) @pytest.mark.parametrize("stack", stacks.stacks()) @@ -73,9 +89,12 @@ def test_ready_entries_are_single_doc(cloud: stacks.Cloud, stack: stacks.Stack) # A readiness CEL query applies to every doc in an entry, so an # entry carrying one keeps to a single manifest - a Service or # ServiceAccount has no status conditions to satisfy it. - for c in stacks.join(cloud, stack): - if isinstance(c, stacks.Manifests) and c.ready is not None: - assert len(c.manifests) == 1, c.key + not_single = [ + c.key + for c in stacks.join(cloud, stack) + if isinstance(c, stacks.Manifests) and c.ready is not None and len(c.manifests) != 1 + ] + assert not_single == [], "a readiness query applies to every doc, so an entry carrying one has one manifest" @pytest.mark.parametrize("stack", stacks.stacks()) @@ -88,16 +107,14 @@ def test_depended_on_charts_wait(cloud: stacks.Cloud, stack: stacks.Stack) -> No # gate would open the moment Helm accepted the manifests. joined = stacks.join(cloud, stack) depended_on = {dep for c in joined for dep in c.depends_on} - for c in joined: - if isinstance(c, stacks.Chart) and c.key in depended_on: - assert c.wait, f"{c.key}: a depended-on chart must set wait" + not_waiting = [c.key for c in joined if isinstance(c, stacks.Chart) and c.key in depended_on and not c.wait] + assert not_waiting == [], "a depended-on chart must set wait" @pytest.mark.parametrize("stack", stacks.stacks()) @pytest.mark.parametrize("cloud", stacks.clouds()) def test_no_wildcard_tolerations(cloud: stacks.Cloud, stack: stacks.Stack) -> None: """No component of a joined stack carries a keyless toleration.""" - # A keyless toleration tolerates every taint, so the pod lands # on tainted GPU nodes: control-plane charts squat on # accelerated capacity and their eviction stalls autoscaler @@ -106,13 +123,13 @@ def test_no_wildcard_tolerations(cloud: stacks.Cloud, stack: stacks.Stack) -> No # (TOLERATIONS in generate.py). This pins that no keyless # toleration survives in any joined stack, chart values and # manifests alike. + wildcards = [] + def check(node: object, where: str) -> None: if isinstance(node, dict): for key, val in node.items(): if key == "tolerations" and isinstance(val, list): - for toleration in val: - assert isinstance(toleration, dict), f"keyless (wildcard) toleration in {where}" - assert "key" in toleration, f"keyless (wildcard) toleration in {where}" + wildcards.extend((where, t) for t in val if not (isinstance(t, dict) and "key" in t)) else: check(val, where) elif isinstance(node, list): @@ -121,14 +138,20 @@ def check(node: object, where: str) -> None: for c in stacks.join(cloud, stack): check(c.values if isinstance(c, stacks.Chart) else c.manifests, c.key) + assert wildcards == [], "keyless (wildcard) tolerations, by component" + + +# The Literal types reject these at type-checking time; these exercise the +# runtime guard behind them, which catches the API and the stacks package +# disagreeing on a value. +JOIN_FAILS_CLOSED_CASES = [ + Case(name="unknown cloud", cloud="Mars", stack="Standard", want="unknown cloud 'Mars'"), + Case(name="unknown stack", cloud="Nebius", stack="Turbo", want="unknown stack 'Turbo'"), +] -def test_unknown_cloud_and_stack_fail_closed() -> None: - """join rejects an unknown cloud or stack.""" - # The Literal types reject these at type-checking time; this - # exercises the runtime guard behind them, which catches the API - # and the stacks package disagreeing on a value. - with pytest.raises(ValueError, match="unknown cloud 'Mars'"): - stacks.join("Mars", "Standard") # ty: ignore[invalid-argument-type] - with pytest.raises(ValueError, match="unknown stack 'Turbo'"): - stacks.join("Nebius", "Turbo") # ty: ignore[invalid-argument-type] +@pytest.mark.parametrize("case", JOIN_FAILS_CLOSED_CASES, ids=lambda case: case.name) +def test_join_fails_closed(case: Case) -> None: + """join rejects a cloud or stack it doesn't know.""" + with pytest.raises(ValueError, match=case.want): + stacks.join(case.cloud, case.stack) # ty: ignore[invalid-argument-type] # cases pass values outside the Literals diff --git a/functions/compose-usages/tests/test_fn.py b/functions/compose-usages/tests/test_fn.py index 912f59a70..d64855bdb 100644 --- a/functions/compose-usages/tests/test_fn.py +++ b/functions/compose-usages/tests/test_fn.py @@ -26,166 +26,281 @@ from google.protobuf import json_format, message from google.protobuf import struct_pb2 as structpb -_NAMESPACE = "test-ns" -_PC = "test-cluster" - -_RELEASE = { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "Release", - "metadata": {"namespace": _NAMESPACE}, - "spec": { - "providerConfigRef": {"kind": "ProviderConfig", "name": _PC}, - "forProvider": {"chart": {"name": "cert-manager"}}, - }, -} - -_OBJECT = { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": _NAMESPACE}, - "spec": { - "providerConfigRef": {"kind": "ProviderConfig", "name": _PC}, - "forProvider": {"manifest": {"apiVersion": "v1", "kind": "Namespace"}}, - }, -} - -# Not a consumer kind: a ProviderConfig gets no Usage of its own. -_PROVIDER_CONFIG = { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "ProviderConfig", - "metadata": {"name": _PC, "namespace": _NAMESPACE}, - "spec": {}, -} - -# A consumer kind (Object) that references no ProviderConfig: gets no Usage. -_OBJECT_NO_PC = { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": _NAMESPACE}, - "spec": {"forProvider": {"manifest": {"apiVersion": "v1", "kind": "ConfigMap"}}}, -} - -# A Release that already carries a label, to check relabeling preserves it. -_RELEASE_WITH_LABEL = { - "apiVersion": "helm.m.crossplane.io/v1beta1", - "kind": "Release", - "metadata": {"namespace": _NAMESPACE, "labels": {"existing": "keep"}}, - "spec": { - "providerConfigRef": {"kind": "ProviderConfig", "name": _PC}, - "forProvider": {"chart": {"name": "prometheus"}}, - }, -} - - -def _labelled(d: dict, consumer: str) -> dict: - """A copy of d with the usage-consumer label stamped on it.""" - out = {**d, "metadata": {**d.get("metadata", {})}} - out["metadata"]["labels"] = { - **d.get("metadata", {}).get("labels", {}), - "modelplane.ai/usage-consumer": consumer, - } - return out - - -def _usage(api_version: str, kind: str, consumer: str) -> dict: - return { - "apiVersion": "protection.crossplane.io/v1beta1", - "kind": "Usage", - "metadata": {"namespace": _NAMESPACE}, - "spec": { - "of": { - "apiVersion": api_version, - "kind": "ProviderConfig", - "resourceRef": {"name": _PC}, - }, - "by": { - "apiVersion": api_version, - "kind": kind, - "resourceSelector": { - "matchControllerRef": True, - "matchLabels": {"modelplane.ai/usage-consumer": consumer}, - }, - }, - "replayDeletion": True, - }, - } + +@dataclasses.dataclass +class Case: + """A test case for compose-usages.""" + + name: str + req: fnv1.RunFunctionRequest + want: fnv1.RunFunctionResponse -def _composite(namespace: str | None = _NAMESPACE) -> structpb.Struct: +# compose-usages reads only the observed composite's namespace, and passes the +# desired one through. The ServingStack model requires a spec.cloud the function +# never reads, so both composites are bare dicts rather than built from the +# model. +def _serving_stack(*, namespace: str | None) -> fnv1.Resource: + """The bare observed ServingStack composite, in namespace unless it's None.""" metadata = {"name": "test"} if namespace is not None: metadata["namespace"] = namespace - return resource.dict_to_struct( - { - "apiVersion": "infrastructure.modelplane.ai/v1alpha1", - "kind": "ServingStack", - "metadata": metadata, - } + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": metadata, + } + ) ) -@dataclasses.dataclass -class Case: - """A test case for compose-usages.""" +def _desired_serving_stack(*, namespace: str | None) -> fnv1.Resource: + """The bare desired ServingStack composite, in namespace unless it's None.""" + metadata = {"name": "test"} + if namespace is not None: + metadata["namespace"] = namespace + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "infrastructure.modelplane.ai/v1alpha1", + "kind": "ServingStack", + "metadata": metadata, + } + ) + ) - name: str - req: fnv1.RunFunctionRequest - want: fnv1.RunFunctionResponse + +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) COMPOSE_CASES = [ Case( name="labels each consumer and composes a Usage per ProviderConfig reference", req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=_composite())), + observed=fnv1.State( + composite=_serving_stack(namespace="test-ns"), + ), desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), + composite=_desired_serving_stack(namespace="test-ns"), resources={ - "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), - "gateway-namespace": fnv1.Resource(resource=resource.dict_to_struct(_OBJECT)), - "prometheus": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE_WITH_LABEL)), - "config-map": fnv1.Resource(resource=resource.dict_to_struct(_OBJECT_NO_PC)), - "provider-config-helm": fnv1.Resource(resource=resource.dict_to_struct(_PROVIDER_CONFIG)), + "cert-manager": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "test-ns"}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"chart": {"name": "cert-manager"}}, + }, + } + ) + ), + "gateway-namespace": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "test-ns"}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"manifest": {"apiVersion": "v1", "kind": "Namespace"}}, + }, + } + ) + ), + # A Release that already carries a label, which the + # function's label must not replace. + "prometheus": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "test-ns", "labels": {"existing": "keep"}}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"chart": {"name": "prometheus"}}, + }, + } + ) + ), + # A consumer kind that references no ProviderConfig. + "config-map": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "test-ns"}, + "spec": {"forProvider": {"manifest": {"apiVersion": "v1", "kind": "ConfigMap"}}}, + } + ) + ), + # Not a consumer kind. + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster", "namespace": "test-ns"}, + "spec": {}, + } + ) + ), }, ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), + composite=_desired_serving_stack(namespace="test-ns"), resources={ "cert-manager": fnv1.Resource( - resource=resource.dict_to_struct(_labelled(_RELEASE, "cert-manager")), + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "namespace": "test-ns", + "labels": {"modelplane.ai/usage-consumer": "cert-manager"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"chart": {"name": "cert-manager"}}, + }, + } + ) ), "gateway-namespace": fnv1.Resource( - resource=resource.dict_to_struct(_labelled(_OBJECT, "gateway-namespace")), + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": { + "namespace": "test-ns", + "labels": {"modelplane.ai/usage-consumer": "gateway-namespace"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"manifest": {"apiVersion": "v1", "kind": "Namespace"}}, + }, + } + ) ), - # Existing labels are preserved when the consumer label is stamped. "prometheus": fnv1.Resource( - resource=resource.dict_to_struct(_labelled(_RELEASE_WITH_LABEL, "prometheus")), + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": { + "namespace": "test-ns", + "labels": {"existing": "keep", "modelplane.ai/usage-consumer": "prometheus"}, + }, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"chart": {"name": "prometheus"}}, + }, + } + ) ), - # An Object with no providerConfigRef is left untouched, no Usage. "config-map": fnv1.Resource( - resource=resource.dict_to_struct(_OBJECT_NO_PC), + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "test-ns"}, + "spec": {"forProvider": {"manifest": {"apiVersion": "v1", "kind": "ConfigMap"}}}, + } + ) ), "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_PROVIDER_CONFIG), + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster", "namespace": "test-ns"}, + "spec": {}, + } + ) ), "usage-pc-cert-manager": fnv1.Resource( resource=resource.dict_to_struct( - _usage("helm.m.crossplane.io/v1beta1", "Release", "cert-manager") + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "test-ns"}, + "spec": { + "of": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "resourceRef": {"name": "test-cluster"}, + }, + "by": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/usage-consumer": "cert-manager"}, + }, + }, + "replayDeletion": True, + }, + } ), ready=fnv1.READY_TRUE, ), "usage-pc-gateway-namespace": fnv1.Resource( resource=resource.dict_to_struct( - _usage("kubernetes.m.crossplane.io/v1alpha1", "Object", "gateway-namespace") + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "test-ns"}, + "spec": { + "of": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ProviderConfig", + "resourceRef": {"name": "test-cluster"}, + }, + "by": { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/usage-consumer": "gateway-namespace"}, + }, + }, + "replayDeletion": True, + }, + } ), ready=fnv1.READY_TRUE, ), "usage-pc-prometheus": fnv1.Resource( resource=resource.dict_to_struct( - _usage("helm.m.crossplane.io/v1beta1", "Release", "prometheus") + { + "apiVersion": "protection.crossplane.io/v1beta1", + "kind": "Usage", + "metadata": {"namespace": "test-ns"}, + "spec": { + "of": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "resourceRef": {"name": "test-cluster"}, + }, + "by": { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "resourceSelector": { + "matchControllerRef": True, + "matchLabels": {"modelplane.ai/usage-consumer": "prometheus"}, + }, + }, + "replayDeletion": True, + }, + } ), ready=fnv1.READY_TRUE, ), @@ -197,20 +312,46 @@ class Case: Case( name="no Usages when the composite has no namespace", req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=_composite(namespace=None))), + observed=fnv1.State( + composite=_serving_stack(namespace=None), + ), desired=fnv1.State( - composite=fnv1.Resource(resource=_composite(namespace=None)), + composite=_desired_serving_stack(namespace=None), resources={ - "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), + "cert-manager": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "test-ns"}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"chart": {"name": "cert-manager"}}, + }, + } + ) + ), }, ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=_composite(namespace=None)), + composite=_desired_serving_stack(namespace=None), resources={ - "cert-manager": fnv1.Resource(resource=resource.dict_to_struct(_RELEASE)), + "cert-manager": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "Release", + "metadata": {"namespace": "test-ns"}, + "spec": { + "providerConfigRef": {"kind": "ProviderConfig", "name": "test-cluster"}, + "forProvider": {"chart": {"name": "cert-manager"}}, + }, + } + ) + ), }, ), context=structpb.Struct(), @@ -219,21 +360,39 @@ class Case: Case( name="no consumers means no Usages", req=fnv1.RunFunctionRequest( - observed=fnv1.State(composite=fnv1.Resource(resource=_composite())), + observed=fnv1.State( + composite=_serving_stack(namespace="test-ns"), + ), desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), + composite=_desired_serving_stack(namespace="test-ns"), resources={ - "provider-config-helm": fnv1.Resource(resource=resource.dict_to_struct(_PROVIDER_CONFIG)), + "provider-config-helm": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster", "namespace": "test-ns"}, + "spec": {}, + } + ) + ), }, ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=_composite()), + composite=_desired_serving_stack(namespace="test-ns"), resources={ "provider-config-helm": fnv1.Resource( - resource=resource.dict_to_struct(_PROVIDER_CONFIG), + resource=resource.dict_to_struct( + { + "apiVersion": "helm.m.crossplane.io/v1beta1", + "kind": "ProviderConfig", + "metadata": {"name": "test-cluster", "namespace": "test-ns"}, + "spec": {}, + } + ) ), }, ), @@ -243,11 +402,6 @@ class Case: ] -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - @pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction labels consumers and composes their Usages.""" diff --git a/functions/compose-vultr-cluster/tests/test_fn.py b/functions/compose-vultr-cluster/tests/test_fn.py index 608ed76f7..a883da015 100644 --- a/functions/compose-vultr-cluster/tests/test_fn.py +++ b/functions/compose-vultr-cluster/tests/test_fn.py @@ -39,254 +39,240 @@ class Case: want: fnv1.RunFunctionResponse -# Name of the cluster's connection secret. Derived like the function derives -# it - the hash suffix depends only on the parent and child names. -_KUBECONFIG_SECRET_NAME = resource.child_name("test-cluster", "kubeconfig") - -# The system node pool injected inline into every cluster. -_SYSTEM_POOL = { - "label": "system", - "plan": "vc2-6c-16gb", - "nodeQuantity": 1, - "autoScaler": True, - "minNodes": 1, - "maxNodes": 2, - "labels": [{"key": "modelplane.ai/pool", "value": "system"}], -} - -# The taint every GPU pool carries. -_GPU_TAINTS = [ - {"key": "nvidia.com/gpu", "value": "true", "effect": "NoSchedule"}, -] - - -def _xr(pools: list[v1alpha1.NodePool]) -> dict: - """A VultrCluster XR with the given node pools, as a request dict.""" - return v1alpha1.VultrCluster( +def _xr(*, node_pools: list[v1alpha1.NodePool]) -> fnv1.Resource: + """The observed VultrCluster XR, with the given node pools.""" + xr = v1alpha1.VultrCluster( metadata=metav1.ObjectMeta( name="test-cluster", namespace="modelplane-system", ), spec=v1alpha1.Spec( region="ewr", - nodePools=pools, + nodePools=node_pools, ), - ).model_dump(exclude_none=True, mode="json") - - -def _req( - pools: list[v1alpha1.NodePool], - observed_resources: dict[str, fnv1.Resource] | None = None, -) -> fnv1.RunFunctionRequest: - return fnv1.RunFunctionRequest( - observed=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_xr(pools))), - resources=observed_resources or {}, + ) + return fnv1.Resource(resource=resource.dict_to_struct(xr.model_dump(exclude_none=True, mode="json", by_alias=True))) + + +def _desired_xr() -> fnv1.Resource: + """The desired XR, publishing the cluster's kubeconfig Secret.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "status": { + "secrets": [ + { + "type": "Kubeconfig", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + ], + }, + } ), ) -def _cluster( - cred_kind: str = "ClusterProviderConfig", - cred_name: str = "default", -) -> dict: - """A Kubernetes cluster golden with only the system pool.""" - return { - "apiVersion": "vke.vultr.m.upbound.io/v1beta1", - "kind": "Kubernetes", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": { - "label": "test-cluster", - "region": "ewr", - "version": "v1.36.2+1", - "haControlplanes": True, - "nodePools": _SYSTEM_POOL, - }, - "writeConnectionSecretToRef": {"name": _KUBECONFIG_SECRET_NAME}, - }, - } - - -def _provider_config() -> dict: - """A provider-kubernetes ProviderConfig golden pointing at the kubeconfig.""" - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "ProviderConfig", - "metadata": { - "name": _KUBECONFIG_SECRET_NAME, - "namespace": "modelplane-system", - }, - "spec": { - "credentials": { - "source": "Secret", - "secretRef": { - "namespace": "modelplane-system", - "name": _KUBECONFIG_SECRET_NAME, - "key": "kubeconfig", +def _cluster(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed VKE cluster.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vke.vultr.m.upbound.io/v1beta1", + "kind": "Kubernetes", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "label": "test-cluster", + "region": "ewr", + "version": "v1.36.2+1", + "haControlplanes": True, + # The system node pool the function adds inline to every cluster. + "nodePools": { + "label": "system", + "plan": "vc2-6c-16gb", + "nodeQuantity": 1, + "autoScaler": True, + "minNodes": 1, + "maxNodes": 2, + "labels": [{"key": "modelplane.ai/pool", "value": "system"}], + }, + }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, }, - }, - }, - } + } + ), + ready=ready, + ) -def _gpu_observer() -> dict: - """A provider-kubernetes Object golden that observes the GPU validator DS.""" - return { - "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", - "kind": "Object", - "metadata": {"namespace": "modelplane-system"}, - "spec": { - "managementPolicies": ["Observe"], - "providerConfigRef": { - "kind": "ProviderConfig", - "name": _KUBECONFIG_SECRET_NAME, - }, - "readiness": { - "policy": "DeriveFromCelQuery", - "celQuery": ( - "has(object.status.numberReady)" - " && object.status.desiredNumberScheduled >= 1" - " && object.status.numberReady == object.status.desiredNumberScheduled" - ), - }, - "forProvider": { - "manifest": { - "apiVersion": "apps/v1", - "kind": "DaemonSet", - "metadata": { - "name": "nvidia-operator-validator", - "namespace": "gpu-operator", +def _observed_cluster(*, ready: bool) -> fnv1.Resource: + """The VKE cluster as observed, with a Ready condition that's True if ready and False if not.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vke.vultr.m.upbound.io/v1beta1", + "kind": "Kubernetes", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "label": "test-cluster", + "region": "ewr", + "version": "v1.36.2+1", + "haControlplanes": True, + "nodePools": { + "label": "system", + "plan": "vc2-6c-16gb", + "nodeQuantity": 1, + "autoScaler": True, + "minNodes": 1, + "maxNodes": 2, + "labels": [{"key": "modelplane.ai/pool", "value": "system"}], + }, }, + "writeConnectionSecretToRef": {"name": "test-cluster-kubeconfig-55b57"}, }, - }, - }, - } + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True" if ready else "False", + "reason": "Available" if ready else "Unavailable", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ) -def _node_pool( - label: str, - plan: str, - node_quantity: int, - labels: list, - taints: list | None = None, - *, - auto_scaler: bool = False, - min_nodes: int | None = None, - max_nodes: int | None = None, - cred_kind: str = "ClusterProviderConfig", - cred_name: str = "default", -) -> dict: - """A KubernetesNodePool golden.""" - fp: dict[str, Any] = { - "label": label, - "plan": plan, +def _gpu_node_pool(*, node_quantity: int, autoscaling: dict | None, ready: fnv1.Ready) -> fnv1.Resource: + """The composed KubernetesNodePool for the gpu-l40s pool, with autoscaling if given.""" + for_provider: dict[str, Any] = { + "label": "gpu-l40s", + "plan": "vcg-l40s-16c-180g-48vram", "nodeQuantity": node_quantity, - "labels": labels, + "labels": [ + {"key": "modelplane.ai/pool", "value": "gpu-l40s"}, + {"key": "modelplane.ai/gpu", "value": "nvidia-l40s"}, + {"key": "nvidia.com/gpu.deploy.device-plugin", "value": "false"}, + ], "clusterIdSelector": {"matchControllerRef": True}, + "taints": [{"key": "nvidia.com/gpu", "value": "true", "effect": "NoSchedule"}], } - if taints: - fp["taints"] = taints - if auto_scaler: - fp["autoScaler"] = True - fp["minNodes"] = min_nodes - fp["maxNodes"] = max_nodes - return { - "apiVersion": "vke.vultr.m.upbound.io/v1beta1", - "kind": "KubernetesNodePool", - "spec": { - "providerConfigRef": {"kind": cred_kind, "name": cred_name}, - "forProvider": fp, - }, - } - - -def _status() -> dict: - return { - "status": { - "secrets": [ - { - "type": "Kubeconfig", - "name": _KUBECONFIG_SECRET_NAME, - "key": "kubeconfig", + if autoscaling is not None: + for_provider.update(autoscaling) + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vke.vultr.m.upbound.io/v1beta1", + "kind": "KubernetesNodePool", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": for_provider, }, - ], - }, - } + } + ), + ready=ready, + ) -def _observed_ready(desired: dict) -> fnv1.Resource: - """An observed variant of a desired resource with a Ready=True condition.""" - observed = { - **desired, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "True", - "reason": "Available", - "lastTransitionTime": "2024-01-01T00:00:00Z", +def _provider_config() -> fnv1.Resource: + """The composed provider-kubernetes ProviderConfig for the cluster, which is always ready.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ProviderConfig", + "metadata": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", }, - ], - }, - } - return fnv1.Resource(resource=resource.dict_to_struct(observed)) + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + }, + }, + } + ), + ready=fnv1.READY_TRUE, + ) -def _observed_unready(desired: dict) -> fnv1.Resource: - """An observed variant of a desired resource with a Ready=False condition.""" - observed = { - **desired, - "status": { - "conditions": [ - { - "type": "Ready", - "status": "False", - "reason": "Unavailable", - "lastTransitionTime": "2024-01-01T00:00:00Z", +def _gpu_observer(*, ready: fnv1.Ready) -> fnv1.Resource: + """The composed Object observing the GPU validator DaemonSet.""" + return fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status.numberReady)" + " && object.status.desiredNumberScheduled >= 1" + " && object.status.numberReady == object.status.desiredNumberScheduled" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "DaemonSet", + "metadata": { + "name": "nvidia-operator-validator", + "namespace": "gpu-operator", + }, + }, + }, }, - ], - }, - } - return fnv1.Resource(resource=resource.dict_to_struct(observed)) - + } + ), + ready=ready, + ) -_GPU_POOL = v1alpha1.NodePool( - name="gpu-l40s", - role="GPU", - plan="vcg-l40s-16c-180g-48vram", - maxNodeCount=4, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), -) -_GPU_POOL_GOLDEN = _node_pool( - label="gpu-l40s", - plan="vcg-l40s-16c-180g-48vram", - node_quantity=1, - labels=[ - {"key": "modelplane.ai/pool", "value": "gpu-l40s"}, - {"key": "modelplane.ai/gpu", "value": "nvidia-l40s"}, - {"key": "nvidia.com/gpu.deploy.device-plugin", "value": "false"}, - ], - taints=_GPU_TAINTS, - auto_scaler=True, - min_nodes=1, - max_nodes=4, -) +def _to_dict(msg: message.Message) -> dict: + """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" + return json.loads(json_format.MessageToJson(msg, sort_keys=True)) COMPOSE_CASES = [ Case( name="cluster composed first; node pools withheld until cluster Ready", - req=_req([_GPU_POOL]), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + plan="vcg-l40s-16c-180g-48vram", + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], + ), + ), + ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), + "cluster": _cluster(ready=fnv1.READY_UNSPECIFIED), }, ), context=structpb.Struct(), @@ -294,31 +280,37 @@ def _observed_unready(desired: dict) -> fnv1.Resource: ), Case( name="node pools and GPU observer composed once cluster is Ready; autoscaling from maxNodeCount", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + plan="vcg-l40s-16c-180g-48vram", + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], + ), + resources={ + "cluster": _observed_cluster(ready=True), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), + "cluster": _cluster(ready=fnv1.READY_TRUE), + "node-pool-gpu-l40s": _gpu_node_pool( + node_quantity=1, + autoscaling={"autoScaler": True, "minNodes": 1, "maxNodes": 4}, + ready=fnv1.READY_UNSPECIFIED, ), + "provider-config-kubernetes": _provider_config(), + "gpu-observer": _gpu_observer(ready=fnv1.READY_UNSPECIFIED), }, ), context=structpb.Struct(), @@ -326,31 +318,69 @@ def _observed_unready(desired: dict) -> fnv1.Resource: ), Case( name="dependents kept when the cluster Ready condition transiently regresses", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_unready(_cluster()), - "provider-config-kubernetes": _observed_ready(_provider_config()), - }, + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + plan="vcg-l40s-16c-180g-48vram", + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], + ), + resources={ + "cluster": _observed_cluster(ready=False), + "provider-config-kubernetes": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "ProviderConfig", + "metadata": { + "name": "test-cluster-kubeconfig-55b57", + "namespace": "modelplane-system", + }, + "spec": { + "credentials": { + "source": "Secret", + "secretRef": { + "namespace": "modelplane-system", + "name": "test-cluster-kubeconfig-55b57", + "key": "kubeconfig", + }, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), + "cluster": _cluster(ready=fnv1.READY_UNSPECIFIED), + "node-pool-gpu-l40s": _gpu_node_pool( + node_quantity=1, + autoscaling={"autoScaler": True, "minNodes": 1, "maxNodes": 4}, + ready=fnv1.READY_UNSPECIFIED, ), + "provider-config-kubernetes": _provider_config(), + "gpu-observer": _gpu_observer(ready=fnv1.READY_UNSPECIFIED), }, ), context=structpb.Struct(), @@ -358,32 +388,73 @@ def _observed_unready(desired: dict) -> fnv1.Resource: ), Case( name="observed node pool alone keeps dependents composed", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_unready(_cluster()), - "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN), - }, + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + plan="vcg-l40s-16c-180g-48vram", + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], + ), + resources={ + "cluster": _observed_cluster(ready=False), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vke.vultr.m.upbound.io/v1beta1", + "kind": "KubernetesNodePool", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "label": "gpu-l40s", + "plan": "vcg-l40s-16c-180g-48vram", + "nodeQuantity": 1, + "labels": [ + {"key": "modelplane.ai/pool", "value": "gpu-l40s"}, + {"key": "modelplane.ai/gpu", "value": "nvidia-l40s"}, + {"key": "nvidia.com/gpu.deploy.device-plugin", "value": "false"}, + ], + "clusterIdSelector": {"matchControllerRef": True}, + "taints": [{"key": "nvidia.com/gpu", "value": "true", "effect": "NoSchedule"}], + "autoScaler": True, + "minNodes": 1, + "maxNodes": 4, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), + "cluster": _cluster(ready=fnv1.READY_UNSPECIFIED), + "node-pool-gpu-l40s": _gpu_node_pool( + node_quantity=1, + autoscaling={"autoScaler": True, "minNodes": 1, "maxNodes": 4}, ready=fnv1.READY_TRUE, ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), - ), + "provider-config-kubernetes": _provider_config(), + "gpu-observer": _gpu_observer(ready=fnv1.READY_UNSPECIFIED), }, ), context=structpb.Struct(), @@ -391,51 +462,37 @@ def _observed_unready(desired: dict) -> fnv1.Resource: ), Case( name="fixed-size GPU pool", - req=_req( - [ - v1alpha1.NodePool( - name="gpu-l40s", - role="GPU", - plan="vcg-l40s-16c-180g-48vram", - nodeCount=2, - gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + plan="vcg-l40s-16c-180g-48vram", + nodeCount=2, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], ), - ], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, + resources={ + "cluster": _observed_cluster(ready=True), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct( - _node_pool( - label="gpu-l40s", - plan="vcg-l40s-16c-180g-48vram", - node_quantity=2, - labels=[ - {"key": "modelplane.ai/pool", "value": "gpu-l40s"}, - {"key": "modelplane.ai/gpu", "value": "nvidia-l40s"}, - {"key": "nvidia.com/gpu.deploy.device-plugin", "value": "false"}, - ], - taints=_GPU_TAINTS, - ), - ), - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), + "cluster": _cluster(ready=fnv1.READY_TRUE), + "node-pool-gpu-l40s": _gpu_node_pool( + node_quantity=2, + autoscaling=None, + ready=fnv1.READY_UNSPECIFIED, ), + "provider-config-kubernetes": _provider_config(), + "gpu-observer": _gpu_observer(ready=fnv1.READY_UNSPECIFIED), }, ), context=structpb.Struct(), @@ -443,50 +500,54 @@ def _observed_unready(desired: dict) -> fnv1.Resource: ), Case( name="minNodeCount sets the autoscaler floor; System pool carries no taint", - req=_req( - [ - v1alpha1.NodePool( - name="workers", - role="System", - plan="vc2-6c-16gb", - nodeCount=2, - minNodeCount=2, - maxNodeCount=5, + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + node_pools=[ + v1alpha1.NodePool( + name="workers", + role="System", + plan="vc2-6c-16gb", + nodeCount=2, + minNodeCount=2, + maxNodeCount=5, + ), + ], ), - ], - observed_resources={ - "cluster": _observed_ready(_cluster()), - }, + resources={ + "cluster": _observed_cluster(ready=True), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), + "cluster": _cluster(ready=fnv1.READY_TRUE), "node-pool-workers": fnv1.Resource( resource=resource.dict_to_struct( - _node_pool( - label="workers", - plan="vc2-6c-16gb", - node_quantity=2, - labels=[{"key": "modelplane.ai/pool", "value": "workers"}], - auto_scaler=True, - min_nodes=2, - max_nodes=5, - ), + { + "apiVersion": "vke.vultr.m.upbound.io/v1beta1", + "kind": "KubernetesNodePool", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "label": "workers", + "plan": "vc2-6c-16gb", + "nodeQuantity": 2, + "labels": [{"key": "modelplane.ai/pool", "value": "workers"}], + "clusterIdSelector": {"matchControllerRef": True}, + "autoScaler": True, + "minNodes": 2, + "maxNodes": 5, + }, + }, + } ), ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), - ), + "provider-config-kubernetes": _provider_config(), + "gpu-observer": _gpu_observer(ready=fnv1.READY_UNSPECIFIED), }, ), context=structpb.Struct(), @@ -494,35 +555,117 @@ def _observed_unready(desired: dict) -> fnv1.Resource: ), Case( name="VultrCluster Ready only once the gpu-observer is Ready", - req=_req( - [_GPU_POOL], - observed_resources={ - "cluster": _observed_ready(_cluster()), - "node-pool-gpu-l40s": _observed_ready(_GPU_POOL_GOLDEN), - "gpu-observer": _observed_ready(_gpu_observer()), - }, + req=fnv1.RunFunctionRequest( + observed=fnv1.State( + composite=_xr( + node_pools=[ + v1alpha1.NodePool( + name="gpu-l40s", + role="GPU", + plan="vcg-l40s-16c-180g-48vram", + maxNodeCount=4, + gpu=v1alpha1.Gpu(acceleratorType="nvidia-l40s"), + ), + ], + ), + resources={ + "cluster": _observed_cluster(ready=True), + "node-pool-gpu-l40s": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "vke.vultr.m.upbound.io/v1beta1", + "kind": "KubernetesNodePool", + "spec": { + "providerConfigRef": {"kind": "ClusterProviderConfig", "name": "default"}, + "forProvider": { + "label": "gpu-l40s", + "plan": "vcg-l40s-16c-180g-48vram", + "nodeQuantity": 1, + "labels": [ + {"key": "modelplane.ai/pool", "value": "gpu-l40s"}, + {"key": "modelplane.ai/gpu", "value": "nvidia-l40s"}, + {"key": "nvidia.com/gpu.deploy.device-plugin", "value": "false"}, + ], + "clusterIdSelector": {"matchControllerRef": True}, + "taints": [{"key": "nvidia.com/gpu", "value": "true", "effect": "NoSchedule"}], + "autoScaler": True, + "minNodes": 1, + "maxNodes": 4, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ), + "gpu-observer": fnv1.Resource( + resource=resource.dict_to_struct( + { + "apiVersion": "kubernetes.m.crossplane.io/v1alpha1", + "kind": "Object", + "metadata": {"namespace": "modelplane-system"}, + "spec": { + "managementPolicies": ["Observe"], + "providerConfigRef": { + "kind": "ProviderConfig", + "name": "test-cluster-kubeconfig-55b57", + }, + "readiness": { + "policy": "DeriveFromCelQuery", + "celQuery": ( + "has(object.status.numberReady)" + " && object.status.desiredNumberScheduled >= 1" + " && object.status.numberReady == object.status.desiredNumberScheduled" + ), + }, + "forProvider": { + "manifest": { + "apiVersion": "apps/v1", + "kind": "DaemonSet", + "metadata": { + "name": "nvidia-operator-validator", + "namespace": "gpu-operator", + }, + }, + }, + }, + "status": { + "conditions": [ + { + "type": "Ready", + "status": "True", + "reason": "Available", + "lastTransitionTime": "2024-01-01T00:00:00Z", + }, + ], + }, + } + ), + ), + }, + ), ), want=fnv1.RunFunctionResponse( meta=fnv1.ResponseMeta(ttl=durationpb.Duration(seconds=60)), desired=fnv1.State( - composite=fnv1.Resource(resource=resource.dict_to_struct(_status())), + composite=_desired_xr(), resources={ - "cluster": fnv1.Resource( - resource=resource.dict_to_struct(_cluster()), - ready=fnv1.READY_TRUE, - ), - "node-pool-gpu-l40s": fnv1.Resource( - resource=resource.dict_to_struct(_GPU_POOL_GOLDEN), - ready=fnv1.READY_TRUE, - ), - "provider-config-kubernetes": fnv1.Resource( - resource=resource.dict_to_struct(_provider_config()), - ready=fnv1.READY_TRUE, - ), - "gpu-observer": fnv1.Resource( - resource=resource.dict_to_struct(_gpu_observer()), + "cluster": _cluster(ready=fnv1.READY_TRUE), + "node-pool-gpu-l40s": _gpu_node_pool( + node_quantity=1, + autoscaling={"autoScaler": True, "minNodes": 1, "maxNodes": 4}, ready=fnv1.READY_TRUE, ), + "provider-config-kubernetes": _provider_config(), + "gpu-observer": _gpu_observer(ready=fnv1.READY_TRUE), }, ), context=structpb.Struct(), @@ -531,11 +674,6 @@ def _observed_unready(desired: dict) -> fnv1.Resource: ] -def _to_dict(msg: message.Message) -> dict: - """msg as a dict with sorted keys, so pytest's diff of two lines them up.""" - return json.loads(json_format.MessageToJson(msg, sort_keys=True)) - - @pytest.mark.parametrize("case", COMPOSE_CASES, ids=lambda case: case.name) def test_compose(case: Case) -> None: """RunFunction composes VKE cluster infrastructure.""" From 82c16c7ca9c5b685f5c346fcfdd2ab610a95ccf0 Mon Sep 17 00:00:00 2001 From: Nic Cope Date: Wed, 30 Sep 2026 21:44:55 -0700 Subject: [PATCH 4/7] Build the host Python package set in its own file The checks build their virtualenvs from a uv2nix package set that checks.nix defines for itself. The e2e app needs a virtualenv from the same set, so this commit moves the set to nix/python.nix, and flake.nix passes it to the checks. Signed-off-by: Nic Cope --- flake.nix | 18 ++++++++++-------- nix/checks.nix | 14 +------------- nix/python.nix | 19 +++++++++++++++++++ 3 files changed, 30 insertions(+), 21 deletions(-) create mode 100644 nix/python.nix diff --git a/flake.nix b/flake.nix index 66c39f777..ad3b03836 100644 --- a/flake.nix +++ b/flake.nix @@ -112,14 +112,16 @@ checks = forAllSystems ( { pkgs, ... }: import ./nix/checks.nix { - inherit - pkgs - self - functionNames - pyproject-nix - uv2nix - pyproject-build-systems - ; + inherit pkgs self functionNames; + pythonSet = import ./nix/python.nix { + inherit + pkgs + self + pyproject-nix + uv2nix + pyproject-build-systems + ; + }; } ); diff --git a/nix/checks.nix b/nix/checks.nix index 6ced53cc6..04e7d4e8f 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -7,23 +7,11 @@ pkgs, self, functionNames, - pyproject-nix, - uv2nix, - pyproject-build-systems, + pythonSet, }: let docs = import ./docs.nix { inherit pkgs self; }; - workspace = uv2nix.lib.workspace.loadWorkspace { workspaceRoot = self; }; - pythonSet = - (pkgs.callPackage pyproject-nix.build.packages { python = pkgs.python312; }).overrideScope - ( - pkgs.lib.composeManyExtensions [ - pyproject-build-systems.overlays.wheel - (workspace.mkPyprojectOverlay { sourcePreference = "wheel"; }) - ] - ); - # Each function exports a 'function' Python module, so tests must run from # a directory where that module is importable via the venv, and one pytest # session can't hold two functions' tests. We copy tests/ from the source diff --git a/nix/python.nix b/nix/python.nix new file mode 100644 index 000000000..740b9e7d3 --- /dev/null +++ b/nix/python.nix @@ -0,0 +1,19 @@ +# The uv workspace's Python packages, built by uv2nix from uv.lock, for +# virtualenvs that run on the build host. The function images build their own, +# per target architecture (see functions.nix). +{ + pkgs, + self, + pyproject-nix, + uv2nix, + pyproject-build-systems, +}: +let + workspace = uv2nix.lib.workspace.loadWorkspace { workspaceRoot = self; }; +in +(pkgs.callPackage pyproject-nix.build.packages { python = pkgs.python312; }).overrideScope ( + pkgs.lib.composeManyExtensions [ + pyproject-build-systems.overlays.wheel + (workspace.mkPyprojectOverlay { sourcePreference = "wheel"; }) + ] +) From c0ff0bd59de56f2d59419847b21fe9b1b8889c09 Mon Sep 17 00:00:00 2001 From: Nic Cope Date: Wed, 30 Sep 2026 22:55:05 -0700 Subject: [PATCH 5/7] Write the e2e tests as a pytest suite The local e2e was a shell script, e2e/run.sh. Most of it checked the running environment, with polling loops written out by hand, JSON matched as text, and an exit at the first failed check, so one fault hid every check after it. This commit replaces run.sh with a Python package. environment.py brings the two clusters up and tears them down as run.sh did, and a pytest suite makes the checks run.sh's --verify made, one test each. Each test waits only for what it needs, so a gateway that refuses every caller fails only the tests that need it to serve. nix run .#e2e keeps its flags. Four checks are stricter. The response must name the served model in its model field, where the name anywhere in the body passed. An unclaimed model must get a 404, where any status but 200 passed. /v1/models must list the model by its exact name, where a substring of another name passed. And the usage record must be a new one with the full endpoint name, where any earlier matching record passed. Towards #473. Signed-off-by: Nic Cope --- .github/workflows/e2e.yml | 4 +- CONTRIBUTING.md | 5 +- e2e/README.md | 58 ++- e2e/__init__.py | 15 + e2e/client.yaml | 22 ++ e2e/conftest.py | 167 ++++++++ e2e/dra-example-driver.yaml | 4 +- e2e/environment.py | 291 ++++++++++++++ e2e/gateway.py | 102 +++++ e2e/kube.py | 131 +++++++ e2e/lean-control-plane.yaml | 6 +- e2e/manifests/00-namespaces.yaml | 5 +- e2e/manifests/10-inference-gateway.yaml | 4 +- e2e/manifests/20-inference-class.yaml | 2 +- e2e/manifests/30-inference-cluster.yaml | 2 +- e2e/run.sh | 496 ------------------------ e2e/test_serving.py | 178 +++++++++ e2e/wait.py | 49 +++ flake.nix | 11 +- nix/apps.nix | 66 ++-- nix/checks.nix | 32 +- pyproject.toml | 4 +- 22 files changed, 1094 insertions(+), 560 deletions(-) create mode 100644 e2e/__init__.py create mode 100644 e2e/client.yaml create mode 100644 e2e/conftest.py create mode 100644 e2e/environment.py create mode 100644 e2e/gateway.py create mode 100644 e2e/kube.py delete mode 100644 e2e/run.sh create mode 100644 e2e/test_serving.py create mode 100644 e2e/wait.py diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml index dae613934..70d2ae20e 100644 --- a/.github/workflows/e2e.yml +++ b/.github/workflows/e2e.yml @@ -66,8 +66,8 @@ jobs: # The exact command a developer runs locally (--print-build-logs is only CI # verbosity), so a green run here and a green run on a laptop mean the same - # thing. It brings up both clusters, deploys the mock model, waits for the - # ModelService, and asserts a live 200 — exiting non-zero on any failure. + # thing. It brings up both clusters, deploys the mock model, and runs the + # tests in e2e/, exiting non-zero if any fails. - name: Run e2e run: nix run .#e2e --print-build-logs -- --verify diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index de878e32a..da1c9d377 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -105,7 +105,8 @@ curl -fsSL https://install.determinate.systems/nix | sh -s -- install `nix flake check` runs all of the project's checks inside the Nix sandbox: Python, shell, and Nix linters and formatters, the [ty](https://docs.astral.sh/ty) -type checker on every composition function, plus unit tests for every function. +type checker on every composition function and the end-to-end tests, plus unit +tests for every function. Run `nix flake show` to see what else is available. ```bash @@ -120,7 +121,7 @@ composition function renders the right resources. The integration layer is `nix run .#e2e`, which brings up two local `kind` clusters and runs the whole path — scheduling, the serving-stack install on a registered cluster, gateway routing, a live request — with no cloud credentials. Add `-- --verify` -and it waits for readiness, asserts a 200, and exits non-zero on failure. That +and it runs the pytest suite in `e2e/`, exiting non-zero if a test fails. That verify command is what the label-gated `E2E` workflow runs on CI (add the `test-e2e` label to a PR), so a green local `--verify` and a green CI run mean the same thing. See `e2e/README.md`. diff --git a/e2e/README.md b/e2e/README.md index d8bac4ee5..126290441 100644 --- a/e2e/README.md +++ b/e2e/README.md @@ -80,10 +80,10 @@ runs, so prefer a separate control plane for cloud work. (kube-prometheus-stack et al.) add up fast; a full Docker disk surfaces as `no space left on device`. Reclaim between runs with `docker builder prune -af` and `docker image prune -af`. -- Everything else — `kind`, `kubectl`, `curl`, `git`, and the `crossplane` CLI — - is provided by the flake via `nix run .#e2e`. +- Everything else — `kind`, `kubectl`, the `docker` CLI, Python and pytest, and + the `crossplane` CLI — is provided by the flake via `nix run .#e2e`. -The workload cluster is pinned to **k8s v1.34** (in `run.sh`) for the +The workload cluster is pinned to **k8s v1.34** (in `environment.py`) for the `resource.k8s.io` (DRA) APIs, on-by-default in 1.34: both the serving stack's NVIDIA DRA driver and the dra-example-driver register DeviceClasses, and the example driver publishes the `ResourceSlice`s the engine's `ResourceClaim` binds @@ -93,10 +93,19 @@ against. The control-plane cluster needs no DRA. ```bash nix run .#e2e # bring up both clusters + deploy the mock model -nix run .#e2e -- --verify # same, then wait for readiness and assert a live 200 +nix run .#e2e -- --verify # same, then run the tests nix run .#e2e -- --clean # tear both clusters down ``` +Arguments after `--verify` go to pytest, so `nix run .#e2e -- --verify -k usage` +runs only the tests whose names match. Bring-up reuses clusters that are already +up. To rerun the tests against an environment that's up, without bringing it up +again: + +```bash +uv run --isolated --package crossplane-models --group dev pytest e2e +``` + `crossplane project run` installs the config and applies the resources, then returns; the serving-stack install and model rollout reconcile in the background. So wait for the `ModelService` to become ready before curling. The gateway's @@ -122,16 +131,17 @@ kubectl run curl -n ml-team --rm -it --image=curlimages/curl@sha256:7c12af72ceb3 -d '{"model":"ml-team/mock","max_tokens":16,"messages":[{"role":"user","content":"hi"}]}' ``` -`--verify` runs both of those, among other checks, and exits non-zero on -failure. It's the exact command the `E2E` CI workflow runs, so a green `--verify` -locally and a green CI run mean the same thing; use the manual curls above to -poke the endpoints interactively. +`--verify` runs the tests in `test_serving.py`, which send both of those +requests among others, and exits non-zero if any fails. It's the exact command +the `E2E` CI workflow runs, so a green `--verify` locally and a green CI run +mean the same thing; use the manual curls above to poke the endpoints +interactively. ## How it's structured `nix run .#e2e` materialises the Nix-built function images and hands off to -`run.sh`, which does the cross-cluster orchestration that `crossplane project -run` flags can't express: +`environment.py`, which does the cross-cluster orchestration that `crossplane +project run` flags can't express: 1. Create the **workload** kind cluster (pinned v1.34). 2. Install MetalLB on it (the serving stack doesn't) with a pool inside the @@ -147,13 +157,20 @@ run` flags can't express: control-plane pods over the shared kind network), then apply the Modelplane manifests. -Everything the control plane needs is a declarative manifest; the shell in -`run.sh` is only the irreducible cross-cluster setup (a second cluster, its -MetalLB and DRA driver, and the cross-cluster kubeconfig). +Everything the control plane needs is a declarative manifest; `environment.py` +is only the irreducible cross-cluster setup (a second cluster, its MetalLB and +DRA driver, and the cross-cluster kubeconfig). With `--verify`, pytest then runs +`test_serving.py`. Its fixtures in `conftest.py` wait for the model to serve, +then start a curl pod on each cluster to send requests from. ``` e2e/ - run.sh # two-cluster orchestration + environment.py # two-cluster bring-up and teardown + conftest.py # fixtures: the clusters, curl pods, readiness + test_serving.py # the tests + kube.py, gateway.py # kubectl and curl helpers + wait.py # polling until a condition holds + client.yaml # the curl pod the tests send requests from dra-example-driver.yaml # vendored fake DRA GPU driver (applied to workload) manifests/ # applied to the control plane after setup 00-namespaces.yaml @@ -169,11 +186,11 @@ e2e/ - **MetalLB on the workload cluster.** Both gateways run there, and both need `LoadBalancer` addresses kind can't provide: the serving stack gates the cluster gateway's readiness on having one (`READY_CEL` in its `gateway.py`). - Nothing Modelplane composes installs MetalLB, so `run.sh` does, with a pool + Nothing Modelplane composes installs MetalLB, so bring-up does, with a pool inside the detected kind Docker subnet (see caveat) so the control plane can route to the addresses it hands out. - **Fake DRA driver.** A `claim: DRA` engine emits a `ResourceClaim`; with no DRA - driver it stays Pending and the pod never schedules. `run.sh` applies the + driver it stays Pending and the pod never schedules. Bring-up applies the vendored **dra-example-driver**, which publishes fake `gpu.example.com` devices so the claim binds on a GPU-less node. - **Cross-cluster kubeconfig.** `source: Existing` needs a kubeconfig the @@ -181,12 +198,12 @@ e2e/ `kind get kubeconfig --internal` gives an address routable across the shared kind network; a host kubeconfig (`127.0.0.1:`) wouldn't be. - **Node label.** On a BYO cluster Modelplane doesn't provision/label pools, so - `run.sh` labels the workload node `modelplane.ai/pool=gpu-synthetic` (matching + bring-up labels the workload node `modelplane.ai/pool=gpu-synthetic` (matching `nodePools[].name`); without it worker pods stay Pending. ## Caveats / open questions -- **Cross-cluster networking uses the detected kind subnet.** `run.sh` reads the +- **Cross-cluster networking uses the detected kind subnet.** Bring-up reads the `kind` Docker network's subnet (usually 172.18.0.0/16, but kind bumps to 172.19/... when earlier networks already hold 172.18) and derives the workload cluster's MetalLB pool from it. A hardcoded 172.18 would leave the LB @@ -195,10 +212,11 @@ e2e/ for the config to install, then applies the resources and exits — it doesn't block on XR readiness. The serving-stack install (the long pole) and the model rollout happen after, so watch the `ModelService`'s `RoutingReady` rather than - the command's exit. `--timeout` in `run.sh` bounds the build and config install. + the command's exit. `--timeout` in `environment.py` bounds the build and config + install. - **Two DRA drivers on a GPU-less node.** The serving stack's **NVIDIA** DRA driver targets NFD-GPU-labelled nodes, so it sits at 0/0 (inert) yet its Helm - release still reports Ready. The **dra-example-driver** `run.sh` installs is the + release still reports Ready. The **dra-example-driver** bring-up installs is the active one — it publishes the fake `gpu.example.com` devices the engine binds. - **Serving-stack weight.** cert-manager, Envoy Gateway, Envoy AI Gateway, GAIE CRDs, kube-prometheus-stack, LeaderWorkerSet, NFD, DRA driver — all on the diff --git a/e2e/__init__.py b/e2e/__init__.py new file mode 100644 index 000000000..ac2a31de4 --- /dev/null +++ b/e2e/__init__.py @@ -0,0 +1,15 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""End-to-end tests that bring Modelplane up on kind and send it traffic. See README.md.""" diff --git a/e2e/client.yaml b/e2e/client.yaml new file mode 100644 index 000000000..9d596a22e --- /dev/null +++ b/e2e/client.yaml @@ -0,0 +1,22 @@ +# A pod to send requests to the gateways from. Their addresses are on the kind +# Docker network, which a macOS host can't route to but a pod can. The tests +# apply this to both clusters, and exec curl in it. +apiVersion: v1 +kind: Namespace +metadata: + name: e2e +--- +apiVersion: v1 +kind: Pod +metadata: + name: curl + namespace: e2e +spec: + containers: + - name: curl + # Pinned by digest (a multi-arch manifest list) so a moving tag can't flake + # the tests. + image: curlimages/curl@sha256:7c12af72ceb38b7432ab85e1a265cff6ae58e06f95539d539b654f2cfa64bb13 + command: ["sleep", "infinity"] + # As PID 1, sleep ignores SIGTERM, so don't wait for it to exit. + terminationGracePeriodSeconds: 0 diff --git a/e2e/conftest.py b/e2e/conftest.py new file mode 100644 index 000000000..50bb1a46d --- /dev/null +++ b/e2e/conftest.py @@ -0,0 +1,167 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Fixtures the end-to-end tests share. See README.md. + +The tests run against clusters that are already up. `nix run .#e2e -- --verify` +brings them up first. +""" + +import pathlib +from collections.abc import Iterator + +import pytest +from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 +from models.ai.modelplane.inferencegateway import v1alpha1 as igv1alpha1 +from models.ai.modelplane.modelservice import v1alpha1 as msv1alpha1 + +from e2e import environment, gateway, kube, wait + +CLIENT = pathlib.Path(__file__).parent / "client.yaml" + + +@pytest.fixture(scope="session") +def control_plane() -> kube.Cluster: + """The control plane, running Crossplane and Modelplane.""" + return kube.Cluster(environment.CONTROL_PLANE_CONTEXT) + + +@pytest.fixture(scope="session") +def workload() -> kube.Cluster: + """The workload cluster, running the serving stack, both gateways and the model.""" + return kube.Cluster(environment.WORKLOAD_CONTEXT) + + +@pytest.fixture(scope="session") +def control_plane_client(control_plane: kube.Cluster) -> Iterator[gateway.Client]: + """A pod on the control plane, which sends requests across clusters to the InferenceGateway.""" + yield from client(control_plane) + + +@pytest.fixture(scope="session") +def workload_client(workload: kube.Cluster) -> Iterator[gateway.Client]: + """A pod on the workload cluster, which can resolve the cluster gateway's Service name.""" + yield from client(workload) + + +def client(cluster: kube.Cluster) -> Iterator[gateway.Client]: + """Start a curl pod on a cluster, and delete it afterwards.""" + try: + cluster.apply(CLIENT) + cluster.wait_for_condition("pod", "curl", "e2e", "Ready", timeout=2 * 60) + yield gateway.Client(cluster, "e2e", "curl") + finally: + # Wait, so a run that follows doesn't create the pod in a namespace + # that's still terminating. + cluster.delete("namespace", "e2e", None) + cluster.wait_until_gone("namespace", "e2e", None, timeout=2 * 60) + + +@pytest.fixture(scope="session") +def routed(control_plane: kube.Cluster, workload: kube.Cluster) -> gateway.Serving: + """Wait for the InferenceGateway to route ModelService ml-team/mock. + + Bring-up returns once Modelplane is installed and the manifests are applied, + so the serving stack and the model are still reconciling. On an environment + that's already up, each wait returns at once. + """ + # RoutingReady means the route is composed and applied on every gateway + # serving the ModelService. Its status.model and the gateway's endpoints + # both publish before that, without a replica, so waiting on either would + # start sending requests while the engine is still rolling out. + obj = control_plane.wait_for_condition("modelservice", "mock", "ml-team", "RoutingReady", timeout=20 * 60) + ms = msv1alpha1.ModelService.model_validate(obj) + assert ms.status is not None + assert ms.status.model is not None, "ModelService ml-team/mock is RoutingReady but publishes no model name" + + # AI Gateway rolls the gateway's proxy pods once the first route reaches + # it, to stamp them with the hash of its sidecar's config, so a fresh + # gateway is still replacing its pods when the route goes ready. Requests + # can fail during that rollout, which starts only once AI Gateway has seen + # the route. So wait for the stamp, then for the rollout. + def proxies_stamped() -> None: + deployments = workload.list_objects("deployment", gateway.PROXY_NAMESPACE, gateway.PROXY_SELECTOR) + annotations = [d["spec"]["template"]["metadata"].get("annotations", {}) for d in deployments] + assert annotations, "the InferenceGateway has no proxy Deployment" + assert all("aigateway.envoyproxy.io/extproc-config-hash" in a for a in annotations), ( + "AI Gateway hasn't stamped the InferenceGateway's proxy pods" + ) + + wait.until( + proxies_stamped, timeout=2 * 60, what="AI Gateway to stamp the InferenceGateway's proxy pods", retry=kube.RETRY + ) + workload.kubectl( + "rollout", "status", "deployment", f"--namespace={gateway.PROXY_NAMESPACE}", + f"--selector={gateway.PROXY_SELECTOR}", "--timeout=5m", + timeout=6 * 60, + ) # fmt: skip + + def endpoints_published() -> igv1alpha1.Endpoints: + obj = control_plane.get("inferencegateway", "local", None) + assert obj is not None, "InferenceGateway local doesn't exist" + ig = igv1alpha1.InferenceGateway.model_validate(obj) + assert ig.status is not None + assert ig.status.endpoints is not None, "InferenceGateway local publishes no endpoints" + return ig.status.endpoints + + endpoints = wait.until( + endpoints_published, timeout=5 * 60, what="InferenceGateway local to publish its endpoints", retry=kube.RETRY + ) + assert endpoints.openAI is not None, "InferenceGateway local publishes no OpenAI endpoint" + assert endpoints.anthropic is not None, "InferenceGateway local publishes no Anthropic endpoint" + return gateway.Serving(model=ms.status.model, openai=endpoints.openAI, anthropic=endpoints.anthropic) + + +@pytest.fixture(scope="session") +def serving(routed: gateway.Serving, control_plane_client: gateway.Client) -> gateway.Serving: + """Wait for the InferenceGateway to serve ModelService ml-team/mock to caller e2e. + + The gateway can publish its endpoints a moment before the route serves, and + a slow CI runner widens that gap. Tests that expect a refusal use routed + instead, so a gateway that refuses everyone still fails only the tests that + expect it to serve. + """ + + def serves() -> None: + r = control_plane_client.request( + f"{routed.openai}/chat/completions", + {"authorization": f"Bearer {gateway.CALLER_KEY}"}, + {"model": routed.model, "messages": [{"role": "user", "content": "ping"}]}, + ) + assert r.status == 200, r + + wait.until(serves, timeout=2 * 60, what=f"the InferenceGateway to serve {routed.model}", retry=kube.RETRY) + return routed + + +@pytest.fixture(scope="session") +def cluster_gateway(control_plane: kube.Cluster) -> str: + """Wait for the gateway fronting the engines on the workload cluster, and return its hostname. + + InferenceCluster local publishes the hostname once the gateway requires + client certificates. + """ + + def published() -> str: + obj = control_plane.get("inferencecluster", "local", None) + assert obj is not None, "InferenceCluster local doesn't exist" + ic = icv1alpha1.InferenceCluster.model_validate(obj) + assert ic.status is not None + assert ic.status.gateway is not None, "InferenceCluster local publishes no gateway" + assert ic.status.gateway.hostname is not None, "InferenceCluster local publishes no gateway hostname" + return ic.status.gateway.hostname + + return wait.until( + published, timeout=20 * 60, what="InferenceCluster local to publish its gateway hostname", retry=kube.RETRY + ) diff --git a/e2e/dra-example-driver.yaml b/e2e/dra-example-driver.yaml index f81566700..793786c1d 100644 --- a/e2e/dra-example-driver.yaml +++ b/e2e/dra-example-driver.yaml @@ -11,8 +11,8 @@ # # then prepend the Namespace below. The template output already starts with # # a --- separator, so don't add another after the Namespace.) # -# run.sh applies this to the workload cluster. The kubeletplugin publishes 8 fake -# gpu.example.com devices via a ResourceSlice, so a `claim: DRA` engine's +# Bring-up applies this to the workload cluster. The kubeletplugin publishes 8 +# fake gpu.example.com devices via a ResourceSlice, so a `claim: DRA` engine's # ResourceClaim binds on a GPU-less node — exercising the real DRA allocation # path the fleet scheduler and composition emit, without a GPU. apiVersion: v1 diff --git a/e2e/environment.py b/e2e/environment.py new file mode 100644 index 000000000..7c514f5ff --- /dev/null +++ b/e2e/environment.py @@ -0,0 +1,291 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Bring up, and tear down, the two kind clusters the end-to-end tests run on. + +The workload cluster runs the serving stack, both gateways and the model. The +control plane runs Crossplane and the Configuration, managed by `crossplane +project run`, and registers the workload cluster as InferenceCluster local with +source: Existing. See README.md. + + python -m e2e.environment up [--no-apply] + python -m e2e.environment down +""" + +import argparse +import ipaddress +import json +import logging +import os +import pathlib +import shlex +import subprocess +import tempfile + +log = logging.getLogger(__name__) + +ROOT = pathlib.Path(__file__).resolve().parent.parent + +CONTROL_PLANE = "modelplane-e2e-local" +WORKLOAD = "modelplane-e2e-workload" +CONTROL_PLANE_CONTEXT = f"kind-{CONTROL_PLANE}" +WORKLOAD_CONTEXT = f"kind-{WORKLOAD}" + +# Pinned so the workload cluster has the DRA APIs the serving stack's NVIDIA DRA +# driver needs (resource.k8s.io, GA in k8s 1.34). The control plane needs no DRA, +# so its image doesn't matter. v1.34.2 or newer: older kubelets deadlock on an +# idle DRA connection (k/k#133934). +WORKLOAD_NODE_IMAGE = "kindest/node:v1.34.8@sha256:02722c2dedddcfc00febf5d27fbeb9b7b2c14294c82109ff4a85d89ac9ba3256" +DEADLOCKING_KUBELETS = ("v1.34.0", "v1.34.1") + +METALLB_URL = "https://raw.githubusercontent.com/metallb/metallb/v0.14.8/config/manifests/metallb-native.yaml" +CROSSPLANE_VERSION = "2.4.0" + +MANIFESTS = ROOT / "e2e" / "manifests" + + +class BringUpError(Exception): + """The environment can't be brought up as asked.""" + + +def up(*, apply_manifests: bool) -> None: + """Bring up both clusters and install Modelplane, then apply the manifests the tests use.""" + up_workload() + up_control_plane() + if not apply_manifests: + log.info("Control plane ready. Skipped applying %s", MANIFESTS) + return + log.info("Applying %s", MANIFESTS) + kubectl(CONTROL_PLANE_CONTEXT, "apply", f"--filename={MANIFESTS}") + + +def up_workload() -> None: + """Create the workload cluster, and install what Modelplane expects a cluster to have already.""" + create_workload_cluster() + + # Both kind clusters share one Docker network. MetalLB hands out + # LoadBalancer addresses from it, and the control plane must route to them, + # so the pool has to sit inside the network's actual subnet. That's usually + # 172.18.0.0/16, but kind moves to 172.19 and beyond when another Docker + # network holds 172.18. The serving stack doesn't install MetalLB, and the + # pool needs room for two Services, one per gateway. + prefix = kind_subnet_prefix() + log.info("Installing MetalLB on the workload cluster, with pool %s.255.100-149", prefix) + kubectl(WORKLOAD_CONTEXT, "apply", f"--filename={METALLB_URL}") + kubectl(WORKLOAD_CONTEXT, "rollout", "status", "--namespace=metallb-system", "deploy/controller", "--timeout=180s") + pool = { + "apiVersion": "v1", + "kind": "List", + "items": [ + { + "apiVersion": "metallb.io/v1beta1", + "kind": "IPAddressPool", + "metadata": {"name": "kind-pool", "namespace": "metallb-system"}, + "spec": {"addresses": [f"{prefix}.255.100-{prefix}.255.149"]}, + }, + { + "apiVersion": "metallb.io/v1beta1", + "kind": "L2Advertisement", + "metadata": {"name": "kind-l2", "namespace": "metallb-system"}, + "spec": {"ipAddressPools": ["kind-pool"]}, + }, + ], + } + kubectl(WORKLOAD_CONTEXT, "apply", "--filename=-", stdin=json.dumps(pool)) + + # Fake DRA GPUs, so a `claim: DRA` engine's ResourceClaim binds on this + # GPU-less node. Without a DRA driver the claim stays Pending and the engine + # never schedules, and the fleet scheduler rejects an engine whose only + # device is Synthetic. + log.info("Installing dra-example-driver (fake GPUs) on the workload cluster") + kubectl(WORKLOAD_CONTEXT, "apply", f"--filename={ROOT / 'e2e' / 'dra-example-driver.yaml'}") + kubectl( + WORKLOAD_CONTEXT, + "rollout", "status", "--namespace=dra-example-driver", "ds/dra-example-driver-kubeletplugin", "--timeout=120s", + ) # fmt: skip + + # Modelplane doesn't label a BYO cluster's nodes. The gpu-synthetic pool + # selects on this. + log.info("Labelling the workload node for pool gpu-synthetic") + kubectl( + WORKLOAD_CONTEXT, + "label", + "node", + f"{WORKLOAD}-control-plane", + "modelplane.ai/pool=gpu-synthetic", + "--overwrite", + ) + + +def create_workload_cluster() -> None: + """Create the workload cluster, or reuse one running the pinned Kubernetes minor version.""" + if WORKLOAD not in output("kind", "get", "clusters").split(): + log.info("Creating workload cluster %s (k8s v1.34, for DRA)", WORKLOAD) + run("kind", "create", "cluster", f"--name={WORKLOAD}", f"--image={WORKLOAD_NODE_IMAGE}") + return + + # An older cluster lacks the DRA APIs, and would fail the run confusingly + # later. + try: + version = output( + "kubectl", f"--context={WORKLOAD_CONTEXT}", + "get", "nodes", "--output=jsonpath={.items[0].status.nodeInfo.kubeletVersion}", + ) # fmt: skip + except subprocess.CalledProcessError: + version = "unreachable" + if version in DEADLOCKING_KUBELETS: + msg = ( + f"workload cluster {WORKLOAD} is {version}, whose kubelet deadlocks on an idle DRA connection " + "(fixed in v1.34.2); recreate it with: nix run .#e2e -- --clean" + ) + raise BringUpError(msg) + if not version.startswith("v1.34."): + msg = ( + f"workload cluster {WORKLOAD} is {version}, but v1.34 is required for the DRA APIs; " + "recreate it with: nix run .#e2e -- --clean" + ) + raise BringUpError(msg) + log.info("Reusing workload cluster %s (%s)", WORKLOAD, version) + + +def kind_subnet_prefix() -> str: + """Return the first two octets of the kind Docker network's IPv4 subnet.""" + subnets = output( + "docker", "network", "inspect", "kind", "--format={{range .IPAM.Config}}{{println .Subnet}}{{end}}" + ) + for subnet in subnets.split(): + network = ipaddress.ip_network(subnet) + if isinstance(network, ipaddress.IPv4Network): + log.info("kind Docker subnet is %s", network) + return ".".join(str(network.network_address).split(".")[:2]) + msg = f"could not find an IPv4 subnet on the kind Docker network: {subnets!r}" + raise BringUpError(msg) + + +def up_control_plane() -> None: + """Build Modelplane and install it on a kind control plane, then register the workload cluster with it.""" + log.info("Building and running the control plane %s", CONTROL_PLANE) + # The nix app runs with no system PATH, so a Docker config whose credsStore + # is "desktop" would break package resolution. The provider packages are + # public, so an empty config is enough. + with tempfile.TemporaryDirectory() as docker_config: + pathlib.Path(docker_config, "config.json").write_text("{}") + # The lean control plane's narrowed MRAP goes in with --init-resources, + # ahead of the providers, so the cloud providers stay dormant (safe-start + # scales them to zero). prerequisites.yaml can't go the same way: it + # opens with a comment-only YAML document, which `crossplane project run` + # rejects and kubectl skips. + run( + "crossplane", "project", "run", + f"--control-plane-name={CONTROL_PLANE}", "--cluster-admin", "--timeout=25m", + f"--init-resources={ROOT / 'e2e' / 'lean-control-plane.yaml'}", + f"--crossplane-version={CROSSPLANE_VERSION}", + env={"DOCKER_CONFIG": docker_config}, + ) # fmt: skip + + # Finish the setup the install guide does by hand: apply the RBAC + # prerequisites, then point the two providers at the DeploymentRuntimeConfigs + # they define. The providers install before prerequisites.yaml, and an + # ImageConfig binds only when a ProviderRevision is created, so provider-helm + # would otherwise come up without the RBAC it grants, and + # provider-kubernetes without --sanitize-secrets. + log.info("Applying the prerequisites and the provider runtime configs") + prerequisites = ROOT / "docs" / "manifests" / "install" / "prerequisites.yaml" + kubectl(CONTROL_PLANE_CONTEXT, "apply", f"--filename={prerequisites}") + for provider, runtime_config in ( + ("upbound-provider-helm", "provider-helm-modelplane"), + ("upbound-provider-kubernetes", "provider-kubernetes-modelplane"), + ): + patch = { + "spec": { + "runtimeConfigRef": { + "apiVersion": "pkg.crossplane.io/v1beta1", + "kind": "DeploymentRuntimeConfig", + "name": runtime_config, + } + } + } + kubectl( + CONTROL_PLANE_CONTEXT, + "patch", f"provider.pkg.crossplane.io/{provider}", "--type=merge", f"--patch={json.dumps(patch)}", + ) # fmt: skip + + # InferenceCluster local reads this kubeconfig to reach the workload + # cluster. --internal gives the address on the kind network, which the + # control plane's provider pods can reach and 127.0.0.1 isn't. + # prerequisites.yaml creates its namespace. + log.info("Registering the workload cluster's kubeconfig with the control plane") + secret = { + "apiVersion": "v1", + "kind": "Secret", + "metadata": {"name": "local-cluster-kubeconfig", "namespace": "modelplane-system"}, + "stringData": {"kubeconfig": output("kind", "get", "kubeconfig", "--internal", f"--name={WORKLOAD}")}, + } + kubectl(CONTROL_PLANE_CONTEXT, "apply", "--filename=-", stdin=json.dumps(secret)) + + +def down() -> None: + """Delete both clusters, and the control plane's local registry.""" + # Delete both clusters whatever project stop returns: it can exit 0 without + # removing the cluster. It's here for the local registry it also manages. + for cmd in ( + ("crossplane", "project", "stop", f"--control-plane-name={CONTROL_PLANE}"), + ("kind", "delete", "cluster", f"--name={CONTROL_PLANE}"), + ("kind", "delete", "cluster", f"--name={WORKLOAD}"), + ("docker", "rm", "--force", f"{CONTROL_PLANE}-registry"), + ): + try: + run(*cmd) + except subprocess.CalledProcessError as e: + log.warning("%s", e) + + +def kubectl(context: str, *args: str, stdin: str | None = None) -> None: + """Run kubectl against a cluster, logging what it prints.""" + run("kubectl", f"--context={context}", *args, stdin=stdin) + + +def run(*cmd: str, stdin: str | None = None, env: dict[str, str] | None = None) -> None: + """Run a command from the repository root, letting it print as it runs. + + Bring-up takes most of a CI run, so its progress shows live rather than + being captured and shown only if it fails. + """ + log.info("$ %s", shlex.join(cmd)) + subprocess.run(cmd, cwd=ROOT, env={**os.environ, **(env or {})}, input=stdin, text=True, check=True) + + +def output(*cmd: str) -> str: + """Run a command from the repository root, and return what it wrote to stdout.""" + return subprocess.run(cmd, cwd=ROOT, stdout=subprocess.PIPE, text=True, check=True).stdout.strip() + + +def main() -> None: + """Bring the environment up or down.""" + parser = argparse.ArgumentParser(prog="python -m e2e.environment", description=__doc__.split("\n\n")[0]) + commands = parser.add_subparsers(dest="command", required=True) + up_command = commands.add_parser("up", help="bring up both clusters, install Modelplane and apply the manifests") + up_command.add_argument("--no-apply", action="store_true", help="skip the manifests, to apply them by hand") + commands.add_parser("down", help="delete both clusters") + args = parser.parse_args() + + logging.basicConfig(level=logging.INFO, format="%(message)s") + if args.command == "down": + down() + return + up(apply_manifests=not args.no_apply) + + +if __name__ == "__main__": + main() diff --git a/e2e/gateway.py b/e2e/gateway.py new file mode 100644 index 000000000..f14efb1cf --- /dev/null +++ b/e2e/gateway.py @@ -0,0 +1,102 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Send requests through Modelplane's gateways, and read what they log. + +The gateways' addresses are on the kind Docker network, which a macOS host +can't route to. So requests go from a pod on one of the clusters, which can. +""" + +import dataclasses +import json +from typing import Any + +from e2e import kube + +# The InferenceGateway's Envoy proxy pods, on the workload cluster. +PROXY_NAMESPACE = "envoy-gateway-system" +PROXY_SELECTOR = "gateway.envoyproxy.io/owning-gateway-name=inference-gateway" + +# The key manifests/10-inference-gateway.yaml gives caller e2e. +CALLER_KEY = "sk-e2e-caller" + + +@dataclasses.dataclass(frozen=True) +class Serving: + """A model the InferenceGateway routes, and where to reach it.""" + + # The name a caller sends as the request's model: the ModelService's + # status.model. + model: str + openai: str + anthropic: str + + +@dataclasses.dataclass(frozen=True) +class Response: + """What came back from a request.""" + + status: int + body: str + + def json(self) -> Any: # noqa: ANN401 - a JSON body can decode to any type. + """Decode the body as JSON.""" + return json.loads(self.body) + + +@dataclasses.dataclass(frozen=True) +class Client: + """Runs curl in a pod on a cluster that can reach the gateways.""" + + cluster: kube.Cluster + namespace: str + pod: str + + def request(self, url: str, headers: dict[str, str], body: object | None = None) -> Response: + """Send a request, a POST of body as JSON if there is one and otherwise a GET.""" + curl = ["curl", "--silent", "--show-error", "--max-time", "15", "--write-out", "\n%{http_code}", url] + for name, value in headers.items(): + curl += ["--header", f"{name}: {value}"] + if body is not None: + curl += ["--header", "content-type: application/json", "--data", json.dumps(body)] + # curl exits non-zero only when no HTTP response came back, which fails + # the request rather than answering it. + out = self.cluster.kubectl("exec", f"--namespace={self.namespace}", self.pod, "--", *curl) + # --write-out puts the status on a line of its own, after the body. + text, _, status = out.rpartition("\n") + return Response(status=int(status), body=text) + + def connect(self, url: str) -> int: + """GET a URL without verifying the server's certificate, and return curl's exit code.""" + curl = ["curl", "--silent", "--show-error", "--insecure", "--max-time", "15", "--output", "/dev/null", url] + return self.cluster.run("exec", f"--namespace={self.namespace}", self.pod, "--", *curl).returncode + + +def usage_records(cluster: kube.Cluster) -> list[dict[str, Any]]: + """Return every usage record the InferenceGateway's proxy pods have logged. + + A request lands on any one of the proxy pods, so this reads them all. + """ + records = [] + for pod in cluster.list_objects("pod", PROXY_NAMESPACE, PROXY_SELECTOR): + logs = cluster.kubectl("logs", f"--namespace={PROXY_NAMESPACE}", pod["metadata"]["name"], "--container=envoy") + for line in logs.splitlines(): + # Envoy logs other things too. The access log is the JSON objects. + try: + record = json.loads(line) + except json.JSONDecodeError: + continue + if isinstance(record, dict) and "caller" in record: + records.append(record) + return records diff --git a/e2e/kube.py b/e2e/kube.py new file mode 100644 index 000000000..24a8ffb0b --- /dev/null +++ b/e2e/kube.py @@ -0,0 +1,131 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Read and change a cluster's resources with kubectl. + +Objects come back as the dicts kubectl's JSON output decodes to. Tests that +read Modelplane's own fields validate them into the generated models. +""" + +import dataclasses +import json +import pathlib +import shlex +import subprocess +from typing import Any + +from e2e import wait + +# A bound on each kubectl call, so a hung API server fails the test that hit it +# rather than the whole run. +KUBECTL_TIMEOUT_SECONDS = 60 + +type Object = dict[str, Any] + + +class KubectlError(Exception): + """kubectl exited non-zero, or timed out.""" + + +# What a wait retries: the condition not holding yet, or a kubectl call failing +# while the cluster converges. +RETRY = (AssertionError, KubectlError) + + +@dataclasses.dataclass(frozen=True) +class Cluster: + """A cluster, addressed by its kubeconfig context.""" + + context: str + + def run(self, *args: str, timeout: float = KUBECTL_TIMEOUT_SECONDS) -> subprocess.CompletedProcess[str]: + """Run kubectl against this cluster, and return how it exited.""" + cmd = ["kubectl", f"--context={self.context}", *args] + try: + return subprocess.run(cmd, capture_output=True, text=True, timeout=timeout, check=False) + except subprocess.TimeoutExpired as e: + msg = f"{shlex.join(cmd)}: timed out after {timeout:.0f}s" + raise KubectlError(msg) from e + + def kubectl(self, *args: str, timeout: float = KUBECTL_TIMEOUT_SECONDS) -> str: + """Run kubectl against this cluster, and return what it wrote to stdout.""" + result = self.run(*args, timeout=timeout) + if result.returncode != 0: + msg = f"{shlex.join(result.args)}: {result.stderr.strip()}" + raise KubectlError(msg) + return result.stdout + + def get(self, kind: str, name: str, namespace: str | None) -> Object | None: + """Return the named object, or None if it doesn't exist.""" + out = self.kubectl("get", kind, name, *namespaced(namespace), "--ignore-not-found", "--output=json") + return json.loads(out) if out else None + + def list_objects(self, kind: str, namespace: str, selector: str) -> list[Object]: + """Return the objects of a kind in a namespace that match a label selector.""" + out = self.kubectl("get", kind, f"--namespace={namespace}", f"--selector={selector}", "--output=json") + return json.loads(out)["items"] + + def apply(self, path: pathlib.Path) -> None: + """Apply a manifest file.""" + self.kubectl("apply", f"--filename={path}") + + def delete(self, kind: str, name: str, namespace: str | None) -> None: + """Delete an object without waiting for it to go. Deleting one that doesn't exist is a no-op.""" + self.kubectl("delete", kind, name, *namespaced(namespace), "--ignore-not-found", "--wait=false") + + def wait_for_condition(self, kind: str, name: str, namespace: str | None, condition: str, timeout: float) -> Object: + """Wait for an object's condition to be True, and return the object.""" + + def condition_is_true() -> Object: + obj = self.get(kind, name, namespace) + assert obj is not None, f"{kind} {qualified(namespace, name)} doesn't exist" + assert is_true(obj, condition), f"{describe(obj)} isn't {condition}" + return obj + + what = f"{kind} {qualified(namespace, name)} {condition}" + return wait.until(condition_is_true, timeout=timeout, what=what, retry=RETRY) + + def wait_until_gone(self, kind: str, name: str, namespace: str | None, timeout: float) -> None: + """Wait for an object to stop existing.""" + + def is_gone() -> None: + obj = self.get(kind, name, namespace) + assert obj is None, f"{describe(obj)} still exists" + + wait.until(is_gone, timeout=timeout, what=f"{kind} {qualified(namespace, name)} to be deleted", retry=RETRY) + + +def namespaced(namespace: str | None) -> list[str]: + """Return the kubectl flag that scopes a command to a namespace, if there is one.""" + return [f"--namespace={namespace}"] if namespace else [] + + +def qualified(namespace: str | None, name: str) -> str: + """Return namespace/name, or just name for a cluster-scoped object.""" + return f"{namespace}/{name}" if namespace else name + + +def is_true(obj: Object, condition: str) -> bool: + """Report whether an object's status condition is True.""" + return any(c["type"] == condition and c["status"] == "True" for c in obj.get("status", {}).get("conditions", [])) + + +def describe(obj: Object) -> str: + """Summarise an object and its status conditions on one line, for failure messages.""" + meta = obj["metadata"] + conditions = [] + for c in obj.get("status", {}).get("conditions", []): + why = ": ".join(v for v in (c.get("reason"), c.get("message")) if v) + conditions.append(f"{c['type']}={c['status']} ({why})" if why else f"{c['type']}={c['status']}") + return f"{obj['kind']} {qualified(meta.get('namespace'), meta['name'])}: {', '.join(conditions) or 'no conditions'}" diff --git a/e2e/lean-control-plane.yaml b/e2e/lean-control-plane.yaml index d02bea4a5..a8843bdee 100644 --- a/e2e/lean-control-plane.yaml +++ b/e2e/lean-control-plane.yaml @@ -3,9 +3,9 @@ # that only exist to *provision* clusters; here the workload cluster already # exists, so the compositions use only provider-kubernetes and provider-helm. # -# run.sh feeds this via `crossplane project run --init-resources`, which applies -# it BEFORE the Configuration's dependencies install, so the cloud providers -# never activate their managed resources or run their controllers. +# Bring-up feeds this via `crossplane project run --init-resources`, which +# applies it BEFORE the Configuration's dependencies install, so the cloud +# providers never activate their managed resources or run their controllers. # # Replace the Helm chart's default catch-all MRAP (activate: ["*"]) with one # that activates only the managed resources the BYO compositions use. The cloud diff --git a/e2e/manifests/00-namespaces.yaml b/e2e/manifests/00-namespaces.yaml index 96630af9d..a71a90288 100644 --- a/e2e/manifests/00-namespaces.yaml +++ b/e2e/manifests/00-namespaces.yaml @@ -1,6 +1,7 @@ -# The model resources live in ml-team; run.sh applies these manifests with +# The model resources live in ml-team; bring-up applies these manifests with # kubectl once the control plane is ready. modelplane-system is created earlier -# by prerequisites.yaml, before run.sh adds the workload kubeconfig Secret to it. +# by prerequisites.yaml, before bring-up adds the workload kubeconfig Secret to +# it. apiVersion: v1 kind: Namespace metadata: diff --git a/e2e/manifests/10-inference-gateway.yaml b/e2e/manifests/10-inference-gateway.yaml index b6d524121..29bcede14 100644 --- a/e2e/manifests/10-inference-gateway.yaml +++ b/e2e/manifests/10-inference-gateway.yaml @@ -3,7 +3,7 @@ # # Co-locating the InferenceGateway with the models it serves is a supported shape, # and the one this test uses: the workload cluster hosts both gateways, each with -# its own LoadBalancer address from the MetalLB run.sh installs there. +# its own LoadBalancer address from the MetalLB bring-up installs there. # # No TLS, so the gateway answers on its address over plain HTTP. It # authenticates callers by API key, against the Secret below. @@ -20,7 +20,7 @@ spec: matchLabels: modelplane.ai/inference-keys: "true" --- -# One caller, identified as e2e. run.sh sends its key in Authorization for +# One caller, identified as e2e. The tests send its key in Authorization for # OpenAI requests and in x-api-key for Anthropic ones. apiVersion: v1 kind: Secret diff --git a/e2e/manifests/20-inference-class.yaml b/e2e/manifests/20-inference-class.yaml index 146266dc9..d86864804 100644 --- a/e2e/manifests/20-inference-class.yaml +++ b/e2e/manifests/20-inference-class.yaml @@ -1,5 +1,5 @@ # A fake-GPU class backed by kubernetes-sigs/dra-example-driver (installed on the -# workload cluster by run.sh). claim: DRA is required — the fleet scheduler +# workload cluster by bring-up). claim: DRA is required — the fleet scheduler # rejects an engine whose only device is Synthetic, because it needs a claimable # device to bind a ResourceClaim through (see scheduling.py). The example driver # publishes fake gpu.example.com devices (memory 80Gi) with no real hardware, so diff --git a/e2e/manifests/30-inference-cluster.yaml b/e2e/manifests/30-inference-cluster.yaml index d44d9f1f0..1c9c16e49 100644 --- a/e2e/manifests/30-inference-cluster.yaml +++ b/e2e/manifests/30-inference-cluster.yaml @@ -1,4 +1,4 @@ -# Registers the separate workload kind cluster via source: Existing. run.sh +# Registers the separate workload kind cluster via source: Existing. Bring-up # creates that cluster, builds the referenced kubeconfig Secret in # modelplane-system (--internal address, reachable from the control plane's # provider pods over the shared kind network), and labels its node diff --git a/e2e/run.sh b/e2e/run.sh deleted file mode 100644 index 84662ebbb..000000000 --- a/e2e/run.sh +++ /dev/null @@ -1,496 +0,0 @@ -#!/usr/bin/env bash -# Two-cluster local e2e (no cloud, no GPU). Usually invoked via -# `nix run .#e2e` (which provides the tooling and the Nix-built function -# images). See README.md. -# -# Two clusters: -# - a workload kind cluster (this script creates it), registered via -# source: Existing, where the serving stack, the InferenceGateway and the -# model run; -# - a control-plane cluster (crossplane project run manages it) with crossplane -# + the config. -set -euo pipefail - -CP=modelplane-e2e-local -WL=modelplane-e2e-workload -# Pinned so the workload cluster has the DRA APIs the serving stack's NVIDIA DRA -# driver needs (resource.k8s.io, GA in k8s 1.34). The control-plane cluster that -# project run creates needs no DRA, so its image doesn't matter here. -# v1.34.2 or newer: older kubelets deadlock on an idle DRA connection (k/k#133934). -WL_NODE_IMAGE=kindest/node:v1.34.8@sha256:02722c2dedddcfc00febf5d27fbeb9b7b2c14294c82109ff4a85d89ac9ba3256 -METALLB_URL=https://raw.githubusercontent.com/metallb/metallb/v0.14.8/config/manifests/metallb-native.yaml -# Pinned by digest (a multi-arch manifest list) so a moving :latest can't flake -# the verify curl pod. -CURL_IMAGE=curlimages/curl@sha256:7c12af72ceb38b7432ab85e1a265cff6ae58e06f95539d539b654f2cfa64bb13 -ROOT="$(git rev-parse --show-toplevel)" - -log() { printf '\n\033[1;34m==> %s\033[0m\n' "$*"; } - -if [ "${1:-}" = "--clean" ]; then - # Always delete both clusters — don't gate kind delete on project stop's exit - # code (it can exit 0 without removing the cluster). project stop is - # best-effort for the local registry it also manages. - crossplane project stop --control-plane-name "$CP" 2>/dev/null || true - kind delete cluster --name "$CP" || true - kind delete cluster --name "$WL" || true - docker rm -f "${CP}-registry" >/dev/null 2>&1 || true - exit 0 -fi - -# One temp dir for everything this run creates (the isolated Docker config), -# removed on exit. mktemp -d gives a fresh unique path and the trap captures it -# on the next line, so the rm -rf can never reach a real directory. -work="$(mktemp -d)" -trap 'rm -rf "$work"' EXIT - -WLCTX="kind-$WL" -if kind get clusters 2>/dev/null | grep -qx "$WL"; then - # Reuse an existing workload cluster only if it's the pinned version. An - # older one lacks the DRA APIs and would fail the run confusingly later. - ver="$(kubectl --context "$WLCTX" get nodes -o jsonpath='{.items[0].status.nodeInfo.kubeletVersion}' 2>/dev/null || true)" - case "$ver" in - v1.34.0 | v1.34.1) - echo "workload cluster $WL is $ver, whose kubelet deadlocks on an idle DRA connection (fixed in v1.34.2); recreate it with: nix run .#e2e -- --clean" >&2 - exit 1 - ;; - v1.34.*) log "Reusing workload cluster $WL ($ver)" ;; - *) - echo "workload cluster $WL is ${ver:-unreachable}, but v1.34 is required for the DRA APIs; recreate it with: nix run .#e2e -- --clean" >&2 - exit 1 - ;; - esac -else - log "Creating workload cluster $WL (k8s v1.34, for DRA)" - kind create cluster --name "$WL" --image "$WL_NODE_IMAGE" -fi - -# Both kind clusters share one Docker network; MetalLB hands out LoadBalancer IPs -# from it, and the control plane must ROUTE to the workload gateways' IPs across -# it. So the pool must sit inside the *actual* kind subnet — normally -# 172.18.0.0/16, but kind bumps to 172.19/172.20/... when earlier Docker networks -# already hold 172.18. Detect it and derive the pool from its prefix; a hardcoded -# 172.18 leaves the LB IPs off-subnet and silently breaks cross-cluster routing -# (curl times out). -# `|| true` so a detection miss (grep finds nothing) doesn't trip set -e here — -# the explicit check below then reports it instead of an opaque abort. -SUBNET="$(docker network inspect kind -f '{{range .IPAM.Config}}{{println .Subnet}}{{end}}' | grep -E '^[0-9]+\.' | head -1 || true)" -PREFIX="$(printf '%s' "$SUBNET" | cut -d. -f1-2)" -[ -n "$PREFIX" ] || { - echo "could not detect the kind Docker subnet" >&2 - exit 1 -} -log "kind Docker subnet ${SUBNET} -> MetalLB pool ${PREFIX}.255.x" - -# The serving stack doesn't install MetalLB, so the workload cluster needs it -# here. Both gateways live on this cluster: the cluster gateway fronting the -# engine pods, and the InferenceGateway callers reach. The pool has to be big -# enough for two LoadBalancer Services. -log "Installing MetalLB on the workload cluster (pool ${PREFIX}.255.100-.149)" -kubectl --context "$WLCTX" apply -f "$METALLB_URL" -kubectl --context "$WLCTX" -n metallb-system rollout status deploy/controller --timeout=180s -kubectl --context "$WLCTX" apply -f - <"$docker_config/config.json" -export DOCKER_CONFIG="$docker_config" - -# Flags pick the mode: -# --no-apply install and finish the control plane but skip the model -# manifests — for gradual, manual apply/debugging. -# --verify after apply, wait for the ModelService and assert a live 200, -# exiting non-zero on failure. This is exactly what CI runs, so -# running it locally gives the same pass/fail signal (dev/CI parity). -manifests="$ROOT/e2e/manifests" -cpctx="kind-$CP" -apply_manifests=1 -verify=0 -case "${1:-}" in ---no-apply) apply_manifests=0 ;; ---verify) verify=1 ;; -esac - -log "Building + running the control plane" -cd "$ROOT" -# Install the config with the lean control-plane's narrowed MRAP applied before -# the providers, so the cloud providers stay dormant (safe-start scales them to -# zero). prerequisites.yaml is applied afterwards with kubectl, not through -# --init-resources: it opens with a comment-only YAML document that `crossplane -# project run` rejects but kubectl skips. -crossplane project run \ - --control-plane-name "$CP" --cluster-admin --timeout 25m \ - --init-resources "$ROOT/e2e/lean-control-plane.yaml" \ - --crossplane-version=2.4.0 - -# Config healthy. Finish the setup the install guide does by hand (as the -# nix run app now does too, PR #375): apply the RBAC prerequisites, then point -# the two providers at the DeploymentRuntimeConfigs they define. Providers -# install before prerequisites.yaml, and an ImageConfig binds only at -# ProviderRevision creation, so provider-helm otherwise comes up without the -# granted RBAC and provider-kubernetes without --sanitize-secrets. -log "Finishing control-plane setup: prerequisites + provider runtime configs" -kubectl --context "$cpctx" apply -f "$ROOT/docs/manifests/install/prerequisites.yaml" -kubectl --context "$cpctx" patch provider.pkg.crossplane.io upbound-provider-helm --type merge \ - -p '{"spec":{"runtimeConfigRef":{"apiVersion":"pkg.crossplane.io/v1beta1","kind":"DeploymentRuntimeConfig","name":"provider-helm-modelplane"}}}' -kubectl --context "$cpctx" patch provider.pkg.crossplane.io upbound-provider-kubernetes --type merge \ - -p '{"spec":{"runtimeConfigRef":{"apiVersion":"pkg.crossplane.io/v1beta1","kind":"DeploymentRuntimeConfig","name":"provider-kubernetes-modelplane"}}}' - -# The InferenceCluster (source: Existing) reads this kubeconfig to reach the -# workload cluster; --internal gives an address routable from the control plane's -# provider pods. It lives in modelplane-system, created by prerequisites.yaml above. -{ - printf 'apiVersion: v1\nkind: Secret\nmetadata: {name: local-cluster-kubeconfig, namespace: modelplane-system}\nstringData:\n kubeconfig: |\n' - kind get kubeconfig --internal --name "$WL" | sed 's/^/ /' -} | kubectl --context "$cpctx" apply -f - - -if [ "$apply_manifests" = 0 ]; then - log "--no-apply: control plane ready; apply manifests from $manifests" - exit 0 -fi - -# RBAC is in place, so the compositions can reach the workload cluster. Apply the -# model manifests. -kubectl --context "$cpctx" apply -f "$manifests/" - -if [ "$verify" = 0 ]; then - log "Done. Curl the ModelService per the README; clean up with: nix run .#e2e -- --clean" - exit 0 -fi - -# --verify: project run returns once the config is healthy and the resources are -# applied, so the serving-stack install and model rollout are still reconciling. -# Wait for the ModelService to report RoutingReady, then route a real request to -# the engine and assert a 200. Any failure exits non-zero — that is what makes -# this usable as a CI gate. -log "Verifying the model serves end to end" -ns=ml-team -svc=mock - -# Wait for the ModelService to report RoutingReady, which means its route is -# composed and applied on every gateway serving it. status.model and the -# gateway's endpoints both publish long before that - neither depends on a -# replica existing - so gating on either would start curling while the engine is -# still rolling out. -ready="" -for _ in $(seq 1 80); do - ready="$(kubectl --context "$cpctx" -n "$ns" get modelservice "$svc" \ - -o jsonpath='{.status.conditions[?(@.type=="RoutingReady")].status}' 2>/dev/null || true)" - [ "$ready" = "True" ] && break - sleep 15 -done -[ "$ready" = "True" ] || { - echo "verify: ModelService $ns/$svc never became RoutingReady" >&2 - kubectl --context "$cpctx" -n "$ns" get modelservice "$svc" -o jsonpath='{range .status.conditions[*]}{.type}={.status} {.reason}: {.message}{"\n"}{end}' >&2 || true - kubectl --context "$cpctx" -n "$ns" get modelendpoint -o wide >&2 || true - kubectl --context "$cpctx" -n "$ns" get modelreplica -o wide >&2 || true - exit 1 -} - -# AI Gateway rolls the gateway's proxy pods once the first route reaches it, to -# stamp them with the hash of its sidecar's config, so a fresh gateway is still -# replacing its pods when the route goes ready. Requests during that rollout -# can fail, so wait for it to finish before asserting anything. The rollout -# starts only once AI Gateway has seen the route, so first wait for the stamp. -proxy=gateway.envoyproxy.io/owning-gateway-name=inference-gateway -stamp="" -for _ in $(seq 1 40); do - stamp="$(kubectl --context "$WLCTX" -n envoy-gateway-system get deploy -l "$proxy" \ - -o jsonpath='{.items[0].spec.template.metadata.annotations.aigateway\.envoyproxy\.io/extproc-config-hash}' 2>/dev/null || true)" - [ -n "$stamp" ] && break - sleep 3 -done -[ -n "$stamp" ] || { - echo "verify: AI Gateway never stamped the InferenceGateway's proxy pods" >&2 - exit 1 -} -kubectl --context "$WLCTX" -n envoy-gateway-system rollout status deploy -l "$proxy" --timeout=5m || { - echo "verify: the InferenceGateway's proxy pods never finished rolling out" >&2 - kubectl --context "$WLCTX" -n envoy-gateway-system get pods -l "$proxy" -o wide >&2 || true - exit 1 -} - -# A caller names a ModelService as the request's model, so read it from status. -model="$(kubectl --context "$cpctx" -n "$ns" get modelservice "$svc" -o jsonpath='{.status.model}')" -[ -n "$model" ] || { - echo "verify: ModelService $ns/$svc published no model name" >&2 - exit 1 -} - -base="" -for _ in $(seq 1 80); do - base="$(kubectl --context "$cpctx" get inferencegateway local -o jsonpath='{.status.endpoints.openAI}' 2>/dev/null || true)" - [ -n "$base" ] && break - sleep 15 -done -[ -n "$base" ] || { - echo "verify: InferenceGateway local never published an OpenAI endpoint" >&2 - exit 1 -} -log "Gateway ${base}, model ${model}" - -# GET a URL from the workload cluster, reporting curl's own exit code rather -# than an HTTP status. Used to assert a request is refused before there is any -# HTTP response to report. -k skips server verification, so a non-zero exit is -# the server rejecting us rather than us rejecting its certificate. -wl_curl_exit() { - local pod="$1" url="$2" - kubectl --context "$WLCTX" -n default run "$pod" --restart=Never \ - --labels=app.kubernetes.io/name=e2e-verify --image="$CURL_IMAGE" \ - --command -- sh -c "curl -sS -k --max-time 15 -o /dev/null \"$url\"; echo EXIT=\$?" \ - >/dev/null 2>&1 || true - local c="" - for _ in $(seq 1 30); do - c="$(kubectl --context "$WLCTX" -n default logs "$pod" 2>/dev/null | sed -n 's/.*EXIT=\([0-9]*\).*/\1/p' || true)" - [ -n "$c" ] && break - sleep 2 - done - kubectl --context "$WLCTX" -n default delete pod "$pod" --now >/dev/null 2>&1 || true - printf '%s' "$c" -} - -# The address is on the kind Docker subnet the host can't route to on macOS, so -# curl from a pod on the control plane, reading the status from the pod's logs -# (not `run -i`, whose attach drops output on a headless runner). curl_status -# runs one throwaway pod per call and echoes the HTTP code; it polls the logs -# (curl writes the code once, then exits) so a failed attempt costs seconds, and -# a unique pod name per call keeps retries from reading a prior pod's output. -curl_status() { - local pod="$1" url="$2" - shift 2 - kubectl --context "$cpctx" -n "$ns" run "$pod" --restart=Never \ - --labels=app.kubernetes.io/name=e2e-verify --image="$CURL_IMAGE" \ - --command -- curl -sS --max-time 15 -o /dev/null -w '%{http_code}' "$url" "$@" \ - >/dev/null 2>&1 || true - local c="" - for _ in $(seq 1 30); do - c="$(kubectl --context "$cpctx" -n "$ns" logs "$pod" 2>/dev/null | tr -dc '0-9' || true)" - [ -n "$c" ] && break - sleep 2 - done - printf '%s' "$c" -} - -# curl_body is the same, but returns the response body. Used where the assertion -# is about what came back rather than only that something did. -curl_body() { - local pod="$1" url="$2" - shift 2 - kubectl --context "$cpctx" -n "$ns" run "$pod" --restart=Never \ - --labels=app.kubernetes.io/name=e2e-verify --image="$CURL_IMAGE" \ - --command -- curl -sS --max-time 15 "$url" "$@" >/dev/null 2>&1 || true - local b="" - for _ in $(seq 1 30); do - b="$(kubectl --context "$cpctx" -n "$ns" logs "$pod" 2>/dev/null || true)" - [ -n "$b" ] && break - sleep 2 - done - printf '%s' "$b" -} - -cleanup_verify_pods() { - kubectl --context "$cpctx" -n "$ns" delete pod -l app.kubernetes.io/name=e2e-verify --now >/dev/null 2>&1 || true -} - -# The gateway authenticates callers against the key in -# e2e/manifests/10-inference-gateway.yaml. OpenAI requests send it as a bearer -# token, Anthropic ones in x-api-key. -caller_key=sk-e2e-caller - -# OpenAI /v1/chat/completions, retried: the gateway can publish an endpoint a -# moment before the route is serving, and a slower CI runner widens that gap. -oai='{"model":"'"$model"'","messages":[{"role":"user","content":"ping"}]}' -code="" -for attempt in $(seq 1 10); do - code="$(curl_status "e2e-verify-oai-$attempt" "$base/chat/completions" -H "authorization: Bearer $caller_key" -H 'content-type: application/json' -d "$oai")" - log "verify attempt $attempt (OpenAI): HTTP ${code:-none}" - [ "$code" = "200" ] && break - sleep 10 -done -[ "$code" = "200" ] || { - echo "verify: $base/chat/completions did not return 200 within retries (last: ${code:-none})" >&2 - cleanup_verify_pods - exit 1 -} - -# The same request with no key, and with a key no Secret holds, is refused. -for nokey in none wrong; do - if [ "$nokey" = none ]; then - kcode="$(curl_status "e2e-verify-nokey-$nokey" "$base/chat/completions" -H 'content-type: application/json' -d "$oai")" - else - kcode="$(curl_status "e2e-verify-nokey-$nokey" "$base/chat/completions" -H 'authorization: Bearer sk-wrong' -H 'content-type: application/json' -d "$oai")" - fi - log "verify (key: ${nokey}): HTTP ${kcode:-none}" - [ "$kcode" = "401" ] || { - echo "verify: expected 401 for a request with key ${nokey}, got ${kcode:-none}" >&2 - cleanup_verify_pods - exit 1 - } -done - -# The engine only answers to the name Modelplane started it under, and rejects -# anything else with a 404. So a 200 above already proves the gateway rewrote the -# caller's ModelService name to the deployment's. Assert the response reports the -# served model rather than what the caller asked for, which is the visible half -# of the same mechanism. -body="$(curl_body e2e-verify-served "$base/chat/completions" -H "authorization: Bearer $caller_key" -H 'content-type: application/json' -d "$oai")" -case "$body" in -*'"model": "ml-team/mock-demo"'* | *'"model":"ml-team/mock-demo"'*) - log "verify (model rewriting): caller asked for ${model}, engine served ml-team/mock-demo" - ;; -*) - echo "verify: response did not report the served model; got: $body" >&2 - cleanup_verify_pods - exit 1 - ;; -esac - -# A model no ModelService claims must not route anywhere. Catches a route -# matching too broadly, which would send a caller to an arbitrary backend. -ncode="$(curl_status e2e-verify-unknown "$base/chat/completions" -H "authorization: Bearer $caller_key" \ - -H 'content-type: application/json' -d '{"model":"ml-team/nope","messages":[{"role":"user","content":"ping"}]}')" -log "verify (unknown model): HTTP ${ncode:-none}" -[ "$ncode" = "200" ] && { - echo "verify: an unclaimed model name was routed and served" >&2 - cleanup_verify_pods - exit 1 -} - -# The cluster gateway must refuse a caller that presents no client certificate. -# Every check above goes through the InferenceGateway, which holds a -# certificate, so none of them would notice this lapsing. A ClientTrafficPolicy -# that stopped applying, or an HTTP listener beside the HTTPS one, would leave -# the engines open to anything that can reach the load balancer. -# -# Run from the workload cluster, where the Service compose-inference-gateway -# composed resolves the gateway's name. A plain GET is enough: the handshake -# fails before any request is sent. The trailing dot skips the pod's search -# domains, which ndots:5 would otherwise try ahead of the name itself. A resolve -# failure returns curl 6, which the checks below reject rather than pass. -cluster_gw_name="$(kubectl --context "$cpctx" get inferencecluster local -o jsonpath='{.status.gateway.hostname}')" -cluster_gw="https://${cluster_gw_name}./v1/models" -ecode="$(wl_curl_exit e2e-verify-nocert "$cluster_gw")" -log "verify (cluster gateway, no client certificate): curl exit ${ecode:-none}" -case "$ecode" in -0) - echo "verify: the cluster gateway served a caller presenting no client certificate" >&2 - cleanup_verify_pods - exit 1 - ;; -35 | 52 | 55 | 56) ;; -*) - echo "verify: expected the cluster gateway to refuse an uncertified caller mid-handshake," >&2 - echo "verify: but curl failed with ${ecode:-no exit code}, which is a different failure" >&2 - cleanup_verify_pods - exit 1 - ;; -esac - -# And nothing on port 80. The serving HTTPRoutes carry no sectionName, so they -# attach to every listener there is, and an HTTP listener would serve the -# engines without a certificate. The gateway's only listener is HTTPS, and the -# load balancer publishes a port per listener, so the connection is refused. -hcode="$(wl_curl_exit e2e-verify-plaintext "http://${cluster_gw_name}./v1/models")" -log "verify (cluster gateway, plaintext): curl exit ${hcode:-none}" -case "$hcode" in -0) - echo "verify: the cluster gateway served plaintext HTTP on port 80" >&2 - cleanup_verify_pods - exit 1 - ;; -7 | 28 | 35 | 52 | 56) ;; -*) - echo "verify: expected no listener on port 80, but curl failed with ${hcode:-no exit code}," >&2 - echo "verify: which is a different failure" >&2 - cleanup_verify_pods - exit 1 - ;; -esac - -# /v1/models lists what this gateway serves. Only exact model matches appear, so -# this also proves the route matches exactly rather than by pattern. -models="$(curl_body e2e-verify-models "$base/models" -H "authorization: Bearer $caller_key")" -case "$models" in -*"$model"*) log "verify (/v1/models): lists ${model}" ;; -*) - echo "verify: /v1/models did not list $model; got: $models" >&2 - cleanup_verify_pods - exit 1 - ;; -esac - -# Anthropic's Messages API on the same gateway. The endpoint's API is OpenAI, so -# the gateway translates the request. The mock serves /v1/messages too, the way -# vLLM does, so a 200 alone doesn't tell translation from passthrough. The key -# goes in x-api-key, as Anthropic clients send it. -anthropic_base="${base%/v1}/anthropic/v1" -ant='{"model":"'"$model"'","max_tokens":16,"messages":[{"role":"user","content":"ping"}]}' -mcode="$(curl_status e2e-verify-anthropic "$anthropic_base/messages" -H "x-api-key: $caller_key" -H 'content-type: application/json' -H 'anthropic-version: 2023-06-01' -d "$ant")" -log "verify (Anthropic /v1/messages): HTTP ${mcode:-none}" -[ "$mcode" = "200" ] || { - echo "verify: $anthropic_base/messages did not return 200 (got: ${mcode:-none})" >&2 - cleanup_verify_pods - exit 1 -} - -# Assert the gateway emits a usage record attributing the request's tokens to -# its caller. Read it off the InferenceGateway's Envoy. It runs two proxy pods -# and each request lands on either, so read both. --tail has to be explicit, -# because with a selector kubectl logs keeps only the last 10 lines per pod. -usage="$(kubectl --context "$WLCTX" -n envoy-gateway-system logs \ - -l "$proxy" -c envoy --tail=200 2>/dev/null | - grep '"input_tokens":12' | tail -1 || true)" -# Check each field on its own. The access log serialises its keys -# alphabetically, so a single glob spanning two of them depends on that order. -# -# The endpoint is the ModelRoute's backend for it, in ml-team's mirrored -# namespace (child_name("mp", "ml-team")) and named after the route -# (child_name("mock", "local")) and the endpoint. -missing="" -for want in \ - '"caller":"e2e"' \ - '"service":"'"$model"'"' \ - '"endpoint":"mp-ml-team-51733/mock-local-' \ - '"served_model":"ml-team/mock-demo"' \ - '"input_tokens":12' \ - '"output_tokens":9' \ - '"total_tokens":21' \ - '"status":200'; do - case "$usage" in - *"$want"*) ;; - *) missing="$missing $want" ;; - esac -done -[ -z "$missing" ] || { - echo "verify: usage record missing:$missing" >&2 - echo "verify: record was: ${usage:-none}" >&2 - cleanup_verify_pods - exit 1 -} -log "verify (usage record): ${usage}" - -cleanup_verify_pods -log "End to end OK: ${base} authenticates callers, serves ${model} over OpenAI and Anthropic, rewrites the model, and meters it" diff --git a/e2e/test_serving.py b/e2e/test_serving.py new file mode 100644 index 000000000..3e49b4066 --- /dev/null +++ b/e2e/test_serving.py @@ -0,0 +1,178 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Test that the InferenceGateway serves ml-team/mock, and the cluster gateway guards it. + +Every request goes from a pod, because the gateways' addresses are on the kind +Docker network. See conftest.py for the fixtures. +""" + +import pytest + +from e2e import gateway, kube, wait + +# OpenAI requests send the caller's key as a bearer token. +AUTHORIZED = {"authorization": f"Bearer {gateway.CALLER_KEY}"} + + +def test_chat_completion_succeeds(serving: gateway.Serving, control_plane_client: gateway.Client) -> None: + """An OpenAI chat completion naming the ModelService returns 200.""" + r = control_plane_client.request( + f"{serving.openai}/chat/completions", + AUTHORIZED, + {"model": serving.model, "messages": [{"role": "user", "content": "ping"}]}, + ) + assert r.status == 200, r + + +@pytest.mark.parametrize("headers", [{}, {"authorization": "Bearer sk-wrong"}], ids=["no key", "a key no Secret holds"]) +def test_chat_completion_without_a_valid_key_is_refused( + routed: gateway.Serving, control_plane_client: gateway.Client, headers: dict[str, str] +) -> None: + """The gateway refuses a caller that presents no key, or one it doesn't know.""" + r = control_plane_client.request( + f"{routed.openai}/chat/completions", + headers, + {"model": routed.model, "messages": [{"role": "user", "content": "ping"}]}, + ) + assert r.status == 401, r + + +def test_response_reports_the_served_model(serving: gateway.Serving, control_plane_client: gateway.Client) -> None: + """The response names the model the engine serves, not the ModelService the caller asked for. + + The engine answers only to the name Modelplane started it under, and refuses + anything else with a 404. So a 200 already shows the gateway rewrote the + caller's ModelService name to the deployment's. This asserts the visible + half of the same mechanism. + """ + r = control_plane_client.request( + f"{serving.openai}/chat/completions", + AUTHORIZED, + {"model": serving.model, "messages": [{"role": "user", "content": "ping"}]}, + ) + assert r.status == 200, r + assert r.json()["model"] == "ml-team/mock-demo" + + +def test_unclaimed_model_is_not_routed(serving: gateway.Serving, control_plane_client: gateway.Client) -> None: + """A model no ModelService claims routes nowhere. + + This catches a route that matches too broadly, which would send a caller + to an arbitrary backend. A 404 only means that once the claimed name + serves, so this waits for it. + """ + r = control_plane_client.request( + f"{serving.openai}/chat/completions", + AUTHORIZED, + {"model": "ml-team/nope", "messages": [{"role": "user", "content": "ping"}]}, + ) + assert r.status == 404, r + + +def test_models_lists_the_model_service(serving: gateway.Serving, control_plane_client: gateway.Client) -> None: + """/v1/models lists the ModelService. + + It lists only models a route matches exactly, so this also shows the route + matches the name exactly rather than by pattern. + """ + r = control_plane_client.request(f"{serving.openai}/models", AUTHORIZED) + assert r.status == 200, r + assert serving.model in [m["id"] for m in r.json()["data"]], r + + +def test_anthropic_message_succeeds(serving: gateway.Serving, control_plane_client: gateway.Client) -> None: + """An Anthropic Messages API request returns 200. + + The endpoint's API is OpenAI, so the gateway translates the request. The + mock serves /v1/messages too, the way vLLM does, so a 200 alone doesn't + tell translation from passthrough. Anthropic clients send the key in + x-api-key. + """ + r = control_plane_client.request( + f"{serving.anthropic}/messages", + {"x-api-key": gateway.CALLER_KEY, "anthropic-version": "2023-06-01"}, + {"model": serving.model, "max_tokens": 16, "messages": [{"role": "user", "content": "ping"}]}, + ) + assert r.status == 200, r + + +def test_gateway_logs_a_usage_record( + serving: gateway.Serving, control_plane_client: gateway.Client, workload: kube.Cluster +) -> None: + """The InferenceGateway's access log attributes a request's tokens to its caller. + + The mock engine reports the same tokens for every request, so every request + the tests send logs an identical record. This counts the matching records + before its own request, and waits for one more. + """ + # The endpoint is the ModelRoute's backend for the ModelEndpoint, in + # ml-team's mirrored namespace on the workload cluster. + want = { + "caller": "e2e", + "service": "ml-team/mock", + "endpoint": "mp-ml-team-51733/mock-local-934fc-mock-demo-da96c-f8a13", + "served_model": "ml-team/mock-demo", + "input_tokens": 12, + "output_tokens": 9, + "total_tokens": 21, + "status": 200, + } + + def matching() -> int: + return sum(1 for record in gateway.usage_records(workload) if {k: record.get(k) for k in want} == want) + + before = matching() + r = control_plane_client.request( + f"{serving.openai}/chat/completions", + AUTHORIZED, + {"model": serving.model, "messages": [{"role": "user", "content": "ping"}]}, + ) + assert r.status == 200, r + + def logged() -> None: + assert matching() > before, f"no new usage record matching {want}" + + wait.until(logged, timeout=60, what="the InferenceGateway to log the request", retry=kube.RETRY) + + +def test_cluster_gateway_refuses_a_caller_without_a_client_certificate( + workload_client: gateway.Client, cluster_gateway: str +) -> None: + """The cluster gateway fronting the engines refuses a caller that presents no client certificate. + + Every other test goes through the InferenceGateway, which holds a + certificate, so none of them would notice this lapsing. A + ClientTrafficPolicy that stopped applying, or an HTTP listener beside the + HTTPS one, would leave the engines open to anything that can reach the load + balancer. The handshake fails before a request is sent, so this checks + curl's exit code. The trailing dot skips the pod's search domains, which + ndots:5 would otherwise try first. + """ + # 35: TLS handshake failed. 52: empty reply. 55 and 56: the connection broke + # mid-handshake. A refused or unresolved connection is a different failure. + assert workload_client.connect(f"https://{cluster_gateway}./v1/models") in {35, 52, 55, 56} + + +def test_cluster_gateway_has_no_plaintext_listener(workload_client: gateway.Client, cluster_gateway: str) -> None: + """The cluster gateway serves nothing over plain HTTP on port 80. + + The serving HTTPRoutes carry no sectionName, so they attach to every listener + there is, and an HTTP listener would serve the engines without a + certificate. The gateway's only listener is HTTPS, and the load balancer + publishes a port per listener, so a connection to port 80 is refused. + """ + # 7: connection refused. 28: timed out. The rest mean something answered + # port 80 without serving HTTP. + assert workload_client.connect(f"http://{cluster_gateway}./v1/models") in {7, 28, 35, 52, 56} diff --git a/e2e/wait.py b/e2e/wait.py new file mode 100644 index 000000000..0da1f29a7 --- /dev/null +++ b/e2e/wait.py @@ -0,0 +1,49 @@ +# Copyright 2026 The Modelplane Authors. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Wait for an eventually consistent system to converge.""" + +import logging +import time +from collections.abc import Callable + +log = logging.getLogger(__name__) + + +def until[T]( + check: Callable[[], T], + *, + timeout: float, + what: str, + retry: tuple[type[Exception], ...] = (AssertionError,), + interval: float = 5, +) -> T: + """Call check until it stops raising one of the retry exceptions, and return what it returns. + + Once timeout seconds pass, re-raise check's last exception, so the failure + says what was still wrong rather than only that time ran out. + """ + log.info("Waiting up to %ds for %s", timeout, what) + start = time.monotonic() + while True: + try: + result = check() + except retry as e: + if time.monotonic() - start + interval > timeout: + e.add_note(f"Still failing after waiting {timeout:.0f}s for {what}.") + raise + time.sleep(interval) + continue + log.info("Done after %.0fs", time.monotonic() - start) + return result diff --git a/flake.nix b/flake.nix index ad3b03836..ffc34e264 100644 --- a/flake.nix +++ b/flake.nix @@ -157,6 +157,15 @@ apps = import ./nix/apps.nix { inherit pkgs; }; crossplane = deps.crossplane { inherit system; }; functionsPkg = self.packages.${system}.functions or null; + pythonSet = import ./nix/python.nix { + inherit + pkgs + self + pyproject-nix + uv2nix + pyproject-build-systems + ; + }; in { fix = apps.fix { }; @@ -173,7 +182,7 @@ dockerCredentialUp = pkgs.upbound; }; stop = apps.stop { inherit crossplane; }; - e2e = apps.e2e { inherit crossplane functionsPkg; }; + e2e = apps.e2e { inherit crossplane functionsPkg pythonSet; }; stacks = apps.stacks { inherit (pkgs) aicr; }; } ); diff --git a/nix/apps.nix b/nix/apps.nix index 48c62ff1f..f646fb00a 100644 --- a/nix/apps.nix +++ b/nix/apps.nix @@ -31,7 +31,7 @@ -ignore '**/*.toml' \ -ignore '**/*.yaml' \ -ignore '**/*.yml' \ - functions/ docs/utils/validate/ nix.sh + functions/ docs/utils/validate/ e2e/ nix.sh echo "Formatting and linting Nix..." statix fix . @@ -46,8 +46,8 @@ find . -name '*.sh' -type f -exec shellcheck {} + echo "Formatting and linting Python..." - ruff format functions/ - ruff check --fix functions/ + ruff format functions/ e2e/ + ruff check --fix functions/ e2e/ echo "Refreshing uv.lock..." uv lock @@ -149,8 +149,8 @@ esac done - # Pin Crossplane to the version e2e/run.sh uses: without a pin the - # CLI installs the latest release + # Pin Crossplane to the version e2e/environment.py uses: without a + # pin the CLI installs the latest release version_args=(--crossplane-version=2.4.0) for arg in "$@"; do case "$arg" in @@ -296,21 +296,33 @@ ); }; - # Run the two-cluster local end-to-end test: a workload - # kind cluster registered via source: Existing (serving stack + model) and a - # control-plane cluster (crossplane + the InferenceGateway). Two clusters - # because the control-plane and workload layers both install the Gateway API - # CRDs and collide on a single cluster. See e2e. Tear down with - # `nix run .#e2e -- --clean`. This app just materialises the Nix-built - # function images (as `run` does), then hands off to run.sh, which needs real - # orchestration (a second cluster, a cross-cluster kubeconfig) that - # `crossplane project run` flags can't express — kept a normal shell file so - # it stays shellcheck-clean rather than escaped nix strings. + # Run the two-cluster local end-to-end test: a workload kind cluster + # registered via source: Existing (serving stack + model) and a control-plane + # cluster (crossplane + the Configuration). Two clusters because the + # control-plane and workload layers both install the Gateway API CRDs and + # collide on a single cluster. See e2e/README.md. + # + # With no argument it brings the environment up and applies the manifests, + # --no-apply stops short of the manifests, --verify then runs the tests in + # e2e/ (passing any further arguments to pytest), and --clean tears it all + # down. This app materialises the Nix-built function images (as `run` does) + # for crossplane project run to load. e2e = { crossplane, functionsPkg, + pythonSet, }: + let + # What e2e/ imports: pytest, and the generated models it reads + # Modelplane's status with. pydantic is declared here rather than on + # crossplane-models, whose pyproject.toml the Crossplane CLI generates. + venv = pythonSet.mkVirtualEnv "modelplane-e2e-env" { + pytest = [ ]; + crossplane-models = [ ]; + pydantic = [ ]; + }; + in { type = "app"; meta.description = "Run the local two-cluster end-to-end test"; @@ -319,16 +331,11 @@ name = "modelplane-e2e"; runtimeInputs = [ crossplane + venv pkgs.coreutils - pkgs.gnused - pkgs.gnugrep - pkgs.gawk pkgs.kind pkgs.kubectl - pkgs.curl pkgs.docker-client - pkgs.git - pkgs.bash ]; inheritPath = false; text = '' @@ -336,7 +343,22 @@ rm -f _output/functions ln -s ${functionsPkg} _output/functions - exec bash e2e/run.sh "$@" + case "''${1:-}" in + "") exec python -m e2e.environment up ;; + --no-apply) exec python -m e2e.environment up --no-apply ;; + --clean) exec python -m e2e.environment down ;; + --verify) + shift + python -m e2e.environment up + exec python -m pytest e2e \ + -o log_cli=true --log-cli-level=INFO "$@" + ;; + *) + echo "usage: nix run .#e2e" \ + "[-- --no-apply | --verify [pytest args...] | --clean]" >&2 + exit 2 + ;; + esac ''; } ); diff --git a/nix/checks.nix b/nix/checks.nix index 04e7d4e8f..01c1c991a 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -93,6 +93,28 @@ in touch $out/.docs-manifests-validated ''; + # Type-check the end-to-end tests with ty, against the packages the e2e app + # runs them with (see apps.nix). + ty-e2e = + let + venv = pythonSet.mkVirtualEnv "e2e-ty-env" { + pytest = [ ]; + crossplane-models = [ ]; + pydantic = [ ]; + }; + in + pkgs.runCommand "modelplane-ty-e2e" + { + nativeBuildInputs = [ pkgs.unstable.ty ]; + } + '' + cp -r ${self}/e2e e2e + cp ${self}/pyproject.toml pyproject.toml + ty check e2e --python ${venv} + mkdir -p $out + touch $out/.ty-passed + ''; + python = pkgs.runCommand "modelplane-python-checks" { @@ -102,8 +124,8 @@ in cp -r ${self} src chmod -R u+w src cd src - ruff format --check functions/ docs/utils/validate/ - ruff check functions/ docs/utils/validate/ + ruff format --check functions/ docs/utils/validate/ e2e/ + ruff check functions/ docs/utils/validate/ e2e/ mkdir -p $out touch $out/.python-checks-passed ''; @@ -137,8 +159,8 @@ in ''; # Fail if any hand-written source file is missing its Apache 2.0 license - # header. Scoped to the files we author: the composition functions and the - # docs manifest validator. Generated models under schemas/python carry their + # header. Scoped to the files we author: the composition functions, the docs + # manifest validator, and the end-to-end tests. Generated models under schemas/python carry their # own codegen banner, and config (*.toml) and vendored upstream CRDs (*.yaml) # are excluded. addlicense -check only reads, so it runs against the store # path directly. Run 'nix run .#fix' to add any missing headers. @@ -153,7 +175,7 @@ in -ignore '**/*.toml' \ -ignore '**/*.yaml' \ -ignore '**/*.yml' \ - functions/ docs/utils/validate/ nix.sh + functions/ docs/utils/validate/ e2e/ nix.sh mkdir -p $out touch $out/.license-check-passed ''; diff --git a/pyproject.toml b/pyproject.toml index 350e59cc8..347268664 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -76,8 +76,10 @@ allow-star-arg-any = true # Tests use magic values, many parameters, long hardcoded resource dicts, and # boolean fixture toggles passed positionally. "**/tests/**" = ["E501", "PLR2004", "PLR0913", "FBT001"] +# The end-to-end tests compare HTTP statuses and curl exit codes. +"e2e/**" = ["PLR2004"] # The pytest style rules are for tests. Elsewhere an assert guards an invariant. -"!**/tests/**" = ["PT"] +"!{**/tests/**,e2e/**}" = ["PT"] # fn.py uses gRPC's required PascalCase method name. "functions/*/function/fn.py" = ["N802"] # The docs manifest validator is a CLI script; print is its output. From ee65a5cec3378e83718a1612e841d82e1d93d15e Mon Sep 17 00:00:00 2001 From: Nic Cope Date: Wed, 30 Sep 2026 23:58:02 -0700 Subject: [PATCH 6/7] Run the unit and e2e tests with nix run Running one function's unit tests, or the e2e tests against clusters that were already up, meant entering the dev shell and driving uv, a second tool to learn alongside nix. This commit adds nix run .#test, which runs every function's unit tests, or one function's with pytest arguments after its name, against the virtualenvs the flake checks use. nix run .#e2e -- --test runs the e2e tests against clusters that are already up. Towards #473. Signed-off-by: Nic Cope --- CONTRIBUTING.md | 12 ++++---- e2e/README.md | 7 +++-- flake.nix | 1 + nix/apps.nix | 76 ++++++++++++++++++++++++++++++++++++++++++++++--- 4 files changed, 83 insertions(+), 13 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index da1c9d377..e87ac2b08 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -364,16 +364,16 @@ input is deterministic. Protobuf maps (`desired.resources`, (`conditions`, `results`, status arrays) must match the order the function emits. -`nix flake check` runs every function's tests. To run one function's while -you work on it: +`nix flake check` runs every function's tests, and so does `nix run .#test`, +outside the sandbox. Name a function to run only its tests, and pass pytest +arguments after it: ```bash -uv run --isolated --package compose-usages --group dev pytest functions/compose-usages/tests +nix run .#test -- compose-usages -k namespace ``` -`--isolated` gives each run its own environment, because every function names -its package `function`, so two can't share one. That's also why each function -runs in its own pytest session. +Each function runs in a pytest session of its own, because every function names +its package `function`. ### Running locally diff --git a/e2e/README.md b/e2e/README.md index 126290441..ae4c776bd 100644 --- a/e2e/README.md +++ b/e2e/README.md @@ -94,16 +94,17 @@ against. The control-plane cluster needs no DRA. ```bash nix run .#e2e # bring up both clusters + deploy the mock model nix run .#e2e -- --verify # same, then run the tests +nix run .#e2e -- --test # run the tests against clusters already up nix run .#e2e -- --clean # tear both clusters down ``` Arguments after `--verify` go to pytest, so `nix run .#e2e -- --verify -k usage` runs only the tests whose names match. Bring-up reuses clusters that are already up. To rerun the tests against an environment that's up, without bringing it up -again: +again, use `--test`: ```bash -uv run --isolated --package crossplane-models --group dev pytest e2e +nix run .#e2e -- --test -k usage ``` `crossplane project run` installs the config and applies the resources, then @@ -169,8 +170,8 @@ e2e/ conftest.py # fixtures: the clusters, curl pods, readiness test_serving.py # the tests kube.py, gateway.py # kubectl and curl helpers - wait.py # polling until a condition holds client.yaml # the curl pod the tests send requests from + wait.py # polling until a condition holds dra-example-driver.yaml # vendored fake DRA GPU driver (applied to workload) manifests/ # applied to the control plane after setup 00-namespaces.yaml diff --git a/flake.nix b/flake.nix index ffc34e264..8e502d8f1 100644 --- a/flake.nix +++ b/flake.nix @@ -183,6 +183,7 @@ }; stop = apps.stop { inherit crossplane; }; e2e = apps.e2e { inherit crossplane functionsPkg pythonSet; }; + test = apps.test { inherit pythonSet functionNames; }; stacks = apps.stacks { inherit (pkgs) aicr; }; } ); diff --git a/nix/apps.nix b/nix/apps.nix index f646fb00a..d2abbf65c 100644 --- a/nix/apps.nix +++ b/nix/apps.nix @@ -296,6 +296,68 @@ ); }; + # Run the composition functions' unit tests outside the sandbox, against the + # same virtualenvs nix flake check uses. With no function named it runs every + # function's tests, each in a pytest session of its own because every + # function's package is named `function`. Arguments after the function name + # go to pytest, e.g. nix run .#test -- compose-usages -k namespace. + test = + { + pythonSet, + functionNames, + }: + let + venvs = map (name: { + inherit name; + venv = pythonSet.mkVirtualEnv "${name}-test-env" { + ${name} = [ ]; + pytest = [ ]; + }; + }) functionNames; + cases = pkgs.lib.concatMapStrings (v: '' + ${v.name}) python=${v.venv}/bin/python ;; + '') venvs; + in + { + type = "app"; + meta.description = "Run the composition functions' unit tests"; + program = pkgs.lib.getExe ( + pkgs.writeShellApplication { + name = "modelplane-test"; + runtimeInputs = [ pkgs.coreutils ]; + inheritPath = false; + text = '' + run() { + local fn="$1" python + shift + case "$fn" in + ${cases} + *) + echo "no such function: $fn" >&2 + return 2 + ;; + esac + "$python" -m pytest "functions/$fn/tests" "$@" + } + + if [ $# -gt 0 ] && [[ "$1" != -* ]]; then + run "$@" + exit + fi + + failed=() + for fn in ${pkgs.lib.concatStringsSep " " functionNames}; do + run "$fn" "$@" || failed+=("$fn") + done + if [ ''${#failed[@]} -gt 0 ]; then + echo "failed: ''${failed[*]}" >&2 + exit 1 + fi + ''; + } + ); + }; + # Run the two-cluster local end-to-end test: a workload kind cluster # registered via source: Existing (serving stack + model) and a control-plane # cluster (crossplane + the Configuration). Two clusters because the @@ -304,8 +366,8 @@ # # With no argument it brings the environment up and applies the manifests, # --no-apply stops short of the manifests, --verify then runs the tests in - # e2e/ (passing any further arguments to pytest), and --clean tears it all - # down. This app materialises the Nix-built function images (as `run` does) + # e2e/ (passing any further arguments to pytest), --test runs them against an + # environment that's already up, and --clean tears it all down. This app materialises the Nix-built function images (as `run` does) # for crossplane project run to load. e2e = { @@ -353,9 +415,15 @@ exec python -m pytest e2e \ -o log_cli=true --log-cli-level=INFO "$@" ;; + --test) + shift + exec python -m pytest e2e \ + -o log_cli=true --log-cli-level=INFO "$@" + ;; *) - echo "usage: nix run .#e2e" \ - "[-- --no-apply | --verify [pytest args...] | --clean]" >&2 + echo "usage: nix run .#e2e -- [--no-apply |" \ + "--verify [pytest args...] | --test [pytest args...] |" \ + "--clean]" >&2 exit 2 ;; esac From cd56e3d7bfcfc68816fbb04e2c207f3ac4401359 Mon Sep 17 00:00:00 2001 From: Nic Cope Date: Wed, 30 Sep 2026 23:59:57 -0700 Subject: [PATCH 7/7] Read the e2e clusters through the Kubernetes API The e2e tests shelled out to kubectl for every read, wait and exec. Its errors reached the tests as stderr strings, a failed kubectl exec and a failed curl both showed up as an exit code, and built-in objects came back as untyped dicts. The official Kubernetes Python client returns typed objects and typed errors. Requests to the gateways still run curl in a pod, because only a pod can reach their published addresses from macOS. Towards #473. Signed-off-by: Nic Cope --- e2e/README.md | 7 +- e2e/client.yaml | 22 -- e2e/conftest.py | 135 +++++++++--- e2e/gateway.py | 21 +- e2e/kube.py | 170 +++++++-------- nix/apps.nix | 5 +- nix/checks.nix | 1 + pyproject.toml | 2 + uv.lock | 561 ++++++++++++++++++++++++++++++++++++++++++++++++ 9 files changed, 760 insertions(+), 164 deletions(-) delete mode 100644 e2e/client.yaml diff --git a/e2e/README.md b/e2e/README.md index ae4c776bd..ee3d76ec5 100644 --- a/e2e/README.md +++ b/e2e/README.md @@ -162,15 +162,16 @@ Everything the control plane needs is a declarative manifest; `environment.py` is only the irreducible cross-cluster setup (a second cluster, its MetalLB and DRA driver, and the cross-cluster kubeconfig). With `--verify`, pytest then runs `test_serving.py`. Its fixtures in `conftest.py` wait for the model to serve, -then start a curl pod on each cluster to send requests from. +then start a curl pod on each cluster to send requests from. The tests read the +clusters through the Kubernetes API, with the official Python client, while +bring-up drives the kind, crossplane, docker and kubectl CLIs. ``` e2e/ environment.py # two-cluster bring-up and teardown conftest.py # fixtures: the clusters, curl pods, readiness test_serving.py # the tests - kube.py, gateway.py # kubectl and curl helpers - client.yaml # the curl pod the tests send requests from + kube.py, gateway.py # Kubernetes API and curl helpers wait.py # polling until a condition holds dra-example-driver.yaml # vendored fake DRA GPU driver (applied to workload) manifests/ # applied to the control plane after setup diff --git a/e2e/client.yaml b/e2e/client.yaml deleted file mode 100644 index 9d596a22e..000000000 --- a/e2e/client.yaml +++ /dev/null @@ -1,22 +0,0 @@ -# A pod to send requests to the gateways from. Their addresses are on the kind -# Docker network, which a macOS host can't route to but a pod can. The tests -# apply this to both clusters, and exec curl in it. -apiVersion: v1 -kind: Namespace -metadata: - name: e2e ---- -apiVersion: v1 -kind: Pod -metadata: - name: curl - namespace: e2e -spec: - containers: - - name: curl - # Pinned by digest (a multi-arch manifest list) so a moving tag can't flake - # the tests. - image: curlimages/curl@sha256:7c12af72ceb38b7432ab85e1a265cff6ae58e06f95539d539b654f2cfa64bb13 - command: ["sleep", "infinity"] - # As PID 1, sleep ignores SIGTERM, so don't wait for it to exit. - terminationGracePeriodSeconds: 0 diff --git a/e2e/conftest.py b/e2e/conftest.py index 50bb1a46d..ba35eae04 100644 --- a/e2e/conftest.py +++ b/e2e/conftest.py @@ -18,17 +18,20 @@ brings them up first. """ -import pathlib from collections.abc import Iterator import pytest +from kubernetes import client +from kubernetes.client.rest import ApiException from models.ai.modelplane.inferencecluster import v1alpha1 as icv1alpha1 from models.ai.modelplane.inferencegateway import v1alpha1 as igv1alpha1 from models.ai.modelplane.modelservice import v1alpha1 as msv1alpha1 from e2e import environment, gateway, kube, wait -CLIENT = pathlib.Path(__file__).parent / "client.yaml" +# Where the curl pod the tests send requests from runs, on each cluster. +CLIENT_NAMESPACE = "e2e" +CLIENT_POD = "curl" @pytest.fixture(scope="session") @@ -46,26 +49,80 @@ def workload() -> kube.Cluster: @pytest.fixture(scope="session") def control_plane_client(control_plane: kube.Cluster) -> Iterator[gateway.Client]: """A pod on the control plane, which sends requests across clusters to the InferenceGateway.""" - yield from client(control_plane) + yield from curl_pod(control_plane) @pytest.fixture(scope="session") def workload_client(workload: kube.Cluster) -> Iterator[gateway.Client]: """A pod on the workload cluster, which can resolve the cluster gateway's Service name.""" - yield from client(workload) + yield from curl_pod(workload) -def client(cluster: kube.Cluster) -> Iterator[gateway.Client]: +def curl_pod(cluster: kube.Cluster) -> Iterator[gateway.Client]: """Start a curl pod on a cluster, and delete it afterwards.""" + # A run that was interrupted can leave the namespace behind. + delete_namespace(cluster) try: - cluster.apply(CLIENT) - cluster.wait_for_condition("pod", "curl", "e2e", "Ready", timeout=2 * 60) - yield gateway.Client(cluster, "e2e", "curl") + cluster.core.create_namespace( + client.V1Namespace(metadata=client.V1ObjectMeta(name=CLIENT_NAMESPACE)), + _request_timeout=kube.TIMEOUT_SECONDS, + ) + cluster.core.create_namespaced_pod( + CLIENT_NAMESPACE, + client.V1Pod( + metadata=client.V1ObjectMeta(name=CLIENT_POD), + spec=client.V1PodSpec( + containers=[ + client.V1Container( + name="curl", + # Pinned by digest (a multi-arch manifest list), so + # a moving tag can't flake the tests. + image="curlimages/curl@sha256:7c12af72ceb38b7432ab85e1a265cff6ae58e06f95539d539b654f2cfa64bb13", + command=["sleep", "infinity"], + ) + ], + # As PID 1, sleep ignores SIGTERM, so don't wait for it to + # exit. + termination_grace_period_seconds=0, + ), + ), + _request_timeout=kube.TIMEOUT_SECONDS, + ) + + def ready() -> None: + pod = cluster.core.read_namespaced_pod(CLIENT_POD, CLIENT_NAMESPACE, _request_timeout=kube.TIMEOUT_SECONDS) + conditions = pod.status.conditions or [] + assert any(c.type == "Ready" and c.status == "True" for c in conditions), f"pod is {pod.status.phase}" + + wait.until(ready, timeout=2 * 60, what=f"pod {CLIENT_NAMESPACE}/{CLIENT_POD} to be Ready", retry=kube.RETRY) + yield gateway.Client(cluster, CLIENT_NAMESPACE, CLIENT_POD) finally: - # Wait, so a run that follows doesn't create the pod in a namespace - # that's still terminating. - cluster.delete("namespace", "e2e", None) - cluster.wait_until_gone("namespace", "e2e", None, timeout=2 * 60) + delete_namespace(cluster) + + +def delete_namespace(cluster: kube.Cluster) -> None: + """Delete the curl pod's namespace if it exists, and wait for it to go. + + Waiting means a run that follows doesn't create the pod in a namespace + that's still terminating. + """ + + def gone() -> None: + try: + ns = cluster.core.read_namespace(CLIENT_NAMESPACE, _request_timeout=kube.TIMEOUT_SECONDS) + except ApiException as e: + if e.status == 404: + return + raise + msg = f"namespace {CLIENT_NAMESPACE} is {ns.status.phase}" + raise AssertionError(msg) + + try: + cluster.core.delete_namespace(CLIENT_NAMESPACE, _request_timeout=kube.TIMEOUT_SECONDS) + except ApiException as e: + if e.status != 404: + raise + wait.until(gone, timeout=2 * 60, what=f"namespace {CLIENT_NAMESPACE} to be deleted", retry=kube.RETRY) @pytest.fixture(scope="session") @@ -76,12 +133,22 @@ def routed(control_plane: kube.Cluster, workload: kube.Cluster) -> gateway.Servi so the serving stack and the model are still reconciling. On an environment that's already up, each wait returns at once. """ + # RoutingReady means the route is composed and applied on every gateway # serving the ModelService. Its status.model and the gateway's endpoints # both publish before that, without a replica, so waiting on either would # start sending requests while the engine is still rolling out. - obj = control_plane.wait_for_condition("modelservice", "mock", "ml-team", "RoutingReady", timeout=20 * 60) - ms = msv1alpha1.ModelService.model_validate(obj) + def routing_ready() -> msv1alpha1.ModelService: + obj = control_plane.modelplane("modelservices", "mock", "ml-team") + assert obj is not None, "ModelService ml-team/mock doesn't exist" + ms = msv1alpha1.ModelService.model_validate(obj) + conditions = (ms.status.conditions if ms.status else None) or [] + assert any(c.type == "RoutingReady" and c.status == "True" for c in conditions), ( + f"ModelService ml-team/mock isn't RoutingReady: {[(c.type, c.status, c.reason) for c in conditions]}" + ) + return ms + + ms = wait.until(routing_ready, timeout=20 * 60, what="ModelService ml-team/mock to route", retry=kube.RETRY) assert ms.status is not None assert ms.status.model is not None, "ModelService ml-team/mock is RoutingReady but publishes no model name" @@ -89,26 +156,32 @@ def routed(control_plane: kube.Cluster, workload: kube.Cluster) -> gateway.Servi # it, to stamp them with the hash of its sidecar's config, so a fresh # gateway is still replacing its pods when the route goes ready. Requests # can fail during that rollout, which starts only once AI Gateway has seen - # the route. So wait for the stamp, then for the rollout. - def proxies_stamped() -> None: - deployments = workload.list_objects("deployment", gateway.PROXY_NAMESPACE, gateway.PROXY_SELECTOR) - annotations = [d["spec"]["template"]["metadata"].get("annotations", {}) for d in deployments] - assert annotations, "the InferenceGateway has no proxy Deployment" - assert all("aigateway.envoyproxy.io/extproc-config-hash" in a for a in annotations), ( - "AI Gateway hasn't stamped the InferenceGateway's proxy pods" - ) + # the route. So wait for the stamp, then for the rollout to finish. + def proxies_rolled_out() -> None: + deployments = workload.apps.list_namespaced_deployment( + gateway.PROXY_NAMESPACE, label_selector=gateway.PROXY_SELECTOR, _request_timeout=kube.TIMEOUT_SECONDS + ).items + assert deployments, "the InferenceGateway has no proxy Deployment" + for d in deployments: + annotations = d.spec.template.metadata.annotations or {} + assert "aigateway.envoyproxy.io/extproc-config-hash" in annotations, ( + f"AI Gateway hasn't stamped Deployment {d.metadata.name}'s pods" + ) + # The same test as kubectl rollout status. + assert (d.status.observed_generation or 0) >= d.metadata.generation, ( + f"Deployment {d.metadata.name}'s controller hasn't seen its latest spec" + ) + want = d.spec.replicas + assert (d.status.updated_replicas or 0) == want, f"Deployment {d.metadata.name} is still updating pods" + assert (d.status.replicas or 0) == want, f"Deployment {d.metadata.name} still has old pods" + assert (d.status.available_replicas or 0) == want, f"Deployment {d.metadata.name} has unavailable pods" wait.until( - proxies_stamped, timeout=2 * 60, what="AI Gateway to stamp the InferenceGateway's proxy pods", retry=kube.RETRY + proxies_rolled_out, timeout=5 * 60, what="the InferenceGateway's proxy pods to roll out", retry=kube.RETRY ) - workload.kubectl( - "rollout", "status", "deployment", f"--namespace={gateway.PROXY_NAMESPACE}", - f"--selector={gateway.PROXY_SELECTOR}", "--timeout=5m", - timeout=6 * 60, - ) # fmt: skip def endpoints_published() -> igv1alpha1.Endpoints: - obj = control_plane.get("inferencegateway", "local", None) + obj = control_plane.modelplane("inferencegateways", "local", None) assert obj is not None, "InferenceGateway local doesn't exist" ig = igv1alpha1.InferenceGateway.model_validate(obj) assert ig.status is not None @@ -130,7 +203,7 @@ def serving(routed: gateway.Serving, control_plane_client: gateway.Client) -> ga The gateway can publish its endpoints a moment before the route serves, and a slow CI runner widens that gap. Tests that expect a refusal use routed instead, so a gateway that refuses everyone still fails only the tests that - expect it to serve. + need it to serve. """ def serves() -> None: @@ -154,7 +227,7 @@ def cluster_gateway(control_plane: kube.Cluster) -> str: """ def published() -> str: - obj = control_plane.get("inferencecluster", "local", None) + obj = control_plane.modelplane("inferenceclusters", "local", None) assert obj is not None, "InferenceCluster local doesn't exist" ic = icv1alpha1.InferenceCluster.model_validate(obj) assert ic.status is not None diff --git a/e2e/gateway.py b/e2e/gateway.py index f14efb1cf..c3c04400f 100644 --- a/e2e/gateway.py +++ b/e2e/gateway.py @@ -47,7 +47,9 @@ class Serving: class Response: """What came back from a request.""" + # The HTTP status, or 0 if no HTTP response came back. status: int + # The body, or why no HTTP response came back. body: str def json(self) -> Any: # noqa: ANN401 - a JSON body can decode to any type. @@ -70,17 +72,18 @@ def request(self, url: str, headers: dict[str, str], body: object | None = None) curl += ["--header", f"{name}: {value}"] if body is not None: curl += ["--header", "content-type: application/json", "--data", json.dumps(body)] - # curl exits non-zero only when no HTTP response came back, which fails - # the request rather than answering it. - out = self.cluster.kubectl("exec", f"--namespace={self.namespace}", self.pod, "--", *curl) + result = self.cluster.exec(self.pod, self.namespace, curl) + # curl exits non-zero only when no HTTP response came back. + if result.code != 0: + return Response(status=0, body=result.stderr.strip()) # --write-out puts the status on a line of its own, after the body. - text, _, status = out.rpartition("\n") + text, _, status = result.stdout.rpartition("\n") return Response(status=int(status), body=text) def connect(self, url: str) -> int: """GET a URL without verifying the server's certificate, and return curl's exit code.""" curl = ["curl", "--silent", "--show-error", "--insecure", "--max-time", "15", "--output", "/dev/null", url] - return self.cluster.run("exec", f"--namespace={self.namespace}", self.pod, "--", *curl).returncode + return self.cluster.exec(self.pod, self.namespace, curl).code def usage_records(cluster: kube.Cluster) -> list[dict[str, Any]]: @@ -88,10 +91,12 @@ def usage_records(cluster: kube.Cluster) -> list[dict[str, Any]]: A request lands on any one of the proxy pods, so this reads them all. """ + pods = cluster.core.list_namespaced_pod( + PROXY_NAMESPACE, label_selector=PROXY_SELECTOR, _request_timeout=kube.TIMEOUT_SECONDS + ) records = [] - for pod in cluster.list_objects("pod", PROXY_NAMESPACE, PROXY_SELECTOR): - logs = cluster.kubectl("logs", f"--namespace={PROXY_NAMESPACE}", pod["metadata"]["name"], "--container=envoy") - for line in logs.splitlines(): + for pod in pods.items: + for line in cluster.logs(pod.metadata.name, PROXY_NAMESPACE, "envoy").splitlines(): # Envoy logs other things too. The access log is the JSON objects. try: record = json.loads(line) diff --git a/e2e/kube.py b/e2e/kube.py index 24a8ffb0b..546a618c6 100644 --- a/e2e/kube.py +++ b/e2e/kube.py @@ -12,120 +12,94 @@ # See the License for the specific language governing permissions and # limitations under the License. -"""Read and change a cluster's resources with kubectl. +"""Read a cluster's resources through the Kubernetes API. -Objects come back as the dicts kubectl's JSON output decodes to. Tests that -read Modelplane's own fields validate them into the generated models. +The official Kubernetes client returns typed objects and typed errors for the +built-in kinds. Cluster wraps the parts of it that need care: Modelplane's own +resources, which come back as dicts for the tests to validate into the +generated models, a command's exit code, and container logs. """ import dataclasses -import json -import pathlib import shlex -import subprocess from typing import Any -from e2e import wait +import urllib3 +from kubernetes import client, config +from kubernetes.client.rest import ApiException +from kubernetes.stream import stream -# A bound on each kubectl call, so a hung API server fails the test that hit it -# rather than the whole run. -KUBECTL_TIMEOUT_SECONDS = 60 +# A bound on each API call, so a hung API server fails the test that hit it +# rather than the whole run. Pass it as _request_timeout: the client has no +# default. +TIMEOUT_SECONDS = 60 -type Object = dict[str, Any] +# What a wait retries: the condition not holding yet, or the API server +# failing a request while the cluster converges. +RETRY = (AssertionError, ApiException, urllib3.exceptions.HTTPError) -class KubectlError(Exception): - """kubectl exited non-zero, or timed out.""" - +@dataclasses.dataclass(frozen=True) +class Exec: + """How a command run in a container exited, and what it printed.""" -# What a wait retries: the condition not holding yet, or a kubectl call failing -# while the cluster converges. -RETRY = (AssertionError, KubectlError) + code: int + stdout: str + stderr: str -@dataclasses.dataclass(frozen=True) class Cluster: """A cluster, addressed by its kubeconfig context.""" - context: str + def __init__(self, context: str) -> None: + """Connect to the cluster a kubeconfig context names.""" + api = config.new_client_from_config(context=context) + self.core = client.CoreV1Api(api) + self.apps = client.AppsV1Api(api) + self.custom = client.CustomObjectsApi(api) - def run(self, *args: str, timeout: float = KUBECTL_TIMEOUT_SECONDS) -> subprocess.CompletedProcess[str]: - """Run kubectl against this cluster, and return how it exited.""" - cmd = ["kubectl", f"--context={self.context}", *args] + def modelplane(self, plural: str, name: str, namespace: str | None) -> dict[str, Any] | None: + """Return a Modelplane resource, or None if it doesn't exist.""" try: - return subprocess.run(cmd, capture_output=True, text=True, timeout=timeout, check=False) - except subprocess.TimeoutExpired as e: - msg = f"{shlex.join(cmd)}: timed out after {timeout:.0f}s" - raise KubectlError(msg) from e - - def kubectl(self, *args: str, timeout: float = KUBECTL_TIMEOUT_SECONDS) -> str: - """Run kubectl against this cluster, and return what it wrote to stdout.""" - result = self.run(*args, timeout=timeout) - if result.returncode != 0: - msg = f"{shlex.join(result.args)}: {result.stderr.strip()}" - raise KubectlError(msg) - return result.stdout - - def get(self, kind: str, name: str, namespace: str | None) -> Object | None: - """Return the named object, or None if it doesn't exist.""" - out = self.kubectl("get", kind, name, *namespaced(namespace), "--ignore-not-found", "--output=json") - return json.loads(out) if out else None - - def list_objects(self, kind: str, namespace: str, selector: str) -> list[Object]: - """Return the objects of a kind in a namespace that match a label selector.""" - out = self.kubectl("get", kind, f"--namespace={namespace}", f"--selector={selector}", "--output=json") - return json.loads(out)["items"] - - def apply(self, path: pathlib.Path) -> None: - """Apply a manifest file.""" - self.kubectl("apply", f"--filename={path}") - - def delete(self, kind: str, name: str, namespace: str | None) -> None: - """Delete an object without waiting for it to go. Deleting one that doesn't exist is a no-op.""" - self.kubectl("delete", kind, name, *namespaced(namespace), "--ignore-not-found", "--wait=false") - - def wait_for_condition(self, kind: str, name: str, namespace: str | None, condition: str, timeout: float) -> Object: - """Wait for an object's condition to be True, and return the object.""" - - def condition_is_true() -> Object: - obj = self.get(kind, name, namespace) - assert obj is not None, f"{kind} {qualified(namespace, name)} doesn't exist" - assert is_true(obj, condition), f"{describe(obj)} isn't {condition}" - return obj - - what = f"{kind} {qualified(namespace, name)} {condition}" - return wait.until(condition_is_true, timeout=timeout, what=what, retry=RETRY) - - def wait_until_gone(self, kind: str, name: str, namespace: str | None, timeout: float) -> None: - """Wait for an object to stop existing.""" - - def is_gone() -> None: - obj = self.get(kind, name, namespace) - assert obj is None, f"{describe(obj)} still exists" - - wait.until(is_gone, timeout=timeout, what=f"{kind} {qualified(namespace, name)} to be deleted", retry=RETRY) - - -def namespaced(namespace: str | None) -> list[str]: - """Return the kubectl flag that scopes a command to a namespace, if there is one.""" - return [f"--namespace={namespace}"] if namespace else [] - - -def qualified(namespace: str | None, name: str) -> str: - """Return namespace/name, or just name for a cluster-scoped object.""" - return f"{namespace}/{name}" if namespace else name - - -def is_true(obj: Object, condition: str) -> bool: - """Report whether an object's status condition is True.""" - return any(c["type"] == condition and c["status"] == "True" for c in obj.get("status", {}).get("conditions", [])) - - -def describe(obj: Object) -> str: - """Summarise an object and its status conditions on one line, for failure messages.""" - meta = obj["metadata"] - conditions = [] - for c in obj.get("status", {}).get("conditions", []): - why = ": ".join(v for v in (c.get("reason"), c.get("message")) if v) - conditions.append(f"{c['type']}={c['status']} ({why})" if why else f"{c['type']}={c['status']}") - return f"{obj['kind']} {qualified(meta.get('namespace'), meta['name'])}: {', '.join(conditions) or 'no conditions'}" + if namespace is None: + return self.custom.get_cluster_custom_object( + "modelplane.ai", "v1alpha1", plural, name, _request_timeout=TIMEOUT_SECONDS + ) + return self.custom.get_namespaced_custom_object( + "modelplane.ai", "v1alpha1", namespace, plural, name, _request_timeout=TIMEOUT_SECONDS + ) + except ApiException as e: + if e.status == 404: + return None + raise + + def exec(self, pod: str, namespace: str, command: list[str]) -> Exec: + """Run a command in a pod's only container, and return how it exited.""" + # The exec API streams over a websocket. Without _preload_content the + # stream stays open until the command exits, which is what yields its + # exit code. + resp = stream( + self.core.connect_get_namespaced_pod_exec, + pod, + namespace, + command=command, + stdout=True, + stderr=True, + stdin=False, + tty=False, + _preload_content=False, + ) + resp.run_forever(timeout=TIMEOUT_SECONDS) + if resp.returncode is None: + msg = f"{shlex.join(command)} in {namespace}/{pod} didn't exit within {TIMEOUT_SECONDS}s" + raise TimeoutError(msg) + return Exec(code=resp.returncode, stdout=resp.read_stdout(), stderr=resp.read_stderr()) + + def logs(self, pod: str, namespace: str, container: str) -> str: + """Return a container's logs.""" + # With its content preloaded, the client tries to deserialize the logs, + # and returns JSON log lines as the repr of a bytes object. + resp = self.core.read_namespaced_pod_log( + pod, namespace, container=container, _preload_content=False, _request_timeout=TIMEOUT_SECONDS + ) + return resp.data.decode() diff --git a/nix/apps.nix b/nix/apps.nix index d2abbf65c..a242f8deb 100644 --- a/nix/apps.nix +++ b/nix/apps.nix @@ -376,11 +376,12 @@ pythonSet, }: let - # What e2e/ imports: pytest, and the generated models it reads - # Modelplane's status with. pydantic is declared here rather than on + # What e2e/ imports: pytest, the Kubernetes client, and the generated + # models it reads Modelplane's status with. pydantic is declared here rather than on # crossplane-models, whose pyproject.toml the Crossplane CLI generates. venv = pythonSet.mkVirtualEnv "modelplane-e2e-env" { pytest = [ ]; + kubernetes = [ ]; crossplane-models = [ ]; pydantic = [ ]; }; diff --git a/nix/checks.nix b/nix/checks.nix index 01c1c991a..e24fe6242 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -99,6 +99,7 @@ in let venv = pythonSet.mkVirtualEnv "e2e-ty-env" { pytest = [ ]; + kubernetes = [ ]; crossplane-models = [ ]; pydantic = [ ]; }; diff --git a/pyproject.toml b/pyproject.toml index 347268664..c1f07c182 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -28,6 +28,8 @@ dev = [ "pydantic>=2.0", # 9.0 reports unittest subtests natively. "pytest>=9.0", + # The end-to-end tests read the clusters through the Kubernetes API. + "kubernetes>=36.0", ] [tool.ruff] diff --git a/uv.lock b/uv.lock index ddc2973d6..0980cf8bd 100644 --- a/uv.lock +++ b/uv.lock @@ -28,6 +28,105 @@ members = [ "modelplane", ] +[[package]] +name = "aiohappyeyeballs" +version = "2.7.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/ce/f4/eec0465c2f67b2664688d0240b3212d5196fd89e741df67ddb81f8d35658/aiohappyeyeballs-2.7.1.tar.gz", hash = "sha256:065665c041c42a5938ed220bdcd7230f22527fbec085e1853d2402c8a3615d9d", size = 24757, upload-time = "2026-07-01T17:11:55.501Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/71/43/1947f06babed6b3f1d7f38b0c767f52df66bfb2bc10b468c4a7de9eceff2/aiohappyeyeballs-2.7.1-py3-none-any.whl", hash = "sha256:9243213661e29250eb41368e5daa826fc017156c3b8a11440826b2e3ed376472", size = 15038, upload-time = "2026-07-01T17:11:54.055Z" }, +] + +[[package]] +name = "aiohttp" +version = "3.14.3" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "aiohappyeyeballs" }, + { name = "aiosignal" }, + { name = "attrs" }, + { name = "frozenlist" }, + { name = "multidict" }, + { name = "propcache" }, + { name = "typing-extensions", marker = "python_full_version < '3.13'" }, + { name = "yarl" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/58/d9/22ce5786ac0c1653ae8b6c23bded02c1686d11f0dbb45b31ce128e0df985/aiohttp-3.14.3.tar.gz", hash = "sha256:9491196535a88924a60afd5b5f434b5b203b6cc616250878dbdb223a8f7844bc", size = 7971213, upload-time = "2026-07-23T01:57:27.037Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/f8/5c/b3e4ff8ad43a8afef9602c5e90285936da1beaea8b029016b793891f03c3/aiohttp-3.14.3-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:e568e14940c09955aa51f4e645b6daa18a581c5dcfcd73744dcc86a856e3ced3", size = 764250, upload-time = "2026-07-23T01:52:48.525Z" }, + { url = "https://files.pythonhosted.org/packages/0e/da/f1b384465e51449d844056b75070461da03a9a23e6c1747003695bf4172a/aiohttp-3.14.3-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:54cfcdee2770dac994417cbb0ee1f3eb0e7cb6b30c79bf44f2c02ff79ec5124a", size = 516281, upload-time = "2026-07-23T01:52:51.047Z" }, + { url = "https://files.pythonhosted.org/packages/b9/3f/01264f820ee2e3712a827892b1cd6ff80f3300c1fcbffbb45714a915d47a/aiohttp-3.14.3-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:21c016079415ed3fd676963e9793700a566d85dbbd6bfc564b9b2d209147dcc8", size = 514742, upload-time = "2026-07-23T01:52:53.779Z" }, + { url = "https://files.pythonhosted.org/packages/9e/8d/a71c6f2db52ac1ed142b133f7feddaa6b70539c3f4de24d7e226c95b794c/aiohttp-3.14.3-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d6088ec9894113802bddb3c09e974929aed2c7b3a8c456219b8aab4481f1a239", size = 1780613, upload-time = "2026-07-23T01:52:56.948Z" }, + { url = "https://files.pythonhosted.org/packages/a5/11/3dd9b3fb3a170f6ec9011b5291d876a6fab4086714c9e158600edf01b4fd/aiohttp-3.14.3-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:16ea7e24c309fb7c0bbd505d149abe4fe4dccfb8db911db7dbec0921bc889a6f", size = 1737688, upload-time = "2026-07-23T01:52:59.294Z" }, + { url = "https://files.pythonhosted.org/packages/6d/3e/834c26918be7d88068822b40e0db30fca50b5f4fe79104aa16a93f1d74e6/aiohttp-3.14.3-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:56f355e79f71aef2a85c80305cc915f894b170dba76de5fe84f6351939b83c06", size = 1845742, upload-time = "2026-07-23T01:53:01.641Z" }, + { url = "https://files.pythonhosted.org/packages/cc/c9/49ab8572df7d66bc13d11e31f781292badb04180dd87ba98733066c6aed7/aiohttp-3.14.3-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:18c441d0a8fca6de8d1f546849b9f0ab20d435993e2c5b59562b2fae6be2f929", size = 1928412, upload-time = "2026-07-23T01:53:04.018Z" }, + { url = "https://files.pythonhosted.org/packages/a5/b9/2b8f0c0ce09c87a1daf80fd483431b56b1435d3f62789bc86f572e1245de/aiohttp-3.14.3-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:53e7b4ce82b54a8bcc71b3b67a5cbd177ca1d7f592cbc92cd38b7349f73482db", size = 1786220, upload-time = "2026-07-23T01:53:06.481Z" }, + { url = "https://files.pythonhosted.org/packages/85/00/9c45f81de11710460edfa1dc81317b6e882703b160926c879a9d20da9fcc/aiohttp-3.14.3-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:f55119f7bf25f49ed210f6096090715da24f2943c62102448915fde3c62877ce", size = 1637231, upload-time = "2026-07-23T01:53:10.258Z" }, + { url = "https://files.pythonhosted.org/packages/19/ce/967d628e910756f3539c6107cb7844a1b69440dcb3029a5ee7871b09ab63/aiohttp-3.14.3-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:9aa6e61fdf20105c4144e755bd586008ff450791d67b1c8146fdc15959c4d51c", size = 1753161, upload-time = "2026-07-23T01:53:13.817Z" }, + { url = "https://files.pythonhosted.org/packages/11/b2/0c3d4114f0aee4f580f5b3b4eb71b24d7a23b834ea506a4dfebe76513f35/aiohttp-3.14.3-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:ccd4893707b3e2a13e39c90d43cf80edf2e4d0457935bcc103bf2346214c3f15", size = 1756356, upload-time = "2026-07-23T01:53:16.211Z" }, + { url = "https://files.pythonhosted.org/packages/63/5d/99e7d91c82f1399d1ae2a854e080bd1493fbc31e5e959dbc4ec33dac3bec/aiohttp-3.14.3-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:b2466434105a4e03113c36ec775cc2ebe6676b62eae326fa670bb607ef788c1c", size = 1819846, upload-time = "2026-07-23T01:53:18.289Z" }, + { url = "https://files.pythonhosted.org/packages/ad/05/d5e1cb6480eeffd3f901d40a2c5e2d1e7effdc797837da3b490272699f13/aiohttp-3.14.3-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:ba59d59aba08ac02fc03b0c8983ccd5ee39a199d0552ce9e6d2b4845b34d59ae", size = 1628531, upload-time = "2026-07-23T01:53:23.86Z" }, + { url = "https://files.pythonhosted.org/packages/c9/90/b934682bcaefae18a9e04f3dff5b68522ba810906358ae5029b68110ea3b/aiohttp-3.14.3-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:ed099d105449c4f9e84f24af203cd131349d4761d8813fa7e02c32e7128cd910", size = 1832712, upload-time = "2026-07-23T01:53:27.551Z" }, + { url = "https://files.pythonhosted.org/packages/21/df/6061679faaf81fac746e7307c7adb71e858071a5d34c27583afefc64f543/aiohttp-3.14.3-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:152516815ef926786a0b6ae2b8f1fd2e0c71582dee0b435636865316fd4891b7", size = 1775014, upload-time = "2026-07-23T01:53:30.223Z" }, + { url = "https://files.pythonhosted.org/packages/8a/1d/f854878bbc69b88faefe924b619a34a6f59ec05fd387c77690667eaa75eb/aiohttp-3.14.3-cp311-cp311-win32.whl", hash = "sha256:a4af35c443e0b1a1bd6a8af3f3485d7fda15c142751a00f3ff8090f0b93346fa", size = 456006, upload-time = "2026-07-23T01:53:34.97Z" }, + { url = "https://files.pythonhosted.org/packages/73/0c/2af9d1674baccd1dbd47282a93d660a22e57ef6167c856deb24b4214fbab/aiohttp-3.14.3-cp311-cp311-win_amd64.whl", hash = "sha256:e1e74298bab6ee0d6e749ed4fd1901c7e604bdda32c03d787a2cc71c46d0433d", size = 481069, upload-time = "2026-07-23T01:53:39.673Z" }, + { url = "https://files.pythonhosted.org/packages/8e/76/88401ff3fc95e85c5fc38d588f36f55e61ecb64343b2bc8d69326f453cc0/aiohttp-3.14.3-cp311-cp311-win_arm64.whl", hash = "sha256:03cd2bde3d7f085b64e549c985f4bb928cad7e8ecf5323bfca320db548d81b39", size = 453021, upload-time = "2026-07-23T01:53:43.749Z" }, + { url = "https://files.pythonhosted.org/packages/18/d4/eb96299230e20acf2efae207cb8d69051f1f68e357e5ea5e479bf6fb097a/aiohttp-3.14.3-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:39aded8c7f3b935b54aab1d8d73c70ec0ee2d3ec3b943e0e86611bc150ba47f5", size = 754690, upload-time = "2026-07-23T01:53:47.332Z" }, + { url = "https://files.pythonhosted.org/packages/88/11/e7a70a209eb9a067c0d3212b518a0134e3484f5178c7533878b6b514d469/aiohttp-3.14.3-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:5bcb6ff3fdab1258a192679ff1a05d44f59626430aa05cd1a9d2447423599228", size = 509484, upload-time = "2026-07-23T01:53:51.159Z" }, + { url = "https://files.pythonhosted.org/packages/30/07/4bbc222cc8dbe31d4c3e8a5baad2286e4d42026ac0c570027b89afce6344/aiohttp-3.14.3-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:617105e2c3018ee38d0c8ce5ee3c84f621a6d8b9f723202aacaff28449ca91ee", size = 511949, upload-time = "2026-07-23T01:53:55.083Z" }, + { url = "https://files.pythonhosted.org/packages/54/b9/42e74c46b7b7c794b995bbc1f573fb48950c38b19d8600c62a6804ee2d67/aiohttp-3.14.3-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f631fe87a6f30df5fbe6d79640b25e4cffb38c31c7fb6f10871517b84b0f8c1a", size = 1765282, upload-time = "2026-07-23T01:53:59.662Z" }, + { url = "https://files.pythonhosted.org/packages/6b/ed/62bc4d74363ad346d518e0720363a949f63e2e23439a79eb5813d4d29bb3/aiohttp-3.14.3-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:a94dbaae5ae27bd849c93570669bff91e0510f33a80805738e3de72a7be0447b", size = 1741511, upload-time = "2026-07-23T01:54:04.063Z" }, + { url = "https://files.pythonhosted.org/packages/d0/9f/181e8a8bc79e47d13c7fc4540bd7a3b729d9505609c61f392a8dd2fbfe55/aiohttp-3.14.3-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:8f2f1c4c032c7cedd7d8da6f54c97b70266c6570c3108d3fdffee7188bb70529", size = 1810680, upload-time = "2026-07-23T01:54:09.882Z" }, + { url = "https://files.pythonhosted.org/packages/5c/9a/dec94d6ad694552fe3424e3f1928d7a606a5d9d9433a04e7ecdd9d38ae7f/aiohttp-3.14.3-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:ea05e1f97ceea523942d9b2a7d7c0359d781d683d6b043f5943a602b14da4787", size = 1905646, upload-time = "2026-07-23T01:54:13.475Z" }, + { url = "https://files.pythonhosted.org/packages/52/b7/7cd31f29d6055bd711ae6e669367fba6f5ae9de463910a793e30556a8db7/aiohttp-3.14.3-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:543906c127fb1d929b95076db19b83fa2d46751006ff1e23b093aa5ac4d8db42", size = 1792122, upload-time = "2026-07-23T01:54:15.752Z" }, + { url = "https://files.pythonhosted.org/packages/66/73/10b1ef93afa61f4963c746257b70ced619cf31a4798671de5fdb2608501d/aiohttp-3.14.3-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:0a5ff2dfbb9ce645fa5b8ef3e02c6c0b9cc3f6030ff863d0c51fffc50cb5541b", size = 1591127, upload-time = "2026-07-23T01:54:19.489Z" }, + { url = "https://files.pythonhosted.org/packages/49/ed/3b203fa6de1b338c14acdc06bf6ca9b043b7944f005966958c2ced932cde/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:041badb8f84396357c4d3ad26de6afd7a32b112f43d3c63045c0c8278cfd2043", size = 1725210, upload-time = "2026-07-23T01:54:24.129Z" }, + { url = "https://files.pythonhosted.org/packages/28/b7/1c2aab8c706436dcc28598452488ac9cd7c409da815237c28c27d58993e6/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:530125ee1163c4219af35dc3aa1206e541e7b31b6efc1a3f93b70a136f65d427", size = 1764848, upload-time = "2026-07-23T01:54:27.973Z" }, + { url = "https://files.pythonhosted.org/packages/54/50/94c28f08b131c4bf10984ea2c7a536c9920608bb2d6e7f95642c30cc87b7/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:c8653fd547c93a61aadc612007790f5555cdd18946fa48cf45e26d8ea4ea473d", size = 1777102, upload-time = "2026-07-23T01:54:31.775Z" }, + { url = "https://files.pythonhosted.org/packages/13/d4/e7d09ba7d345fb2d74440fd2fa033c5e079fac05552927705986f41a364f/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:89176250f686cb9853c0fb7ead90e639e915b84a6f43eedc2a4e7ec21f1037f0", size = 1580205, upload-time = "2026-07-23T01:54:34.518Z" }, + { url = "https://files.pythonhosted.org/packages/a3/84/072a91d68e1e1eb587985b54baab94221277f877e8ef274fc213a0ceae28/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:3a26434dafe408229ff3403458ca58de24fb51936504decac49ce6755f77e59d", size = 1797219, upload-time = "2026-07-23T01:54:36.995Z" }, + { url = "https://files.pythonhosted.org/packages/e0/eb/aad34e897e668424d6e995da5dff8a4a09af93363d3392488772957a63aa/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:d1558173930a5a8d3069cee5c92fc91c87c4dbcb099debbb3622053717145a19", size = 1768629, upload-time = "2026-07-23T01:54:40.103Z" }, + { url = "https://files.pythonhosted.org/packages/b6/2b/6bb88ddba0fecd9122aa3ebcad25996cf6c083a4a7040dbb3a4f97972af6/aiohttp-3.14.3-cp312-cp312-win32.whl", hash = "sha256:16100ad3ab8d649fdfbee87602d9d2dcdca9df0b9eda8a1b5fdc0d41f96da559", size = 451481, upload-time = "2026-07-23T01:54:42.547Z" }, + { url = "https://files.pythonhosted.org/packages/76/9b/f2f8f108da17ecef2cc3efc424e8b7ad3782b1a8360f7b8eae8ced84f6ea/aiohttp-3.14.3-cp312-cp312-win_amd64.whl", hash = "sha256:33a2d7c28d33797a2e99923dffa63f83d908a19b6bf26cfe80fa790aa5e1a75a", size = 476845, upload-time = "2026-07-23T01:54:44.853Z" }, + { url = "https://files.pythonhosted.org/packages/3e/44/28dac80a8941b604f4da10ce21097614ca1bf905ce93dca28d8d7de9c1e7/aiohttp-3.14.3-cp312-cp312-win_arm64.whl", hash = "sha256:362a3fd481769cac1a824514bcd86fda51c65e8fe6e051099e008fddde6db17c", size = 448050, upload-time = "2026-07-23T01:54:47.087Z" }, + { url = "https://files.pythonhosted.org/packages/57/be/5afd201cc0ab139029aadb75392efe85a293403d9dd3a3226161c21ce00c/aiohttp-3.14.3-cp313-cp313-android_21_arm64_v8a.whl", hash = "sha256:2e9878ae68e4a5f1c0abe4dd497dbc3d51946f5837b56759e2a02e78fa90ef86", size = 506269, upload-time = "2026-07-23T01:54:49.075Z" }, + { url = "https://files.pythonhosted.org/packages/22/09/dec8189d62b45ade009f6792a2264b942a90cb88aeaf181239933cd72c3c/aiohttp-3.14.3-cp313-cp313-android_21_x86_64.whl", hash = "sha256:f3d2669fe7dec7fc359ecdb5984b29b50d85d5d00f8c1cb61de4f4a24ee42627", size = 515166, upload-time = "2026-07-23T01:54:51.894Z" }, + { url = "https://files.pythonhosted.org/packages/28/24/2854869d29ed8a8b19d74f9ec6629515f7e04d02dd329d9d179201e58e47/aiohttp-3.14.3-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:cc7cb243a68167172f48c1fd43cee91ec4b1d40cefd190edd43369d1a6bc9c82", size = 486263, upload-time = "2026-07-23T01:54:54.223Z" }, + { url = "https://files.pythonhosted.org/packages/d4/dd/57187c8be2a35aea65eaee3bd2c3dcbbcf0204f5106c89637e3610380cd1/aiohttp-3.14.3-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:78253b573e6ffab5028924fc98bc281aae05445969982a10864bc360dea2016c", size = 492299, upload-time = "2026-07-23T01:54:56.236Z" }, + { url = "https://files.pythonhosted.org/packages/b9/11/06ae6ed8f0d414edf4068861e233d8fe23ee699bfd4b3ceb8663db948a62/aiohttp-3.14.3-cp313-cp313-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:7041d52c3a7fa20c9e8c182b534704abb19502c8bdcbde7ab23bfda6f642394f", size = 502235, upload-time = "2026-07-23T01:54:58.377Z" }, + { url = "https://files.pythonhosted.org/packages/7e/a3/559639c34a345d2cf7c52dff6838119f2eaf29eb508227b5b83f573af813/aiohttp-3.14.3-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:ac74facc01463f138b0da5580329cfcc82818dea5656e83ddcd11268fc12ff80", size = 750883, upload-time = "2026-07-23T01:55:00.65Z" }, + { url = "https://files.pythonhosted.org/packages/91/cd/41e131f13afd1e7b0172a9d9eda085ef90eb8439f41f0d279db81ed3ae60/aiohttp-3.14.3-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:d6218d92e450824e9b4881f44e8c09f1853b490f9a64130801024a4793b1b3b0", size = 508473, upload-time = "2026-07-23T01:55:02.945Z" }, + { url = "https://files.pythonhosted.org/packages/bc/6b/e7f13410d391c6e55b4c007a8de024355389d7d459e3d64c42b2d33617e5/aiohttp-3.14.3-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:11fb37ef075669eee52ab1928fbf6e1741fada40409fa309ebde9607a962aebf", size = 509190, upload-time = "2026-07-23T01:55:05.173Z" }, + { url = "https://files.pythonhosted.org/packages/97/21/6464573e53d69672cc1eada3e5c5cb2d2efa82701e8305a0f2047a576967/aiohttp-3.14.3-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:55bdcc472aafe2de4a253045cc128007a64f1e0264fb675791e132ea5edaa3bd", size = 1761478, upload-time = "2026-07-23T01:55:07.383Z" }, + { url = "https://files.pythonhosted.org/packages/1a/81/d217043a4c17fbce360905e3b2bdd20139ebc9a2de836d035d179c4da006/aiohttp-3.14.3-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:c39846c3aad97a8530c89d7a3869a8f8e9e3762c6ac0504481e5c80948f7e807", size = 1735092, upload-time = "2026-07-23T01:55:09.803Z" }, + { url = "https://files.pythonhosted.org/packages/a1/66/e13a02d0eeb1a9a502402a977abb4e4abff9fe4051c26f80558c57a7c975/aiohttp-3.14.3-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:5895ef58c4620afe02fa16044f023dc4dafec08158f9d08874a46a7dbc0341b8", size = 1800546, upload-time = "2026-07-23T01:55:12.012Z" }, + { url = "https://files.pythonhosted.org/packages/26/5e/57d42fca1d18cb5acc1cad945d017fabc5d6ae71d8a08ad66be8dc3ee544/aiohttp-3.14.3-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:fa9467a8113aa69d3d7c55a70ef0b7c636010a40993f3df9d9d0d73b3eb7ef24", size = 1895250, upload-time = "2026-07-23T01:55:14.357Z" }, + { url = "https://files.pythonhosted.org/packages/ca/1c/7da8d08e74d56f00070822f9638ff3f1c563f8ad87d1efa996c87bfc8644/aiohttp-3.14.3-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d7d2deec16eeedf55f2c7cf75b521ea3856a5177e123844f8fd0f114ce252cb5", size = 1789289, upload-time = "2026-07-23T01:55:16.668Z" }, + { url = "https://files.pythonhosted.org/packages/cd/0f/cf16bcf56896981c1a0319f5d5db9337994b5165730c48a8fa07e9b34be6/aiohttp-3.14.3-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:dd54d0e8717de95939766febac482ac0474d8ac3b048115f9f2b1d23a16e7db4", size = 1586706, upload-time = "2026-07-23T01:55:18.913Z" }, + { url = "https://files.pythonhosted.org/packages/fe/6f/76eac12a7f2480e1e304f842efdb07db33256b0d9165b866b6ef0806c202/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:df82f3787c940c94986b34222d59c9e38843fba85139f36e85255a82ad5355a9", size = 1724652, upload-time = "2026-07-23T01:55:21.296Z" }, + { url = "https://files.pythonhosted.org/packages/39/b6/19c8c592baeeb94b75f966547d40c02ac7590902306ec5863d5c027cf506/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:42a67efc36300d052fb4508a53e8b6901b9284b599ae63945c377569c5fcc1e1", size = 1756239, upload-time = "2026-07-23T01:55:23.705Z" }, + { url = "https://files.pythonhosted.org/packages/dc/c9/4e9383150296f97f873b680c4de8fb2cd88608fb9f48c79edcb111611abc/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:7a75aa63cbf9b21cfaf60dc2657e19df2c2867d91707d653fee171ffeedd1371", size = 1769161, upload-time = "2026-07-23T01:55:26.082Z" }, + { url = "https://files.pythonhosted.org/packages/aa/1e/147bdc6cc5de5f3ab011be8bf5d6e786633249f22c20bae06f85e45f5387/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:e92eb8acc45eb6a9f4935071a77edf5b85cc6f8dfad5cd99e97653c26593cdde", size = 1578759, upload-time = "2026-07-23T01:55:28.846Z" }, + { url = "https://files.pythonhosted.org/packages/fd/31/78388a9d6040ece2e11df62ea229a822cf5e52d238374b220ae9975b2623/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:b014a6ed7cf912e787149fdc529166d3ceabac23f26efeea3158c9aba2354e7e", size = 1792025, upload-time = "2026-07-23T01:55:31.457Z" }, + { url = "https://files.pythonhosted.org/packages/03/51/a3d29fdf2c25d796746af8ad6fe56a45d6256c38b0a8a2ed752e1160b3a2/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:3d4f72af88ac2474bb5bca640030320e3d38a0163a1d7533500e87be458eef71", size = 1768477, upload-time = "2026-07-23T01:55:33.87Z" }, + { url = "https://files.pythonhosted.org/packages/29/a6/442e18b5afeade534d877a2dc3c3e392aff8d49787890b0cf84790410267/aiohttp-3.14.3-cp313-cp313-win32.whl", hash = "sha256:5f08ec777f35ee70720233b8b9811d3bb5d728137f30ac91b7457709c3261ac0", size = 451069, upload-time = "2026-07-23T01:55:36.121Z" }, + { url = "https://files.pythonhosted.org/packages/9d/69/3d876ac02659f271cf7f6769f14a8e3de5b6e888ed8b5a7e998086a4cec8/aiohttp-3.14.3-cp313-cp313-win_amd64.whl", hash = "sha256:dff9461ec275f22135650d5ba4b4931a11f3958df7dfbb8db630000d4dee0883", size = 476518, upload-time = "2026-07-23T01:55:38.303Z" }, + { url = "https://files.pythonhosted.org/packages/b2/0e/50d6e6471cd31edce8b282bdec59375a3a69124d8a989a0b1313355cae52/aiohttp-3.14.3-cp313-cp313-win_arm64.whl", hash = "sha256:ddcac3c6b382e81f1dd0499199d4136b877beb4cb5ef770bbbfba56c4b8f55d2", size = 447676, upload-time = "2026-07-23T01:55:40.451Z" }, +] + +[[package]] +name = "aiosignal" +version = "1.4.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "frozenlist" }, + { name = "typing-extensions", marker = "python_full_version < '3.13'" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/61/62/06741b579156360248d1ec624842ad0edf697050bbaf7c3e46394e106ad1/aiosignal-1.4.0.tar.gz", hash = "sha256:f47eecd9468083c2029cc99945502cb7708b082c232f9aca65da147157b251c7", size = 25007, upload-time = "2025-07-03T22:54:43.528Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/fb/76/641ae371508676492379f16e2fa48f4e2c11741bd63c48be4b12a6b09cba/aiosignal-1.4.0-py3-none-any.whl", hash = "sha256:053243f8b92b990551949e63930a839ff0cf0b0ebbe0597b0f3fb19e1a0fe82e", size = 7490, upload-time = "2025-07-03T22:54:42.156Z" }, +] + [[package]] name = "annotated-types" version = "0.7.0" @@ -37,6 +136,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/78/b6/6307fbef88d9b5ee7421e68d78a9f162e0da4900bc5f5793f6d3d0e34fb8/annotated_types-0.7.0-py3-none-any.whl", hash = "sha256:1f02e8b43a8fbbc3f3e0d4f0f4bfc8131bcb4eebe8849b8e5c773f3a1c582a53", size = 13643, upload-time = "2024-05-20T21:33:24.1Z" }, ] +[[package]] +name = "attrs" +version = "26.1.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/9a/8e/82a0fe20a541c03148528be8cac2408564a6c9a0cc7e9171802bc1d26985/attrs-26.1.0.tar.gz", hash = "sha256:d03ceb89cb322a8fd706d4fb91940737b6642aa36998fe130a9bc96c985eff32", size = 952055, upload-time = "2026-03-19T14:22:25.026Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/64/b4/17d4b0b2a2dc85a6df63d1157e028ed19f90d4cd97c36717afef2bc2f395/attrs-26.1.0-py3-none-any.whl", hash = "sha256:c647aa4a12dfbad9333ca4e71fe62ddc36f4e63b2d260a37a8b83d2f043ac309", size = 67548, upload-time = "2026-03-19T14:22:23.645Z" }, +] + [[package]] name = "cel-python" version = "0.5.0" @@ -53,6 +161,93 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/1e/f8/38812adc3f787c2c2e8ba56f524185ed379656c10b40347a32796ba61c08/cel_python-0.5.0-py3-none-any.whl", hash = "sha256:d0f85008b89655c2bb18d797d2fa3f96f2ed80f4a3b43b0e8138c6646581e5f6", size = 84950, upload-time = "2026-01-31T19:07:11.821Z" }, ] +[[package]] +name = "certifi" +version = "2026.7.22" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/a3/c2/24167ea9858356b47a87a50d39908bfdb72ceeefe0041586e704e5376b3a/certifi-2026.7.22.tar.gz", hash = "sha256:741e2c3b351ddf169a738da9f2c048608ff7f2c5cc02f1ebc6b118bb090d5d55", size = 138112, upload-time = "2026-07-22T03:35:12.644Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/0b/a7/71ac2cff56fec219ed242bb11b8efb69fcc4bec75db06fb7bfe35de520e6/certifi-2026.7.22-py3-none-any.whl", hash = "sha256:62f22742b58a1a33014a2b6b706588a8d7e2a88ae7bd1a6ebe8c992928483775", size = 136983, upload-time = "2026-07-22T03:35:11.276Z" }, +] + +[[package]] +name = "charset-normalizer" +version = "3.5.2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/33/1c/f41d4e74c28ab327ff3acd36053f7ea506c55872d7a90b0fa71aa3ab0c89/charset_normalizer-3.5.2.tar.gz", hash = "sha256:39de2a259fc954455c57274dc94c79d5842774e1247a016aff30bc0efed0f4ef", size = 172659, upload-time = "2026-09-30T04:39:23.398Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/22/67/6a0b94a7960d5e1b5eacd2fb529f3fccc47db4644f7f0a7cfdcfc3be578a/charset_normalizer-3.5.2-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:3d21b8b13c7592db2ac5e544a6d83187b995257472b0c9e8351b6d507ae37ed6", size = 370642, upload-time = "2026-09-30T04:35:06.91Z" }, + { url = "https://files.pythonhosted.org/packages/fb/94/01009e13b94041599004edf32e56e382c24e570f60f79bab8efe45cfe1eb/charset_normalizer-3.5.2-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d760fe2a4d7c3b226cb9026d6a842868d52a7901bd98420e1baf14e80da85cf5", size = 259280, upload-time = "2026-09-30T04:35:08.448Z" }, + { url = "https://files.pythonhosted.org/packages/66/85/3b5358f60a13210f0b67d3755c168ef758701b021e655d88d4da28554467/charset_normalizer-3.5.2-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:c9790464842f85f437dbbb54417eda1e0e6bfc52dd8d22d6fd1c994b73b2dc74", size = 245335, upload-time = "2026-09-30T04:35:10.104Z" }, + { url = "https://files.pythonhosted.org/packages/74/75/77c1c479b09ecd751d1e767b251ea5c14d4d50ff757bf404afab2692f600/charset_normalizer-3.5.2-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:4685902cf26edf013ed7a3da0f426ebba7a00ebb9541386d835afbf002c11cab", size = 291707, upload-time = "2026-09-30T04:35:11.575Z" }, + { url = "https://files.pythonhosted.org/packages/0b/0d/363f78cacb70f58f15f4b083961bbd9d292f335d3f5c66fc4f1cfe69cb90/charset_normalizer-3.5.2-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:4495c5002a7b28557e7e222e77e0b661183e432b7d6d2e788101e3f240e05b8c", size = 286747, upload-time = "2026-09-30T04:35:13.022Z" }, + { url = "https://files.pythonhosted.org/packages/e4/ed/cf505d3011ffceb12c2067a7a5d3cfe92b875d4d44bb0ff0d69375e2c184/charset_normalizer-3.5.2-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:211d5a3eb6af8f513b8d4ca19a8c1b7accab1b5f0d3175f9826b03c1a920dc1f", size = 269972, upload-time = "2026-09-30T04:35:14.606Z" }, + { url = "https://files.pythonhosted.org/packages/15/d8/f0a93a431d170e7ca681d4f6650fee3de934d18560e474e7267eb4b0f987/charset_normalizer-3.5.2-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:ef4fcbf3327382cd4c9f540babd61248208af7b93eec4de397b4d5f58a09e288", size = 269339, upload-time = "2026-09-30T04:35:16.087Z" }, + { url = "https://files.pythonhosted.org/packages/86/bd/9b2bd1c5b7af02462c9752d33994834ff972a96b4c483eefde9e594488e2/charset_normalizer-3.5.2-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:bd16aabe4a02a297c23417aa17ac6299dbd8c49f673bcd645b4929b11f5a4400", size = 261683, upload-time = "2026-09-30T04:35:17.488Z" }, + { url = "https://files.pythonhosted.org/packages/76/a5/cac540ab0fd61f3fec88ad3dbb64509e71424593d73cfdfff5ab3e4db279/charset_normalizer-3.5.2-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:fb9e68df06293761f9fe66ade60a9bc6d0f5e42b8acf2939a9158af86ab0e5bd", size = 248143, upload-time = "2026-09-30T04:35:18.849Z" }, + { url = "https://files.pythonhosted.org/packages/71/7a/ff467301deef2089fad87f72df9e000a26a78fec7acbb18e1999371b8369/charset_normalizer-3.5.2-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:59f63901b0031c3136cf64704dcb21de0bbae62ce2c9529bc39d27665463de37", size = 292358, upload-time = "2026-09-30T04:35:20.326Z" }, + { url = "https://files.pythonhosted.org/packages/ad/77/22d7e785d1e210afc2e2f58600dd1799d17a35665faf84383f002826c5f8/charset_normalizer-3.5.2-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:304d5463e65a35d7bb0850550e0780395395f6fcf452f04db7d5ca7cecc425ac", size = 268766, upload-time = "2026-09-30T04:35:21.72Z" }, + { url = "https://files.pythonhosted.org/packages/ae/91/e8e946267f1c2d9e2bd651726e2fbd2addf02c4d36cea5069e32ca9d7bb5/charset_normalizer-3.5.2-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:9cf9b1a857e25c4baceeb3624e92a56df3668f398c4acba74e174d81fb4d1d3a", size = 287432, upload-time = "2026-09-30T04:35:23.273Z" }, + { url = "https://files.pythonhosted.org/packages/4e/88/7561d8a88d555e7df6623abe7c0070b4baf47549b9408783a2ae0a1a6cf7/charset_normalizer-3.5.2-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:114e4d0c92d618409ed82a99e22b5c5e768fe995f2973f78265f4524f49d4640", size = 272470, upload-time = "2026-09-30T04:35:24.655Z" }, + { url = "https://files.pythonhosted.org/packages/35/7e/578c702301ec036f01455f30744a08d2b42f6ab35b9b2d4bf8cae0ef2a80/charset_normalizer-3.5.2-cp311-cp311-win32.whl", hash = "sha256:2625388c6c754520c37abaf3b41eb34d1cc4a373f457898f08606c8e362b891d", size = 186688, upload-time = "2026-09-30T04:35:26.225Z" }, + { url = "https://files.pythonhosted.org/packages/e8/fc/fdf8cf52ff21cd5bf158f20978991cf985325842f74283eb6df26c8a39d8/charset_normalizer-3.5.2-cp311-cp311-win_amd64.whl", hash = "sha256:87e50a3e7cb90af586b6c5faf23e302a970415ac73bd7bd90a515a04b427ef96", size = 214932, upload-time = "2026-09-30T04:35:27.796Z" }, + { url = "https://files.pythonhosted.org/packages/97/66/3e45a506d8110b632541faf9a9470185aa9878f1ed44020f31346c1c5e5b/charset_normalizer-3.5.2-cp311-cp311-win_arm64.whl", hash = "sha256:254eb48b9fa5ee9898a3c445825a1f340fe53712a098904b39b0bddba8ea3cb1", size = 202828, upload-time = "2026-09-30T04:35:29.259Z" }, + { url = "https://files.pythonhosted.org/packages/e7/c8/693809898870237d82785a03f3b2b58fe4c9f14669f84a7d4e623c92a59e/charset_normalizer-3.5.2-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:ed2a239c0ea213acc1908150a3037257083c7c083128f1a4cec2ec4b97dca491", size = 367780, upload-time = "2026-09-30T04:35:30.888Z" }, + { url = "https://files.pythonhosted.org/packages/c9/87/2fea8c13dc24b3ca9c6f803a5b2dfdeae73eb4f9e12c7885ed908ff0433c/charset_normalizer-3.5.2-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:b91363207bd9dc966a691e959bb47f64b30f7ac4b072be9968b366982f7db77c", size = 246730, upload-time = "2026-09-30T04:35:32.286Z" }, + { url = "https://files.pythonhosted.org/packages/a8/9e/09efac30b937722f46d3110ba30b875b24b2e3a266ed746cc4e376a94d80/charset_normalizer-3.5.2-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:38a873987f3be698494da8b2e3085e29da02da7b633dce73e79c699a113d7bf0", size = 237707, upload-time = "2026-09-30T04:35:33.709Z" }, + { url = "https://files.pythonhosted.org/packages/9e/18/70d76670b13686237863a379928d60bd10e021f17d243ab3d7014c4a5f4e/charset_normalizer-3.5.2-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:355ad8011081dec5412240c087a9a0c9d4d5039f3ed11a3f13e18c2b29b56c51", size = 273050, upload-time = "2026-09-30T04:35:35.138Z" }, + { url = "https://files.pythonhosted.org/packages/54/e2/77a8b09d5adc013ed07b95b01b8b8fa5441c4e810e83ee7e4aae2fa4d91a/charset_normalizer-3.5.2-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:ee21e28f0430bd6dc9086c6e525d5e818a44a5ad19720c8a0ef766792f3eb5e5", size = 270345, upload-time = "2026-09-30T04:35:36.502Z" }, + { url = "https://files.pythonhosted.org/packages/7f/c5/38806a25ab5e65fc178f39affeda20858efafede2fce1ffc2556cfc9fe73/charset_normalizer-3.5.2-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:3d31298449090ab8d47b7b1b2a555ff73cac7ed438a08b7ac160980c7ebed649", size = 257601, upload-time = "2026-09-30T04:35:37.919Z" }, + { url = "https://files.pythonhosted.org/packages/ae/8d/213565184708fdb263ae55e2c04ee1ff748129dd65d48ed0e3502da9c85a/charset_normalizer-3.5.2-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:5cde776b7cc66e4f6c99612cea4aa7269aa65863f7a15841b2c264f103822f4e", size = 252222, upload-time = "2026-09-30T04:35:39.544Z" }, + { url = "https://files.pythonhosted.org/packages/7e/24/76d2cefc25472531e4c5c7dfff68865eb1c39b78482f0fdc15b46f047830/charset_normalizer-3.5.2-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:ae4f5fea5b8b8ccff88238cc8569303e5ee95efae67fa62922a311397a71f346", size = 248482, upload-time = "2026-09-30T04:35:41.088Z" }, + { url = "https://files.pythonhosted.org/packages/7d/dc/65a801b66ab4c197e22c433ab25e7ac24324ac6f45a2269aca42cce309bf/charset_normalizer-3.5.2-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:f7d486c83842422badd511868fd8a9a20e9407ace71564b6af47ce7e60a336c1", size = 241206, upload-time = "2026-09-30T04:35:42.59Z" }, + { url = "https://files.pythonhosted.org/packages/a7/95/ca9b5eabde673002c6f1e7ada1b223916fe18f6d661da7aabd4d643718f1/charset_normalizer-3.5.2-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:11a4d68a6ecda3292cb1e50239e111543ba5d709bb62a6b4ea1afcfa729d8875", size = 273190, upload-time = "2026-09-30T04:35:44.347Z" }, + { url = "https://files.pythonhosted.org/packages/2d/8b/803b4d2a3f6e1740f63f1e87b04d14b42f3d4fdfe6ed7d4db2d34102b14f/charset_normalizer-3.5.2-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:d6734d2ef8a50fbf8445c139477da401f50d62a0606bf00e20ec6d87773fefb1", size = 253527, upload-time = "2026-09-30T04:35:45.915Z" }, + { url = "https://files.pythonhosted.org/packages/a9/55/93c0e5dbd085ae0471346026abbe7e0db9ea2d6fea74e51f0b5a46f233a7/charset_normalizer-3.5.2-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:a815775b6c38d4e0ff7bcffbeba67feded90202bb6a226b8dd35f1c855217413", size = 271285, upload-time = "2026-09-30T04:35:47.49Z" }, + { url = "https://files.pythonhosted.org/packages/95/69/0dbd0e0b9b16cfa816cdfcb3e2e3854a1f680dc07fb1245ea125e7448060/charset_normalizer-3.5.2-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:23851fb4e1b85ed3f6c2a27b777cdfe2e19fb5b38429a8faf38c7542b7665869", size = 260010, upload-time = "2026-09-30T04:35:48.996Z" }, + { url = "https://files.pythonhosted.org/packages/58/9d/e7b88e7b1bf403590c3b573277b5e1e488c68c7a6fbacca310a2c324e90c/charset_normalizer-3.5.2-cp312-cp312-win32.whl", hash = "sha256:db19d07e2e0129e974a0e65d0064fc222a446cd5122c2fd4184d2af9fc734a9e", size = 184125, upload-time = "2026-09-30T04:35:50.777Z" }, + { url = "https://files.pythonhosted.org/packages/eb/e6/e6e083884cbcfd49c64865af05027fe7011be7b2d9179524f099a1b611f3/charset_normalizer-3.5.2-cp312-cp312-win_amd64.whl", hash = "sha256:780fbe7cab297b81dad9fb8dc5eb003c0468ffb0d9e5f65068c53a34661a96bc", size = 207486, upload-time = "2026-09-30T04:35:52.194Z" }, + { url = "https://files.pythonhosted.org/packages/c4/e3/017aea0911ada7405a825c7d937eb3a13009664e2f5b38e8c4bbf2abf894/charset_normalizer-3.5.2-cp312-cp312-win_arm64.whl", hash = "sha256:e2af3aad578aa6bd1384bcf4750fc285e5a9de53f40b7d41e5a0bf748edeb2b3", size = 196734, upload-time = "2026-09-30T04:35:53.636Z" }, + { url = "https://files.pythonhosted.org/packages/c5/34/68292d68512768591aaff07c59bb53ee31341c87759433a859c4641a50c2/charset_normalizer-3.5.2-cp313-cp313-android_24_arm64_v8a.whl", hash = "sha256:ed905975ab14056a2e5eb1c376cb2e1ebc5396baf84163939c518556fccde9f5", size = 225946, upload-time = "2026-09-30T04:35:55.313Z" }, + { url = "https://files.pythonhosted.org/packages/e3/80/bee0b01b90ccd5322ae1d0abb33fab1bd95b7c2eadaf02aeccf22e04ee83/charset_normalizer-3.5.2-cp313-cp313-android_24_x86_64.whl", hash = "sha256:a66c3bc5ab1f0ff2164fc9965ddd611ff0802173f4b9d24554c563f6ab7e1d6e", size = 238619, upload-time = "2026-09-30T04:35:56.863Z" }, + { url = "https://files.pythonhosted.org/packages/78/6e/60ce52a85a7fd631ae8482ae6d74521014ca2f255892679484dc04d7ef56/charset_normalizer-3.5.2-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:d2374b62878abb00cd8309b32af6c0b715cd02dec0ca74ef12e5069bdc64144a", size = 206701, upload-time = "2026-09-30T04:35:58.639Z" }, + { url = "https://files.pythonhosted.org/packages/36/8c/71aafad23f971afc84c2b295bc0c560739ce1dac558aad9fec22e39f3639/charset_normalizer-3.5.2-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:d376bbd28b3a8999db1a103b3b388aee6f1ddeb3e51bc2172993efdcd86e064d", size = 210506, upload-time = "2026-09-30T04:36:00.147Z" }, + { url = "https://files.pythonhosted.org/packages/91/da/3c5a7798c046df7d2d68ad653cf5b6c5a8bfee225055a843c6f2f42aac1a/charset_normalizer-3.5.2-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:6045373d5a89a5ec71afde535db987ca28e76dfa276c2d4c818265b375d4b055", size = 366626, upload-time = "2026-09-30T04:36:01.77Z" }, + { url = "https://files.pythonhosted.org/packages/e1/16/710ac3de2ee354e2bd1a9c94efe45a2d27b5c6ad39b2d6a905be2c094b6c/charset_normalizer-3.5.2-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:849df64e889b2e17230d58410a03dba311a65b163508fd33679b2b737d4b7858", size = 244624, upload-time = "2026-09-30T04:36:03.389Z" }, + { url = "https://files.pythonhosted.org/packages/d6/39/45c7439f5b63d24f7d5b2a1d760f34af7628782d7144b4cc8ded45c2d4bc/charset_normalizer-3.5.2-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:15c44f7edfd477b06f517a5cc317fc1707edb9de2c865f43d4b6513907473234", size = 235731, upload-time = "2026-09-30T04:36:04.987Z" }, + { url = "https://files.pythonhosted.org/packages/4d/34/38f3154785ce92e9f56eb226f4d35bdfae6b008480dd055f58837a89c810/charset_normalizer-3.5.2-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:a89012d6d5476ee112d20d998570ed58df2260a852afb1758809cd6900411d21", size = 270352, upload-time = "2026-09-30T04:36:06.412Z" }, + { url = "https://files.pythonhosted.org/packages/04/f3/859f74e7babc977705026b30593b3be04049632a522fb7000f83c033d747/charset_normalizer-3.5.2-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:0c951d5e6dd9c2ff60609476752bee49da4206adde960ebc247766937f72e718", size = 267393, upload-time = "2026-09-30T04:36:07.865Z" }, + { url = "https://files.pythonhosted.org/packages/4b/85/41d27f234b82e47c167a5f6c0f62501dc0c640585ff4aba79e08a390336a/charset_normalizer-3.5.2-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:7218e8f32b0956cfcd048fd42d9d5779809745ca1d86113ca56f66e7ae1549c4", size = 254660, upload-time = "2026-09-30T04:36:09.248Z" }, + { url = "https://files.pythonhosted.org/packages/58/ca/5d1a997587febe5b26d8daffe363b5c1a091cece19828eec6502fd09c5ef/charset_normalizer-3.5.2-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:a19a731138fc27d5682277d3b9df22855cea1239bce7fcec5f78f42ef2d1f3c3", size = 249996, upload-time = "2026-09-30T04:36:10.73Z" }, + { url = "https://files.pythonhosted.org/packages/b3/1f/d1e78246f7ed60c8c8d606b4ac27f66ce49cc3e95f24893ccbeba9f77302/charset_normalizer-3.5.2-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:62603db9a7caa0802eaa28c1c46fecd7b3a263a774069c24c3c28c302448721c", size = 247028, upload-time = "2026-09-30T04:36:12.294Z" }, + { url = "https://files.pythonhosted.org/packages/8e/37/eba316edd4f0c4d3a5d945924c4eeeae59abac4056aa815d8a4268f863a2/charset_normalizer-3.5.2-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:b6856554c4f44d79fc2307d5768854310a8f0096e501c75637542c82292b0429", size = 240562, upload-time = "2026-09-30T04:36:13.887Z" }, + { url = "https://files.pythonhosted.org/packages/c8/8e/aaa037d40ca9ef045977f1a661048b1aa33f223adfce3452fe9be9f79d14/charset_normalizer-3.5.2-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:1bc0baf5ef96b6ede57d47f4b8fe4d9d84019c3bfcbeb20a41edc6a6ee341f1f", size = 271699, upload-time = "2026-09-30T04:36:15.41Z" }, + { url = "https://files.pythonhosted.org/packages/26/19/1c1c9f75974adf523b87f34b8a2adc5a435cd65916812bcbd0dfa45f9a29/charset_normalizer-3.5.2-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:56bc200a365efb37383b7852e4cc5898d3b2da5987289b543956cf8cad71018a", size = 252030, upload-time = "2026-09-30T04:36:16.839Z" }, + { url = "https://files.pythonhosted.org/packages/bc/90/0660ef18e18df0a4d2a1a0edff7dfbba42d4e50ef2425557a5bb7051f77b/charset_normalizer-3.5.2-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:2c9ad19a6cfcd5ea5c0d41161d22f9df1dcc277e9bef2751391334546a314c00", size = 268565, upload-time = "2026-09-30T04:36:18.468Z" }, + { url = "https://files.pythonhosted.org/packages/79/ba/57adc269824e8658f1a0f97a9e514c247445a9632b3419b97e0ba37f16dc/charset_normalizer-3.5.2-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:e243bd13217235fc7290c621941c3f5cc8b66e4872495be821d7436ba2fb838d", size = 257627, upload-time = "2026-09-30T04:36:19.938Z" }, + { url = "https://files.pythonhosted.org/packages/9a/85/33abd4315c052d3d4f54c92b1ee49bfbc0dc7115a981e462a793b6d2ab87/charset_normalizer-3.5.2-cp313-cp313-pyemscripten_2025_0_wasm32.whl", hash = "sha256:a090bb2c68df85450502e3e20d665e3a5af9c65a84d6508ed477badd49166fd3", size = 143407, upload-time = "2026-09-30T04:36:21.376Z" }, + { url = "https://files.pythonhosted.org/packages/4f/de/6435e18d1aaa5d910b896d551411c96af1f42a0c56c29afc2016c61ccc2e/charset_normalizer-3.5.2-cp313-cp313-win32.whl", hash = "sha256:2b7b3bbfb4fe8ef40600792d762fbaa9057559f9d3fad209525b7a22b99e91fd", size = 183207, upload-time = "2026-09-30T04:36:22.776Z" }, + { url = "https://files.pythonhosted.org/packages/9c/76/b8ec57f4e9ee3253541abf95e4a462c0175fe8032dcd070f1f2421240942/charset_normalizer-3.5.2-cp313-cp313-win_amd64.whl", hash = "sha256:78456a747de8dc58360ffa581f30a002baf5aa28cb262536545e91f113ed7639", size = 206489, upload-time = "2026-09-30T04:36:24.306Z" }, + { url = "https://files.pythonhosted.org/packages/3e/60/c647c6ae47480221e875ea5d743ff94946f7416e3c69415ab772928e8d32/charset_normalizer-3.5.2-cp313-cp313-win_arm64.whl", hash = "sha256:11912e4bb14baae7c5d8791aa55ba0a3a03ec6729073307b0f57270abaa713d3", size = 195501, upload-time = "2026-09-30T04:36:25.846Z" }, + { url = "https://files.pythonhosted.org/packages/8c/ab/176fbfd5b64939c55d652366aa5b9ef1d767af207a3aa6ebeb0d226c484d/charset_normalizer-3.5.2-cp37-abi3-macosx_10_9_universal2.whl", hash = "sha256:4275811936e2f06feff5e598fb42a1b7ae852da8e39605211892b56b81a34efd", size = 331815, upload-time = "2026-09-30T04:38:26.216Z" }, + { url = "https://files.pythonhosted.org/packages/7e/84/371eac6b30bdbcbf2d632a1a01809103459216fcaae61b8b8d922c1bfb8a/charset_normalizer-3.5.2-cp37-abi3-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:1c50fe28bbc2ced33386f298650d91218076c05420e6cbd790b913adc41659e7", size = 253276, upload-time = "2026-09-30T04:38:28.032Z" }, + { url = "https://files.pythonhosted.org/packages/43/6f/c4fbae58febff71709c51bc7e18fdfa55341dc382704740f9f0cbf03817b/charset_normalizer-3.5.2-cp37-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d19fbd981a488e22cd04883659ca6b08f50b5974f9fd7c95655ef6a043e5893f", size = 241239, upload-time = "2026-09-30T04:38:29.732Z" }, + { url = "https://files.pythonhosted.org/packages/61/71/458c3f42164a07d0c5210798e9e704b39e540a6793b05aba67f3a35243a9/charset_normalizer-3.5.2-cp37-abi3-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:0fed1d06615f022ee3b13caf5e8b180cfea32bb2c5aded8a9d44277afc040f93", size = 231121, upload-time = "2026-09-30T04:38:31.462Z" }, + { url = "https://files.pythonhosted.org/packages/09/54/ab9e89367076f6331bb6c65c4bf14a5361fa5191cb6561bf534f18504e1b/charset_normalizer-3.5.2-cp37-abi3-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:838dcc90063569a0448120554591a1d6c4a4ffe11babf048908793154ab86ade", size = 260350, upload-time = "2026-09-30T04:38:33.239Z" }, + { url = "https://files.pythonhosted.org/packages/7c/c1/061431ecc688d9d76602502cb57cc01e691e682c18f1beb45f9673b5bbd2/charset_normalizer-3.5.2-cp37-abi3-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:2ce45c6627b22c47e390bc91a41c3d13032192e699fa0bea96e9671b373d69b0", size = 255430, upload-time = "2026-09-30T04:38:34.865Z" }, + { url = "https://files.pythonhosted.org/packages/8d/1f/20c8949f0676f7ab811abdeb7f4d7f1cbc6e61ff20bef08b44edeb092bc8/charset_normalizer-3.5.2-cp37-abi3-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:0774bf9bf620249fee3e0b8b9fd3065de213be30f3aa94ce2494b3b638949e26", size = 250612, upload-time = "2026-09-30T04:38:36.649Z" }, + { url = "https://files.pythonhosted.org/packages/2b/9e/46f2fa4c431fc98c4ae76a8cb5bdca54e0341e3cfc3fcfd8e82740250818/charset_normalizer-3.5.2-cp37-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:1db38f4c5496827c1a501846d64d14c3b80c7e6714e406cd7dc36a9899fa1011", size = 242083, upload-time = "2026-09-30T04:38:38.26Z" }, + { url = "https://files.pythonhosted.org/packages/bd/39/559be29a0c0f086e0bba6922babd38916cc5e0b58ced4de13ee01ea05508/charset_normalizer-3.5.2-cp37-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:304d8e4d493af723536393eee0c689eb7813f4a474c8b479dee63f1fdd98f621", size = 232738, upload-time = "2026-09-30T04:38:39.81Z" }, + { url = "https://files.pythonhosted.org/packages/ff/6c/387b0e4f756a282831c1d9fc6aeb6c51ca4507ca202767c8de15ce9b12e2/charset_normalizer-3.5.2-cp37-abi3-musllinux_1_2_ppc64le.whl", hash = "sha256:9b7f416ff0978e2f2249330527f0ad6fa02f4932e6199692d3b52da2048c19e4", size = 260703, upload-time = "2026-09-30T04:38:41.346Z" }, + { url = "https://files.pythonhosted.org/packages/96/92/1fdf015f09ef449f50d3ac4b67c90887c9c318b727daa95cc4f866e6521d/charset_normalizer-3.5.2-cp37-abi3-musllinux_1_2_riscv64.whl", hash = "sha256:01077390b03f7988f11d700a2194e69b119741a86b1a638b1db88891e3eced8e", size = 247622, upload-time = "2026-09-30T04:38:42.937Z" }, + { url = "https://files.pythonhosted.org/packages/dc/3c/8e7b8a5671ad5d433669fb2a76f1a0164df2d9b1718b0206bc2a16d840cc/charset_normalizer-3.5.2-cp37-abi3-musllinux_1_2_s390x.whl", hash = "sha256:7e841fb9010836c992c9f12fcbd43a831de93a5f726fc1ccd8ca1d0268c5014c", size = 257500, upload-time = "2026-09-30T04:38:44.604Z" }, + { url = "https://files.pythonhosted.org/packages/b4/f0/45b579df5cabc1d5d53ea1cc35e8437d3ca768c0acccc7041517cb6fbb32/charset_normalizer-3.5.2-cp37-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:9cae88599c7219005d879f98e5ed53341e9a122af585e1091200358a3003d2a0", size = 255100, upload-time = "2026-09-30T04:38:46.289Z" }, + { url = "https://files.pythonhosted.org/packages/31/68/fdec18a343f5fb3f310588dd478b09ac4799e0b187dbade3a8cd776f03ef/charset_normalizer-3.5.2-cp37-abi3-win32.whl", hash = "sha256:01b0c0d2262a9e28e8484a278c7e1b5d650e3ac8cf2683d2967e25899f208bdf", size = 174499, upload-time = "2026-09-30T04:38:47.999Z" }, + { url = "https://files.pythonhosted.org/packages/9d/8a/b618149cc5207943a0242068d7a27897f56a62947b5a039085f2a22029f8/charset_normalizer-3.5.2-cp37-abi3-win_amd64.whl", hash = "sha256:9f56f72050826f63dcee7a7f55b0a77168cb3bfc553fd405e7f8f9ece75a4036", size = 200092, upload-time = "2026-09-30T04:38:49.707Z" }, + { url = "https://files.pythonhosted.org/packages/03/cf/4c66866fa9e2b1c78e3c911516d1de497a677b7ac60f1eceda74ce777ca3/charset_normalizer-3.5.2-cp37-abi3-win_arm64.whl", hash = "sha256:40ab6bffa02ae10a0581e6c198be7d2d8ca5c2a0c64e4ed3465d766df457573e", size = 294363, upload-time = "2026-09-30T04:38:51.312Z" }, + { url = "https://files.pythonhosted.org/packages/fc/ad/d07d7862a62ffa6d79d68074d14823243dd235a77c45262acbf6adeb28bf/charset_normalizer-3.5.2-py3-none-any.whl", hash = "sha256:b6b751274acb69d77b3323d6b7dbaa3c7fdfc1eb829b7eb61d262f32e1af9685", size = 68872, upload-time = "2026-09-30T04:39:21.828Z" }, +] + [[package]] name = "click" version = "8.4.1" @@ -403,6 +598,88 @@ name = "crossplane-models" version = "0.0.0" source = { editable = "schemas/python" } +[[package]] +name = "durationpy" +version = "0.11" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/b0/5d/5f8571bd5dedc80863191621ac4be001f3f3dd8315d2ec078705dab7dec1/durationpy-0.11.tar.gz", hash = "sha256:181898e1ae282e288f0a2291829656bf1b6b3aadf30a97993b85db4943642905", size = 3582, upload-time = "2026-08-26T13:56:00.991Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e6/c4/ebdf7837bc4ef6fd98cfb013c28855bb358467bf86c1af011bbc21e21df0/durationpy-0.11-py3-none-any.whl", hash = "sha256:a739fe2b8972c250ff72f8e2c488d18cf25f7b852f49ee76048775d5171df30c", size = 4133, upload-time = "2026-08-26T13:55:59.456Z" }, +] + +[[package]] +name = "frozenlist" +version = "1.8.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/2d/f5/c831fac6cc817d26fd54c7eaccd04ef7e0288806943f7cc5bbf69f3ac1f0/frozenlist-1.8.0.tar.gz", hash = "sha256:3ede829ed8d842f6cd48fc7081d7a41001a56f1f38603f9d49bf3020d59a31ad", size = 45875, upload-time = "2025-10-06T05:38:17.865Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/bc/03/077f869d540370db12165c0aa51640a873fb661d8b315d1d4d67b284d7ac/frozenlist-1.8.0-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:09474e9831bc2b2199fad6da3c14c7b0fbdd377cce9d3d77131be28906cb7d84", size = 86912, upload-time = "2025-10-06T05:35:45.98Z" }, + { url = "https://files.pythonhosted.org/packages/df/b5/7610b6bd13e4ae77b96ba85abea1c8cb249683217ef09ac9e0ae93f25a91/frozenlist-1.8.0-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:17c883ab0ab67200b5f964d2b9ed6b00971917d5d8a92df149dc2c9779208ee9", size = 50046, upload-time = "2025-10-06T05:35:47.009Z" }, + { url = "https://files.pythonhosted.org/packages/6e/ef/0e8f1fe32f8a53dd26bdd1f9347efe0778b0fddf62789ea683f4cc7d787d/frozenlist-1.8.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:fa47e444b8ba08fffd1c18e8cdb9a75db1b6a27f17507522834ad13ed5922b93", size = 50119, upload-time = "2025-10-06T05:35:48.38Z" }, + { url = "https://files.pythonhosted.org/packages/11/b1/71a477adc7c36e5fb628245dfbdea2166feae310757dea848d02bd0689fd/frozenlist-1.8.0-cp311-cp311-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:2552f44204b744fba866e573be4c1f9048d6a324dfe14475103fd51613eb1d1f", size = 231067, upload-time = "2025-10-06T05:35:49.97Z" }, + { url = "https://files.pythonhosted.org/packages/45/7e/afe40eca3a2dc19b9904c0f5d7edfe82b5304cb831391edec0ac04af94c2/frozenlist-1.8.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:957e7c38f250991e48a9a73e6423db1bb9dd14e722a10f6b8bb8e16a0f55f695", size = 233160, upload-time = "2025-10-06T05:35:51.729Z" }, + { url = "https://files.pythonhosted.org/packages/a6/aa/7416eac95603ce428679d273255ffc7c998d4132cfae200103f164b108aa/frozenlist-1.8.0-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:8585e3bb2cdea02fc88ffa245069c36555557ad3609e83be0ec71f54fd4abb52", size = 228544, upload-time = "2025-10-06T05:35:53.246Z" }, + { url = "https://files.pythonhosted.org/packages/8b/3d/2a2d1f683d55ac7e3875e4263d28410063e738384d3adc294f5ff3d7105e/frozenlist-1.8.0-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:edee74874ce20a373d62dc28b0b18b93f645633c2943fd90ee9d898550770581", size = 243797, upload-time = "2025-10-06T05:35:54.497Z" }, + { url = "https://files.pythonhosted.org/packages/78/1e/2d5565b589e580c296d3bb54da08d206e797d941a83a6fdea42af23be79c/frozenlist-1.8.0-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:c9a63152fe95756b85f31186bddf42e4c02c6321207fd6601a1c89ebac4fe567", size = 247923, upload-time = "2025-10-06T05:35:55.861Z" }, + { url = "https://files.pythonhosted.org/packages/aa/c3/65872fcf1d326a7f101ad4d86285c403c87be7d832b7470b77f6d2ed5ddc/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:b6db2185db9be0a04fecf2f241c70b63b1a242e2805be291855078f2b404dd6b", size = 230886, upload-time = "2025-10-06T05:35:57.399Z" }, + { url = "https://files.pythonhosted.org/packages/a0/76/ac9ced601d62f6956f03cc794f9e04c81719509f85255abf96e2510f4265/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:f4be2e3d8bc8aabd566f8d5b8ba7ecc09249d74ba3c9ed52e54dc23a293f0b92", size = 245731, upload-time = "2025-10-06T05:35:58.563Z" }, + { url = "https://files.pythonhosted.org/packages/b9/49/ecccb5f2598daf0b4a1415497eba4c33c1e8ce07495eb07d2860c731b8d5/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:c8d1634419f39ea6f5c427ea2f90ca85126b54b50837f31497f3bf38266e853d", size = 241544, upload-time = "2025-10-06T05:35:59.719Z" }, + { url = "https://files.pythonhosted.org/packages/53/4b/ddf24113323c0bbcc54cb38c8b8916f1da7165e07b8e24a717b4a12cbf10/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:1a7fa382a4a223773ed64242dbe1c9c326ec09457e6b8428efb4118c685c3dfd", size = 241806, upload-time = "2025-10-06T05:36:00.959Z" }, + { url = "https://files.pythonhosted.org/packages/a7/fb/9b9a084d73c67175484ba2789a59f8eebebd0827d186a8102005ce41e1ba/frozenlist-1.8.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:11847b53d722050808926e785df837353bd4d75f1d494377e59b23594d834967", size = 229382, upload-time = "2025-10-06T05:36:02.22Z" }, + { url = "https://files.pythonhosted.org/packages/95/a3/c8fb25aac55bf5e12dae5c5aa6a98f85d436c1dc658f21c3ac73f9fa95e5/frozenlist-1.8.0-cp311-cp311-win32.whl", hash = "sha256:27c6e8077956cf73eadd514be8fb04d77fc946a7fe9f7fe167648b0b9085cc25", size = 39647, upload-time = "2025-10-06T05:36:03.409Z" }, + { url = "https://files.pythonhosted.org/packages/0a/f5/603d0d6a02cfd4c8f2a095a54672b3cf967ad688a60fb9faf04fc4887f65/frozenlist-1.8.0-cp311-cp311-win_amd64.whl", hash = "sha256:ac913f8403b36a2c8610bbfd25b8013488533e71e62b4b4adce9c86c8cea905b", size = 44064, upload-time = "2025-10-06T05:36:04.368Z" }, + { url = "https://files.pythonhosted.org/packages/5d/16/c2c9ab44e181f043a86f9a8f84d5124b62dbcb3a02c0977ec72b9ac1d3e0/frozenlist-1.8.0-cp311-cp311-win_arm64.whl", hash = "sha256:d4d3214a0f8394edfa3e303136d0575eece0745ff2b47bd2cb2e66dd92d4351a", size = 39937, upload-time = "2025-10-06T05:36:05.669Z" }, + { url = "https://files.pythonhosted.org/packages/69/29/948b9aa87e75820a38650af445d2ef2b6b8a6fab1a23b6bb9e4ef0be2d59/frozenlist-1.8.0-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:78f7b9e5d6f2fdb88cdde9440dc147259b62b9d3b019924def9f6478be254ac1", size = 87782, upload-time = "2025-10-06T05:36:06.649Z" }, + { url = "https://files.pythonhosted.org/packages/64/80/4f6e318ee2a7c0750ed724fa33a4bdf1eacdc5a39a7a24e818a773cd91af/frozenlist-1.8.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:229bf37d2e4acdaf808fd3f06e854a4a7a3661e871b10dc1f8f1896a3b05f18b", size = 50594, upload-time = "2025-10-06T05:36:07.69Z" }, + { url = "https://files.pythonhosted.org/packages/2b/94/5c8a2b50a496b11dd519f4a24cb5496cf125681dd99e94c604ccdea9419a/frozenlist-1.8.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:f833670942247a14eafbb675458b4e61c82e002a148f49e68257b79296e865c4", size = 50448, upload-time = "2025-10-06T05:36:08.78Z" }, + { url = "https://files.pythonhosted.org/packages/6a/bd/d91c5e39f490a49df14320f4e8c80161cfcce09f1e2cde1edd16a551abb3/frozenlist-1.8.0-cp312-cp312-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:494a5952b1c597ba44e0e78113a7266e656b9794eec897b19ead706bd7074383", size = 242411, upload-time = "2025-10-06T05:36:09.801Z" }, + { url = "https://files.pythonhosted.org/packages/8f/83/f61505a05109ef3293dfb1ff594d13d64a2324ac3482be2cedc2be818256/frozenlist-1.8.0-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:96f423a119f4777a4a056b66ce11527366a8bb92f54e541ade21f2374433f6d4", size = 243014, upload-time = "2025-10-06T05:36:11.394Z" }, + { url = "https://files.pythonhosted.org/packages/d8/cb/cb6c7b0f7d4023ddda30cf56b8b17494eb3a79e3fda666bf735f63118b35/frozenlist-1.8.0-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:3462dd9475af2025c31cc61be6652dfa25cbfb56cbbf52f4ccfe029f38decaf8", size = 234909, upload-time = "2025-10-06T05:36:12.598Z" }, + { url = "https://files.pythonhosted.org/packages/31/c5/cd7a1f3b8b34af009fb17d4123c5a778b44ae2804e3ad6b86204255f9ec5/frozenlist-1.8.0-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:c4c800524c9cd9bac5166cd6f55285957fcfc907db323e193f2afcd4d9abd69b", size = 250049, upload-time = "2025-10-06T05:36:14.065Z" }, + { url = "https://files.pythonhosted.org/packages/c0/01/2f95d3b416c584a1e7f0e1d6d31998c4a795f7544069ee2e0962a4b60740/frozenlist-1.8.0-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:d6a5df73acd3399d893dafc71663ad22534b5aa4f94e8a2fabfe856c3c1b6a52", size = 256485, upload-time = "2025-10-06T05:36:15.39Z" }, + { url = "https://files.pythonhosted.org/packages/ce/03/024bf7720b3abaebcff6d0793d73c154237b85bdf67b7ed55e5e9596dc9a/frozenlist-1.8.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:405e8fe955c2280ce66428b3ca55e12b3c4e9c336fb2103a4937e891c69a4a29", size = 237619, upload-time = "2025-10-06T05:36:16.558Z" }, + { url = "https://files.pythonhosted.org/packages/69/fa/f8abdfe7d76b731f5d8bd217827cf6764d4f1d9763407e42717b4bed50a0/frozenlist-1.8.0-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:908bd3f6439f2fef9e85031b59fd4f1297af54415fb60e4254a95f75b3cab3f3", size = 250320, upload-time = "2025-10-06T05:36:17.821Z" }, + { url = "https://files.pythonhosted.org/packages/f5/3c/b051329f718b463b22613e269ad72138cc256c540f78a6de89452803a47d/frozenlist-1.8.0-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:294e487f9ec720bd8ffcebc99d575f7eff3568a08a253d1ee1a0378754b74143", size = 246820, upload-time = "2025-10-06T05:36:19.046Z" }, + { url = "https://files.pythonhosted.org/packages/0f/ae/58282e8f98e444b3f4dd42448ff36fa38bef29e40d40f330b22e7108f565/frozenlist-1.8.0-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:74c51543498289c0c43656701be6b077f4b265868fa7f8a8859c197006efb608", size = 250518, upload-time = "2025-10-06T05:36:20.763Z" }, + { url = "https://files.pythonhosted.org/packages/8f/96/007e5944694d66123183845a106547a15944fbbb7154788cbf7272789536/frozenlist-1.8.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:776f352e8329135506a1d6bf16ac3f87bc25b28e765949282dcc627af36123aa", size = 239096, upload-time = "2025-10-06T05:36:22.129Z" }, + { url = "https://files.pythonhosted.org/packages/66/bb/852b9d6db2fa40be96f29c0d1205c306288f0684df8fd26ca1951d461a56/frozenlist-1.8.0-cp312-cp312-win32.whl", hash = "sha256:433403ae80709741ce34038da08511d4a77062aa924baf411ef73d1146e74faf", size = 39985, upload-time = "2025-10-06T05:36:23.661Z" }, + { url = "https://files.pythonhosted.org/packages/b8/af/38e51a553dd66eb064cdf193841f16f077585d4d28394c2fa6235cb41765/frozenlist-1.8.0-cp312-cp312-win_amd64.whl", hash = "sha256:34187385b08f866104f0c0617404c8eb08165ab1272e884abc89c112e9c00746", size = 44591, upload-time = "2025-10-06T05:36:24.958Z" }, + { url = "https://files.pythonhosted.org/packages/a7/06/1dc65480ab147339fecc70797e9c2f69d9cea9cf38934ce08df070fdb9cb/frozenlist-1.8.0-cp312-cp312-win_arm64.whl", hash = "sha256:fe3c58d2f5db5fbd18c2987cba06d51b0529f52bc3a6cdc33d3f4eab725104bd", size = 40102, upload-time = "2025-10-06T05:36:26.333Z" }, + { url = "https://files.pythonhosted.org/packages/2d/40/0832c31a37d60f60ed79e9dfb5a92e1e2af4f40a16a29abcc7992af9edff/frozenlist-1.8.0-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:8d92f1a84bb12d9e56f818b3a746f3efba93c1b63c8387a73dde655e1e42282a", size = 85717, upload-time = "2025-10-06T05:36:27.341Z" }, + { url = "https://files.pythonhosted.org/packages/30/ba/b0b3de23f40bc55a7057bd38434e25c34fa48e17f20ee273bbde5e0650f3/frozenlist-1.8.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:96153e77a591c8adc2ee805756c61f59fef4cf4073a9275ee86fe8cba41241f7", size = 49651, upload-time = "2025-10-06T05:36:28.855Z" }, + { url = "https://files.pythonhosted.org/packages/0c/ab/6e5080ee374f875296c4243c381bbdef97a9ac39c6e3ce1d5f7d42cb78d6/frozenlist-1.8.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:f21f00a91358803399890ab167098c131ec2ddd5f8f5fd5fe9c9f2c6fcd91e40", size = 49417, upload-time = "2025-10-06T05:36:29.877Z" }, + { url = "https://files.pythonhosted.org/packages/d5/4e/e4691508f9477ce67da2015d8c00acd751e6287739123113a9fca6f1604e/frozenlist-1.8.0-cp313-cp313-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:fb30f9626572a76dfe4293c7194a09fb1fe93ba94c7d4f720dfae3b646b45027", size = 234391, upload-time = "2025-10-06T05:36:31.301Z" }, + { url = "https://files.pythonhosted.org/packages/40/76/c202df58e3acdf12969a7895fd6f3bc016c642e6726aa63bd3025e0fc71c/frozenlist-1.8.0-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:eaa352d7047a31d87dafcacbabe89df0aa506abb5b1b85a2fb91bc3faa02d822", size = 233048, upload-time = "2025-10-06T05:36:32.531Z" }, + { url = "https://files.pythonhosted.org/packages/f9/c0/8746afb90f17b73ca5979c7a3958116e105ff796e718575175319b5bb4ce/frozenlist-1.8.0-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:03ae967b4e297f58f8c774c7eabcce57fe3c2434817d4385c50661845a058121", size = 226549, upload-time = "2025-10-06T05:36:33.706Z" }, + { url = "https://files.pythonhosted.org/packages/7e/eb/4c7eefc718ff72f9b6c4893291abaae5fbc0c82226a32dcd8ef4f7a5dbef/frozenlist-1.8.0-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:f6292f1de555ffcc675941d65fffffb0a5bcd992905015f85d0592201793e0e5", size = 239833, upload-time = "2025-10-06T05:36:34.947Z" }, + { url = "https://files.pythonhosted.org/packages/c2/4e/e5c02187cf704224f8b21bee886f3d713ca379535f16893233b9d672ea71/frozenlist-1.8.0-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:29548f9b5b5e3460ce7378144c3010363d8035cea44bc0bf02d57f5a685e084e", size = 245363, upload-time = "2025-10-06T05:36:36.534Z" }, + { url = "https://files.pythonhosted.org/packages/1f/96/cb85ec608464472e82ad37a17f844889c36100eed57bea094518bf270692/frozenlist-1.8.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:ec3cc8c5d4084591b4237c0a272cc4f50a5b03396a47d9caaf76f5d7b38a4f11", size = 229314, upload-time = "2025-10-06T05:36:38.582Z" }, + { url = "https://files.pythonhosted.org/packages/5d/6f/4ae69c550e4cee66b57887daeebe006fe985917c01d0fff9caab9883f6d0/frozenlist-1.8.0-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:517279f58009d0b1f2e7c1b130b377a349405da3f7621ed6bfae50b10adf20c1", size = 243365, upload-time = "2025-10-06T05:36:40.152Z" }, + { url = "https://files.pythonhosted.org/packages/7a/58/afd56de246cf11780a40a2c28dc7cbabbf06337cc8ddb1c780a2d97e88d8/frozenlist-1.8.0-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:db1e72ede2d0d7ccb213f218df6a078a9c09a7de257c2fe8fcef16d5925230b1", size = 237763, upload-time = "2025-10-06T05:36:41.355Z" }, + { url = "https://files.pythonhosted.org/packages/cb/36/cdfaf6ed42e2644740d4a10452d8e97fa1c062e2a8006e4b09f1b5fd7d63/frozenlist-1.8.0-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:b4dec9482a65c54a5044486847b8a66bf10c9cb4926d42927ec4e8fd5db7fed8", size = 240110, upload-time = "2025-10-06T05:36:42.716Z" }, + { url = "https://files.pythonhosted.org/packages/03/a8/9ea226fbefad669f11b52e864c55f0bd57d3c8d7eb07e9f2e9a0b39502e1/frozenlist-1.8.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:21900c48ae04d13d416f0e1e0c4d81f7931f73a9dfa0b7a8746fb2fe7dd970ed", size = 233717, upload-time = "2025-10-06T05:36:44.251Z" }, + { url = "https://files.pythonhosted.org/packages/1e/0b/1b5531611e83ba7d13ccc9988967ea1b51186af64c42b7a7af465dcc9568/frozenlist-1.8.0-cp313-cp313-win32.whl", hash = "sha256:8b7b94a067d1c504ee0b16def57ad5738701e4ba10cec90529f13fa03c833496", size = 39628, upload-time = "2025-10-06T05:36:45.423Z" }, + { url = "https://files.pythonhosted.org/packages/d8/cf/174c91dbc9cc49bc7b7aab74d8b734e974d1faa8f191c74af9b7e80848e6/frozenlist-1.8.0-cp313-cp313-win_amd64.whl", hash = "sha256:878be833caa6a3821caf85eb39c5ba92d28e85df26d57afb06b35b2efd937231", size = 43882, upload-time = "2025-10-06T05:36:46.796Z" }, + { url = "https://files.pythonhosted.org/packages/c1/17/502cd212cbfa96eb1388614fe39a3fc9ab87dbbe042b66f97acb57474834/frozenlist-1.8.0-cp313-cp313-win_arm64.whl", hash = "sha256:44389d135b3ff43ba8cc89ff7f51f5a0bb6b63d829c8300f79a2fe4fe61bcc62", size = 39676, upload-time = "2025-10-06T05:36:47.8Z" }, + { url = "https://files.pythonhosted.org/packages/d2/5c/3bbfaa920dfab09e76946a5d2833a7cbdf7b9b4a91c714666ac4855b88b4/frozenlist-1.8.0-cp313-cp313t-macosx_10_13_universal2.whl", hash = "sha256:e25ac20a2ef37e91c1b39938b591457666a0fa835c7783c3a8f33ea42870db94", size = 89235, upload-time = "2025-10-06T05:36:48.78Z" }, + { url = "https://files.pythonhosted.org/packages/d2/d6/f03961ef72166cec1687e84e8925838442b615bd0b8854b54923ce5b7b8a/frozenlist-1.8.0-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:07cdca25a91a4386d2e76ad992916a85038a9b97561bf7a3fd12d5d9ce31870c", size = 50742, upload-time = "2025-10-06T05:36:49.837Z" }, + { url = "https://files.pythonhosted.org/packages/1e/bb/a6d12b7ba4c3337667d0e421f7181c82dda448ce4e7ad7ecd249a16fa806/frozenlist-1.8.0-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:4e0c11f2cc6717e0a741f84a527c52616140741cd812a50422f83dc31749fb52", size = 51725, upload-time = "2025-10-06T05:36:50.851Z" }, + { url = "https://files.pythonhosted.org/packages/bc/71/d1fed0ffe2c2ccd70b43714c6cab0f4188f09f8a67a7914a6b46ee30f274/frozenlist-1.8.0-cp313-cp313t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:b3210649ee28062ea6099cfda39e147fa1bc039583c8ee4481cb7811e2448c51", size = 284533, upload-time = "2025-10-06T05:36:51.898Z" }, + { url = "https://files.pythonhosted.org/packages/c9/1f/fb1685a7b009d89f9bf78a42d94461bc06581f6e718c39344754a5d9bada/frozenlist-1.8.0-cp313-cp313t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:581ef5194c48035a7de2aefc72ac6539823bb71508189e5de01d60c9dcd5fa65", size = 292506, upload-time = "2025-10-06T05:36:53.101Z" }, + { url = "https://files.pythonhosted.org/packages/e6/3b/b991fe1612703f7e0d05c0cf734c1b77aaf7c7d321df4572e8d36e7048c8/frozenlist-1.8.0-cp313-cp313t-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:3ef2d026f16a2b1866e1d86fc4e1291e1ed8a387b2c333809419a2f8b3a77b82", size = 274161, upload-time = "2025-10-06T05:36:54.309Z" }, + { url = "https://files.pythonhosted.org/packages/ca/ec/c5c618767bcdf66e88945ec0157d7f6c4a1322f1473392319b7a2501ded7/frozenlist-1.8.0-cp313-cp313t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:5500ef82073f599ac84d888e3a8c1f77ac831183244bfd7f11eaa0289fb30714", size = 294676, upload-time = "2025-10-06T05:36:55.566Z" }, + { url = "https://files.pythonhosted.org/packages/7c/ce/3934758637d8f8a88d11f0585d6495ef54b2044ed6ec84492a91fa3b27aa/frozenlist-1.8.0-cp313-cp313t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:50066c3997d0091c411a66e710f4e11752251e6d2d73d70d8d5d4c76442a199d", size = 300638, upload-time = "2025-10-06T05:36:56.758Z" }, + { url = "https://files.pythonhosted.org/packages/fc/4f/a7e4d0d467298f42de4b41cbc7ddaf19d3cfeabaf9ff97c20c6c7ee409f9/frozenlist-1.8.0-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:5c1c8e78426e59b3f8005e9b19f6ff46e5845895adbde20ece9218319eca6506", size = 283067, upload-time = "2025-10-06T05:36:57.965Z" }, + { url = "https://files.pythonhosted.org/packages/dc/48/c7b163063d55a83772b268e6d1affb960771b0e203b632cfe09522d67ea5/frozenlist-1.8.0-cp313-cp313t-musllinux_1_2_armv7l.whl", hash = "sha256:eefdba20de0d938cec6a89bd4d70f346a03108a19b9df4248d3cf0d88f1b0f51", size = 292101, upload-time = "2025-10-06T05:36:59.237Z" }, + { url = "https://files.pythonhosted.org/packages/9f/d0/2366d3c4ecdc2fd391e0afa6e11500bfba0ea772764d631bbf82f0136c9d/frozenlist-1.8.0-cp313-cp313t-musllinux_1_2_ppc64le.whl", hash = "sha256:cf253e0e1c3ceb4aaff6df637ce033ff6535fb8c70a764a8f46aafd3d6ab798e", size = 289901, upload-time = "2025-10-06T05:37:00.811Z" }, + { url = "https://files.pythonhosted.org/packages/b8/94/daff920e82c1b70e3618a2ac39fbc01ae3e2ff6124e80739ce5d71c9b920/frozenlist-1.8.0-cp313-cp313t-musllinux_1_2_s390x.whl", hash = "sha256:032efa2674356903cd0261c4317a561a6850f3ac864a63fc1583147fb05a79b0", size = 289395, upload-time = "2025-10-06T05:37:02.115Z" }, + { url = "https://files.pythonhosted.org/packages/e3/20/bba307ab4235a09fdcd3cc5508dbabd17c4634a1af4b96e0f69bfe551ebd/frozenlist-1.8.0-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:6da155091429aeba16851ecb10a9104a108bcd32f6c1642867eadaee401c1c41", size = 283659, upload-time = "2025-10-06T05:37:03.711Z" }, + { url = "https://files.pythonhosted.org/packages/fd/00/04ca1c3a7a124b6de4f8a9a17cc2fcad138b4608e7a3fc5877804b8715d7/frozenlist-1.8.0-cp313-cp313t-win32.whl", hash = "sha256:0f96534f8bfebc1a394209427d0f8a63d343c9779cda6fc25e8e121b5fd8555b", size = 43492, upload-time = "2025-10-06T05:37:04.915Z" }, + { url = "https://files.pythonhosted.org/packages/59/5e/c69f733a86a94ab10f68e496dc6b7e8bc078ebb415281d5698313e3af3a1/frozenlist-1.8.0-cp313-cp313t-win_amd64.whl", hash = "sha256:5d63a068f978fc69421fb0e6eb91a9603187527c86b7cd3f534a5b77a592b888", size = 48034, upload-time = "2025-10-06T05:37:06.343Z" }, + { url = "https://files.pythonhosted.org/packages/16/6c/be9d79775d8abe79b05fa6d23da99ad6e7763a1d080fbae7290b286093fd/frozenlist-1.8.0-cp313-cp313t-win_arm64.whl", hash = "sha256:bf0a7e10b077bf5fb9380ad3ae8ce20ef919a6ad93b4552896419ac7e1d8e042", size = 41749, upload-time = "2025-10-06T05:37:07.431Z" }, + { url = "https://files.pythonhosted.org/packages/9a/9a/e35b4a917281c0b8419d4207f4334c8e8c5dbf4f3f5f9ada73958d937dcc/frozenlist-1.8.0-py3-none-any.whl", hash = "sha256:0c18a16eab41e82c295618a77502e17b195883241c563b00f0aa5106fc4eaa0d", size = 13409, upload-time = "2025-10-06T05:38:16.721Z" }, +] + [[package]] name = "google-re2" version = "1.1.20251105" @@ -498,6 +775,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/82/54/acc6a6e684827b0f6bb4e2c27f3d7e25b71322c4078ef5b455c07c43260e/grpcio_reflection-1.62.3-py3-none-any.whl", hash = "sha256:a48ef37df81a3bada78261fc92ef382f061112f989d1312398b945cc69838b9c", size = 22232, upload-time = "2024-08-06T00:30:13.131Z" }, ] +[[package]] +name = "idna" +version = "3.20" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/f5/08/8eea9d4b8302028f3abb2c0813953f7aec26d33b7a8960ed760e65ff29fa/idna-3.20.tar.gz", hash = "sha256:a7db850025b95ded1eae8a46181a1a6c56c92c96f0e2b005d9ff8dc0210cab44", size = 216463, upload-time = "2026-09-17T14:11:04.752Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/58/a2/bb081bab032533a855d44de1d56f8e8426114ff1ba5d1f07a438a0a654f8/idna-3.20-py3-none-any.whl", hash = "sha256:ab7ae7122974553370f0bdb919e1a960b2cd1bc1ef0276416d896db81c14582c", size = 69583, upload-time = "2026-09-17T14:11:03.168Z" }, +] + [[package]] name = "iniconfig" version = "2.3.0" @@ -516,6 +802,27 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/14/2f/967ba146e6d58cf6a652da73885f52fc68001525b4197effc174321d70b4/jmespath-1.1.0-py3-none-any.whl", hash = "sha256:a5663118de4908c91729bea0acadca56526eb2698e83de10cd116ae0f4e97c64", size = 20419, upload-time = "2026-01-22T16:35:24.919Z" }, ] +[[package]] +name = "kubernetes" +version = "36.0.3" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "aiohttp" }, + { name = "certifi" }, + { name = "durationpy" }, + { name = "python-dateutil" }, + { name = "pyyaml" }, + { name = "requests" }, + { name = "requests-oauthlib" }, + { name = "six" }, + { name = "urllib3" }, + { name = "websocket-client" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/ca/57/b07b96353f902aa1bdbe00e878e3a12a137977d03a962479785576aa8ec9/kubernetes-36.0.3.tar.gz", hash = "sha256:36993ed25ce59b789c9341473a228fcf268504a2fec7c2b2b1531d73072e5ce7", size = 2337528, upload-time = "2026-07-13T20:38:12.128Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/5b/30/a96d47df739689ac0001ade0afefc16e3b477fc2fb426b568515fdc8afce/kubernetes-36.0.3-py2.py3-none-any.whl", hash = "sha256:8fde9241c4b298e6374a069dcf728359b4e72c2fb29489a975ba4e1c047cf10f", size = 4618066, upload-time = "2026-07-13T20:38:10.172Z" }, +] + [[package]] name = "lark" version = "1.3.1" @@ -533,6 +840,7 @@ source = { virtual = "." } [package.dev-dependencies] dev = [ { name = "crossplane-function-sdk-python" }, + { name = "kubernetes" }, { name = "pydantic" }, { name = "pytest" }, { name = "pyyaml" }, @@ -544,12 +852,94 @@ dev = [ [package.metadata.requires-dev] dev = [ { name = "crossplane-function-sdk-python", specifier = ">=0.14.0" }, + { name = "kubernetes", specifier = ">=36.0" }, { name = "pydantic", specifier = ">=2.0" }, { name = "pytest", specifier = ">=9.0" }, { name = "pyyaml", specifier = ">=6.0" }, { name = "types-protobuf", specifier = ">=4.24" }, ] +[[package]] +name = "multidict" +version = "6.9.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/d6/99/1d4d69c3512d0ddbfa3a1b69cfd9a151012ab2eb4eabbb096201b1f0b7d8/multidict-6.9.1.tar.gz", hash = "sha256:0f06e60fa190aa7abd0914c2a766736fdc8e9f34878c4346338534b73d1b20e2", size = 182404, upload-time = "2026-09-21T17:59:05.362Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e3/24/3823efc330630a1f132bcd0a9b182ddec3c53f452f551c8e8d3aaf5a2d20/multidict-6.9.1-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:910d4260512660484c0dc1588a316fbb35a40c081c36fc51d1225351af17cfe4", size = 96739, upload-time = "2026-09-21T17:54:42.815Z" }, + { url = "https://files.pythonhosted.org/packages/6a/69/36331fe1d3aeb0c525a9972d6cf82791075a395aa96f026f060009d691ae/multidict-6.9.1-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:33fa55b990f81c2927e01399ace0d18926c69d69baa8cdaa819424132fb97987", size = 58514, upload-time = "2026-09-21T17:54:44.064Z" }, + { url = "https://files.pythonhosted.org/packages/8c/ca/5860651f782078ef13f2761c2593f4ba066d76f626b2f30f23a7f838b775/multidict-6.9.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:369b5aa01b241cd3fea6890bdbb11a1425d87bf1515831500d518f4223e9d72c", size = 57303, upload-time = "2026-09-21T17:54:45.304Z" }, + { url = "https://files.pythonhosted.org/packages/c7/9f/c56e2fa223cb5d182660e1a09eec38880f96b7cd22769582f74447c60418/multidict-6.9.1-cp311-cp311-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:803f8b575a71b1b299d677c28db0653459c79b5308874efec813f17b7457c7f1", size = 320076, upload-time = "2026-09-21T17:54:46.826Z" }, + { url = "https://files.pythonhosted.org/packages/59/97/eda4fe0cad51f96096363ed32d3e0f8df28dab00beff4d92adc9db517726/multidict-6.9.1-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:b8a65621b98984a62e59403009591b8a5a7736273aefe1cab64cfb85b365cc07", size = 315489, upload-time = "2026-09-21T17:54:48.25Z" }, + { url = "https://files.pythonhosted.org/packages/24/14/10b1ded0a4085a51c3054125f854da4f47ce4fec724a2d2503231b83f3db/multidict-6.9.1-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:13849a1d4f54c3809ae721e9e83ab28f5ea602f33660cb84eb6ef261eac706c1", size = 296716, upload-time = "2026-09-21T17:54:49.749Z" }, + { url = "https://files.pythonhosted.org/packages/dc/14/04d155dd18528443cb9009f3a9f35f7417773797d072c95104ff17b8b9b7/multidict-6.9.1-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:dd9a137a4a9becda3094f3831cd026380f75f6855e051eefe4c73ade524f1cc3", size = 330244, upload-time = "2026-09-21T17:54:51.26Z" }, + { url = "https://files.pythonhosted.org/packages/8b/90/284d10b3a9e5b312a8ec5b7a175fc7d560eefdc99886320449891a197992/multidict-6.9.1-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:9ae9614c317c50836689ce2dfde07c05fe0b16378562e2221746c3913ede3c80", size = 331288, upload-time = "2026-09-21T17:54:52.595Z" }, + { url = "https://files.pythonhosted.org/packages/27/50/f420de3683f9b047fc5588064d84043ef57a7152ef451b8373dfc7068d22/multidict-6.9.1-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:e8f1e362c9352b50ed120f001046fdbb80810c9d56580f4c3fc13bbe30823387", size = 320663, upload-time = "2026-09-21T17:54:53.91Z" }, + { url = "https://files.pythonhosted.org/packages/01/ab/0120a650d7ce4fe167299c13a13cc4ee2dc72458a9c0e9e9330b824b73c9/multidict-6.9.1-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:0dd655518f136febd96c05131a76a863e32fc2a1d7acd4e3c959e3ceb77d8345", size = 291139, upload-time = "2026-09-21T17:54:55.313Z" }, + { url = "https://files.pythonhosted.org/packages/b3/3f/c66342f73ee5cd2c9b1dda6cfce89b5f348c48921b8c8b6e59aafbbfe31a/multidict-6.9.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:2c5d675da8f1cb5650271c8ad5e95c0a3e5a183c105e72d953b12877b1c8d0fd", size = 310044, upload-time = "2026-09-21T17:54:56.745Z" }, + { url = "https://files.pythonhosted.org/packages/05/0d/e44b90d9e77e44ec3473c4c7ac7ea95764d7e8fbf6d5a085ef9bff7ce0c2/multidict-6.9.1-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:658f90f49cf5af2441cad0a2b801c3ef520471989a1ec55bcb25b255b2ca8d2f", size = 303279, upload-time = "2026-09-21T17:54:58.178Z" }, + { url = "https://files.pythonhosted.org/packages/97/8a/7741f7c23c211fae31ab546f0ba41569dae7671cf7ca30d5afb59e9d92aa/multidict-6.9.1-cp311-cp311-musllinux_1_2_i686.whl", hash = "sha256:43124fe172ada86d03ac3c8dc8179091341f6724d5e5d5b160e1587e4cd3761b", size = 326177, upload-time = "2026-09-21T17:54:59.825Z" }, + { url = "https://files.pythonhosted.org/packages/e8/63/0cdbfbbe36316b2ef4b8e011afaca2cf291a73fe4fabf34dd0b22e3ee48a/multidict-6.9.1-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:b2f0adc22a4eb31e545221d93fc73a0f6a8cc2379d0f4309f71d1d17ba938b82", size = 323526, upload-time = "2026-09-21T17:55:01.43Z" }, + { url = "https://files.pythonhosted.org/packages/98/05/0c86e9678e78f4596df0ed7f59d2aa755107da3f2a92f1b7fa150e94cbb8/multidict-6.9.1-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:7fc59e9ba821b220944ccfe0f89c9dc4745f6d097569992356eb03869f21e953", size = 290278, upload-time = "2026-09-21T17:55:02.993Z" }, + { url = "https://files.pythonhosted.org/packages/95/8e/71bd7f43c4883cba6c5f96db88bd2311d52e74ad7680be5915938490d56f/multidict-6.9.1-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:87cc632c88ee5dc80e12681047839304d98ee5c9a708d686505767001c9b8b9a", size = 321142, upload-time = "2026-09-21T17:55:04.48Z" }, + { url = "https://files.pythonhosted.org/packages/2d/00/bf59d6bb22a4c152e4039574c29490caf3c65107715121807b4bd7cde244/multidict-6.9.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:b828cf64d62dc09ac183f03c1aeedd164ade96a2ce4934109452edf29de1dd13", size = 319045, upload-time = "2026-09-21T17:55:06.119Z" }, + { url = "https://files.pythonhosted.org/packages/3a/3c/f27f4f045198bf1c4bcd6eb3b6cf06be2dceaceb6e67222b903fce7a6749/multidict-6.9.1-cp311-cp311-win32.whl", hash = "sha256:2c1aeb92eea59d824f004341b26d5e4b47a8a441cf9726769b9a90abf9d0e08f", size = 51309, upload-time = "2026-09-21T17:55:07.531Z" }, + { url = "https://files.pythonhosted.org/packages/91/17/c2064aef64efc007bc89a571f843e5772f27946eb9bc018731bf8016e319/multidict-6.9.1-cp311-cp311-win_amd64.whl", hash = "sha256:5f89dad732280e7a10b74d40b91364f88e13c3f2c08c2ef83a8cd42f7a61af2e", size = 59148, upload-time = "2026-09-21T17:55:08.936Z" }, + { url = "https://files.pythonhosted.org/packages/45/a1/3bab1827edd813dfb90712af5cbb7fc1473ed4d9c871d103df4f4bb950c2/multidict-6.9.1-cp311-cp311-win_arm64.whl", hash = "sha256:5800368526647146978389dfaa46da3356291e9f0fff9a4ef12e8c2bef964a0d", size = 54258, upload-time = "2026-09-21T17:55:10.191Z" }, + { url = "https://files.pythonhosted.org/packages/d9/0d/4b5afb6d3e545c9af0cdfe2db8f6f4c6664568c23d863d888674e447e6a4/multidict-6.9.1-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:29138fef49828542e859828107e42e50d0e587c513b7eb4b2d92bade2b0860fe", size = 98360, upload-time = "2026-09-21T17:55:11.508Z" }, + { url = "https://files.pythonhosted.org/packages/89/ab/1b9ca66251899981b21138b87da9d5a9c2c81af12b1ea7d19466972f7fe2/multidict-6.9.1-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:19e31815d41cefc365489e591d105d2baceb2f65aa75d29471fbdbda8651e006", size = 59979, upload-time = "2026-09-21T17:55:12.866Z" }, + { url = "https://files.pythonhosted.org/packages/36/eb/6ae44062466c26c8469ef43f2481a6a48d8cea0587b2d54514ec92e2adfd/multidict-6.9.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:6ed30be8918e18c8bed0a2e8b70639ecf02feb61ed00ca2e41cfcb2a50fa3f42", size = 57736, upload-time = "2026-09-21T17:55:14.223Z" }, + { url = "https://files.pythonhosted.org/packages/b9/5c/a67817593019257a4ac8b0d1b4c426030e637c047b0692ef405439ecea7a/multidict-6.9.1-cp312-cp312-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:637f4ae36264bd7b8d9a60193acddc1d735ad52e8ed53a19931ea6d921fea8e5", size = 333097, upload-time = "2026-09-21T17:55:15.591Z" }, + { url = "https://files.pythonhosted.org/packages/90/bf/599ae2e6222822d88a247a8a7ae82fe6fd25d5700757b79603d5edafe6a0/multidict-6.9.1-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:35fc236507fb1b3138f0af5ecd5f94ed752d4d6d826248eae425f86204013eea", size = 334196, upload-time = "2026-09-21T17:55:17.015Z" }, + { url = "https://files.pythonhosted.org/packages/15/10/d8aac5acacbe7f5c117866c776ec26d5f15a868759b6f37ad8e7ed3b5b02/multidict-6.9.1-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:e58952f04772f59f11c6e007471449809a30165188669bca8fdb19dde40a8f24", size = 317465, upload-time = "2026-09-21T17:55:18.412Z" }, + { url = "https://files.pythonhosted.org/packages/19/0a/714f796f7293a8b1c5c3f465a26996231d5450c54e85d64ef1258c091134/multidict-6.9.1-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:d35a4f1c63f07fbb8c8f9946dea98b21eddf6c57421585f71d91864be3ba2a24", size = 345914, upload-time = "2026-09-21T17:55:20.306Z" }, + { url = "https://files.pythonhosted.org/packages/da/b1/e37fbf769c567be277bcf32df6234035a4384677fcc1bd852752be3d6b93/multidict-6.9.1-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:b5f8771aaaed7f80e84a4e471d2f29ab6721e4595075e54d03ae1ed951b2000a", size = 351163, upload-time = "2026-09-21T17:55:22.21Z" }, + { url = "https://files.pythonhosted.org/packages/ed/5b/db08419c1e1f7c9d60cfd2787b2b517d7ae4ebbda8281b48a33eb5141467/multidict-6.9.1-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:976fd7689d69ec78d67d31d38d396d8adb562f7e8368279f76aed4aa451fa06d", size = 336881, upload-time = "2026-09-21T17:55:23.699Z" }, + { url = "https://files.pythonhosted.org/packages/9b/07/cc9bc8a62651d2d53ab93ee4993b3a71b7cb78eb8ebfc9c757a5b6698617/multidict-6.9.1-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:95052e8777a86bae87c0bd0b5ab22d809e3d1d02bf69e3e66ddda5ba75a05805", size = 303405, upload-time = "2026-09-21T17:55:25.305Z" }, + { url = "https://files.pythonhosted.org/packages/b6/0c/8e912afafa70e944dbb8bec4b66ca6e008511395278c0d3dd0e89567536a/multidict-6.9.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:1a8adfcaf96f587ab138476eaddef95f29b8a2a8a9afbfea8d2fd62180995d02", size = 324265, upload-time = "2026-09-21T17:55:26.989Z" }, + { url = "https://files.pythonhosted.org/packages/66/6a/62c2af80fb085e6234805017af857b8e913dfac7a55e9c1349c27768c58a/multidict-6.9.1-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:1a53de2772cfb74559df2eb4456ec4eeb908435ec55a84b69370d9d745d62aa8", size = 322085, upload-time = "2026-09-21T17:55:28.732Z" }, + { url = "https://files.pythonhosted.org/packages/f8/6b/35bf801b336fd960811207203ffdcdc24acc3558b3e7ff2e1c914b141e70/multidict-6.9.1-cp312-cp312-musllinux_1_2_i686.whl", hash = "sha256:0f2ce963299d42fa3f22a90adc0fdf174792ffef5ff4c7ffb68260548fb05580", size = 338107, upload-time = "2026-09-21T17:55:30.303Z" }, + { url = "https://files.pythonhosted.org/packages/80/41/495ef65bf5bba29d142d81b3fe1b8154b919bed70e96491b1e93c3c26f0f/multidict-6.9.1-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:5b30ddf7234e611ca877575b62840e6af5977f92f1f9d532eedbb05a44ff8004", size = 339631, upload-time = "2026-09-21T17:55:32.012Z" }, + { url = "https://files.pythonhosted.org/packages/19/bd/057fdff5f4e04dcd40a960e38f77d19d3c4b67dd243ffa5f43718edb29fc/multidict-6.9.1-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:63ada7ee2e9345695f9e9bc4c65d72222253f07b1ac94fd0e37555cc6f3c7f60", size = 302080, upload-time = "2026-09-21T17:55:33.817Z" }, + { url = "https://files.pythonhosted.org/packages/3f/de/9ace933ee8dad808632523726f42255b09600087219e3d4ead7369820910/multidict-6.9.1-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:3c95601ed98fad3f6e2f8fe809c3b526b0fab31ef525e00a155e227f3d17f58a", size = 341107, upload-time = "2026-09-21T17:55:35.471Z" }, + { url = "https://files.pythonhosted.org/packages/d8/ac/7c1204406097bfc5c283d4a3287d61166807d189917567df4c1318484fc4/multidict-6.9.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:c148e8b596000dd3e4bfe206e70f3e666be18d72032e0012555f2373c52e35d6", size = 333482, upload-time = "2026-09-21T17:55:37.002Z" }, + { url = "https://files.pythonhosted.org/packages/ff/c0/a70c32ea3299ebe00f44533740cb46905717c51faf29bbd3ce8bb5886d9d/multidict-6.9.1-cp312-cp312-win32.whl", hash = "sha256:f9dad513626a33670f17cddc6078e30e311f444c957e8dbfc5b2b4603c8b4edb", size = 52548, upload-time = "2026-09-21T17:55:38.529Z" }, + { url = "https://files.pythonhosted.org/packages/d7/2e/8c9c2591df01e5692ad1bf92febb082a17b1c6c3a19aa8b7ff49988fd989/multidict-6.9.1-cp312-cp312-win_amd64.whl", hash = "sha256:a16a1dc8529f9e734a41c3b856f3eae7ebacdc061dde3f8a844e0c7889c97203", size = 59622, upload-time = "2026-09-21T17:55:39.86Z" }, + { url = "https://files.pythonhosted.org/packages/d0/95/59f4472ec512bc180fd207899594c7803ece96262ff9aaa3a4b6d7da6940/multidict-6.9.1-cp312-cp312-win_arm64.whl", hash = "sha256:361f7206cf341ba94fb015688f5c8b480f8e63bd58a4c14a48aeca7851a241cc", size = 54930, upload-time = "2026-09-21T17:55:41.155Z" }, + { url = "https://files.pythonhosted.org/packages/eb/45/ddf7c76860f5f553a23ad3ca38463ccf5247081618977742c8dc4625355e/multidict-6.9.1-cp313-cp313-android_24_x86_64.whl", hash = "sha256:d7bf9e43282d69561618e8a0ea33368d532ebef42f15c096f427090521dd74f3", size = 63266, upload-time = "2026-09-21T17:55:42.676Z" }, + { url = "https://files.pythonhosted.org/packages/6d/d3/f4ae5945ea2de597eaeb4eca3a74c59783ace54482c9b49435442b63efa8/multidict-6.9.1-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:03d47df72f084f757c1cb771188d5f4e3a805e4abc4d67e32509272343ae9382", size = 55771, upload-time = "2026-09-21T17:55:44.016Z" }, + { url = "https://files.pythonhosted.org/packages/93/7d/15468239920040d01c686e5ae669e6382f7bf31fc3e5e0a6b8f1c3b32d3b/multidict-6.9.1-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:6bc94fe17c3c56e5418f79515b786b101845f70609b0d19d0c1ba13448e5633a", size = 57116, upload-time = "2026-09-21T17:55:45.315Z" }, + { url = "https://files.pythonhosted.org/packages/17/1b/b958f06aac2d8b1e485eb1c105b1b159cbf3868249e5142142451577dad2/multidict-6.9.1-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:8f2973bbd2bebd9d2e0cd6394c1292a1a19ccd56bdcbe1e174059f1a39be5b40", size = 97692, upload-time = "2026-09-21T17:55:46.674Z" }, + { url = "https://files.pythonhosted.org/packages/93/6c/d6cfe18e61010166d7237d7527c775eb9843e7078feadea45b7628751b60/multidict-6.9.1-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:de7738b8c0bb74c4cc16bbd7fb49fc2bcf6430dba11b3432cd52768ae40933e8", size = 59556, upload-time = "2026-09-21T17:55:48.22Z" }, + { url = "https://files.pythonhosted.org/packages/46/46/9b4c1127cece0289fb207d04aae5116382ddfcb2a4dc8e0cb49a33f3c7b3/multidict-6.9.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:e5ccad4b7bac125722f48d6f862bed3b514d8526deea06316bb72f152cd30a7c", size = 57357, upload-time = "2026-09-21T17:55:49.998Z" }, + { url = "https://files.pythonhosted.org/packages/bc/fc/c18b07100a6064573e49e538d13d47eb2b5221f48080512df78aef54ab04/multidict-6.9.1-cp313-cp313-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", hash = "sha256:a49ff5cdb33654cb7d6a3c377aa2a83ddefaa1db31eb10bcf3c180aa84f9af8a", size = 330433, upload-time = "2026-09-21T17:55:51.435Z" }, + { url = "https://files.pythonhosted.org/packages/62/ff/52a0082adeb69656b634609d4fb0ae65456deaae4cca3d5a12656626dc50/multidict-6.9.1-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:b22ff30006a2f28f8bff878fb93413cbe3a4d1fd517c28081d848c90e9cfd2c8", size = 332487, upload-time = "2026-09-21T17:55:52.937Z" }, + { url = "https://files.pythonhosted.org/packages/39/3e/80bd4729635a7af371ff4dcc312594cf96e73f98854491be8740792de430/multidict-6.9.1-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:3adf06c66041aa21eeb8a71e82379b74773298c8e6d3d839b151aae441a99b94", size = 316911, upload-time = "2026-09-21T17:55:54.751Z" }, + { url = "https://files.pythonhosted.org/packages/15/f4/7ea4a907e053e924e2e1694d90560f79f72056504bfd5e8483695f9527d2/multidict-6.9.1-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:963a8d8f97057082679523d0fd4c53a38f86bc58cabe4556faef682ae53fa2fa", size = 344950, upload-time = "2026-09-21T17:55:56.5Z" }, + { url = "https://files.pythonhosted.org/packages/0d/16/9642ae41fbfbdc546aa481c2b07ae1bc087a94ab5f5020b4d9ab77da401f/multidict-6.9.1-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:b66ccc5c2cdd26e74fa5d4c29ffae424cc6148bf93ce574821783fb3b6d452c5", size = 350264, upload-time = "2026-09-21T17:55:58.073Z" }, + { url = "https://files.pythonhosted.org/packages/93/f2/e06c8e42074d0a4b8419bfe92afbd1d262190b69549ecbcd2dffa4f103c2/multidict-6.9.1-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:7fd79c521f6290c69125fa2b85fa65d9e657e6a8ffaf722dc881b926bef4aa5c", size = 336213, upload-time = "2026-09-21T17:55:59.723Z" }, + { url = "https://files.pythonhosted.org/packages/98/43/d7cf9ef4700e7c25244590d1b4f20cf12cbed86beba6c17be4d9e48c1099/multidict-6.9.1-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:1d804e4caf5d5da37d6dac1325da5629ebef1e27a294c2b568b295814aa36c7b", size = 302555, upload-time = "2026-09-21T17:56:01.666Z" }, + { url = "https://files.pythonhosted.org/packages/7d/3a/54209920324bc3928f5f2ba28b01a6dd957a9d48c3aaff77e38faaf5668d/multidict-6.9.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:bb0d664505f4b112f384cffeee82e91e3f6448d8e574989479db4439b68cba05", size = 322247, upload-time = "2026-09-21T17:56:03.265Z" }, + { url = "https://files.pythonhosted.org/packages/33/50/df96f961b178b621ab0adbf9221d4fd21534e1064fd9f5e85c1fa27fea55/multidict-6.9.1-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:9e14d17773b1b3c758ff153659a1824608a0cb562c45f484b5ed8a433428444a", size = 321787, upload-time = "2026-09-21T17:56:04.894Z" }, + { url = "https://files.pythonhosted.org/packages/a7/9b/37f354562a8f82f9c1f63d3a94300fc82126d50320175f0702bbd544f87c/multidict-6.9.1-cp313-cp313-musllinux_1_2_i686.whl", hash = "sha256:083735b7f395894e43adb278d5dae901448883a835ff8f1977e285fefdb10418", size = 335498, upload-time = "2026-09-21T17:56:06.783Z" }, + { url = "https://files.pythonhosted.org/packages/1e/44/78e366efc6c004185295cc9eb9711c10f13eadf3b547603b9c5526329976/multidict-6.9.1-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:b65121091567847a8cb520d364ab22ba90e00d3cc55fa9eb34bb439f0684bcd1", size = 338244, upload-time = "2026-09-21T17:56:09.162Z" }, + { url = "https://files.pythonhosted.org/packages/ad/2b/ab5bd3964691abe4d14b43bdbdb8a621b77cea2ca352dc93159ec0a4575b/multidict-6.9.1-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:ea999ae6e80e66ad5eea287860951b033d0104ca34d6d87c7b5125ebe0e12721", size = 301641, upload-time = "2026-09-21T17:56:11.227Z" }, + { url = "https://files.pythonhosted.org/packages/ea/a8/f6bc899f5aafaba755edc8e4bb934bc00b1e405f8d3fbc1dd10283738074/multidict-6.9.1-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:095d900c242e00fbe5f321ee072e7278b4153e78c5ce9c1efde167d62c1e4771", size = 340162, upload-time = "2026-09-21T17:56:13.076Z" }, + { url = "https://files.pythonhosted.org/packages/eb/03/a8fc809ef364b8c231c065ea2ea2f97ae86d4f9a0e36b5721759fd606778/multidict-6.9.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:6441cc837aea58be7d9baef1b2383eb8311ab9303f500f99ac90b584cd78bb14", size = 332826, upload-time = "2026-09-21T17:56:15.175Z" }, + { url = "https://files.pythonhosted.org/packages/f2/27/32dde80245e2024e9bb7a6ca3c2fb7c695dc74014cba557cc35be6f91e24/multidict-6.9.1-cp313-cp313-win32.whl", hash = "sha256:9c4880d017555d70dea367dd49271830842d48e3891c2da97da7ce8c4abcee40", size = 52378, upload-time = "2026-09-21T17:56:16.959Z" }, + { url = "https://files.pythonhosted.org/packages/be/78/1bac2987edac273a6b27fa50e9204bc3466c94a7b4f6a44828d730f8810f/multidict-6.9.1-cp313-cp313-win_amd64.whl", hash = "sha256:ac51cd64bae51c462ea58ad2492c9b8209667a4ef60c45c4a304518b67598d5d", size = 59449, upload-time = "2026-09-21T17:56:18.336Z" }, + { url = "https://files.pythonhosted.org/packages/e1/32/2a77ce19eea48cb9b3521a202b513ba3349aafa702c8317be0b645ffb74b/multidict-6.9.1-cp313-cp313-win_arm64.whl", hash = "sha256:37a9ebe00c698279213d56e6c64e1962ab1e092918270649b397cac3dc196ca4", size = 54713, upload-time = "2026-09-21T17:56:19.774Z" }, + { url = "https://files.pythonhosted.org/packages/be/59/e26cb779be4c591d1a910f59d29aca9fba4de70349840a833beba2652371/multidict-6.9.1-py3-none-any.whl", hash = "sha256:7bf6478188f4e47bf5686e8a33da4ae28bf43b1b2528d9ee144d28492bfac60b", size = 19176, upload-time = "2026-09-21T17:59:03.501Z" }, +] + +[[package]] +name = "oauthlib" +version = "4.0.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/7a/d8/a1bcc8ba112a627f8ffbdc212a78ce18d3ac07e91a5ca65d27918eee25a1/oauthlib-4.0.0.tar.gz", hash = "sha256:efb274799819440f95b4ab3b818869f1ce9ae26c5beacba0201d1a1b76b54f86", size = 187232, upload-time = "2026-09-28T06:01:18.77Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/d9/f4/78229a1066068ca14fc60fb26cf7381cabe4382261392b90e5f9552722d4/oauthlib-4.0.0-py3-none-any.whl", hash = "sha256:624c28c13a0a59cabf9747dfa52af63be3e512a7f2714df16e91b5b3a145e6cd", size = 159715, upload-time = "2026-09-28T06:01:17.008Z" }, +] + [[package]] name = "packaging" version = "26.3" @@ -618,6 +1008,66 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/54/20/4d324d65cc6d9205fabedc306948156824eb9f0ee1633355a8f7ec5c66bf/pluggy-1.6.0-py3-none-any.whl", hash = "sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746", size = 20538, upload-time = "2025-05-15T12:30:06.134Z" }, ] +[[package]] +name = "propcache" +version = "0.5.4" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/b3/9a/9fbf4e4ec0c2d7f1c32519fff782ef467859b8faa9fbc5331a96f6395d43/propcache-0.5.4.tar.gz", hash = "sha256:ff6b113f50bc066a698db5d944d2c6dc7507168dd3341e255a8892fd0715a558", size = 61545, upload-time = "2026-09-16T00:17:14.386Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/1e/40/14b21e505b7921617466576423f188a5c9caddfdaa1cf4b2b8a83d8fe216/propcache-0.5.4-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:897d1ddf6716e8f47200f7aad9a0efa6cc7586df66c6defa572f9eab379c078e", size = 86393, upload-time = "2026-09-16T00:14:06.9Z" }, + { url = "https://files.pythonhosted.org/packages/e7/4b/5a52e1a7b43563f7d408814194bb23cc8bf214eb6b86639b667a33a8d0d0/propcache-0.5.4-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:9cbfff4423eef4cc6cafc021469641a2b835f610b2647a6c5281903e21b8670d", size = 50431, upload-time = "2026-09-16T00:14:08.025Z" }, + { url = "https://files.pythonhosted.org/packages/05/cf/b5248180bf056cc76acc60c9c6e8c0ebbfdbd1c6cffd31fd14996927b7c8/propcache-0.5.4-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:3fc24f209c1b7f7f688b66b98293954f5504279760999b58920ee12dd8471c1d", size = 52116, upload-time = "2026-09-16T00:14:09.114Z" }, + { url = "https://files.pythonhosted.org/packages/86/a8/7c6cd6bfead1a11f2e411e688640e6d26574cb0bde7dcaa7423b0b65ed7a/propcache-0.5.4-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:62530ca89187827e4a4fe733f971abe81a7542eeea48ff61995f19b64d7199c8", size = 238729, upload-time = "2026-09-16T00:14:10.357Z" }, + { url = "https://files.pythonhosted.org/packages/5c/b4/442715b2e980df51be52d203549279e027728f24c80b00b5e525e31cd5ea/propcache-0.5.4-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:56fc3f7599528db40b1efa0889a620116e2704144495273d66066e8164e45838", size = 246121, upload-time = "2026-09-16T00:14:11.735Z" }, + { url = "https://files.pythonhosted.org/packages/bc/5d/df0684fc2b1732a01a7bec26d7897369022712422d25b09c37ce7dbc88a2/propcache-0.5.4-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:4f2d880ff60f45898f4acfa152aac8d04e3ee627d90ff4003491bf92239d5757", size = 251735, upload-time = "2026-09-16T00:14:13.053Z" }, + { url = "https://files.pythonhosted.org/packages/c7/06/519a5ebb48b6f94beb48396e55c905f12246a25c3a3608a7ec7bceabf50e/propcache-0.5.4-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:6e9368e87a3efc285e559131092c5db643eb8e56de4ee42064d5baec22ef2bb5", size = 235381, upload-time = "2026-09-16T00:14:14.398Z" }, + { url = "https://files.pythonhosted.org/packages/3c/07/1e0a9bb310830f2245edbd5cd3c6d24a783c053c4efd8e08e386e513c940/propcache-0.5.4-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:004e685b315646c410771836e72a44f143bbe624f29653a42687815069a303d5", size = 208973, upload-time = "2026-09-16T00:14:15.715Z" }, + { url = "https://files.pythonhosted.org/packages/62/5c/9324fab27d6088eecc47fe4332bf7aaf8c1ded93c36f558391e8a06d41a7/propcache-0.5.4-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:594eb4c6ec35e7179b058481f4e9f02521b56de16fa577c4b85c76fb1bf8a9f8", size = 233897, upload-time = "2026-09-16T00:14:17.25Z" }, + { url = "https://files.pythonhosted.org/packages/9c/a3/570d92fc952eae93b676f3a1568f4b89264102abd3c982ab6a9ebec58dcf/propcache-0.5.4-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:2dba2f02d2d5c09ef8a0e6c1a42aeaa451f4be9898cb00b04fe98717da2eb23b", size = 223512, upload-time = "2026-09-16T00:14:18.87Z" }, + { url = "https://files.pythonhosted.org/packages/65/47/26810d889d89bba31db397e6a88f8984af775f5ed6bad0a29dce84324cff/propcache-0.5.4-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:c3ef2818d63bc86071e9d2989ae75a1bc32b8f7059cfd9f5abbbee70c32e2ed6", size = 239043, upload-time = "2026-09-16T00:14:20.366Z" }, + { url = "https://files.pythonhosted.org/packages/89/2d/f9c47691aa024c8299a3afacd78d22a01ab57eb627b481b6089708e71017/propcache-0.5.4-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:dd2ac8f5b643454c2cc6b6118b13da16e88f4a6434fc3ba61aca384029f04f36", size = 208218, upload-time = "2026-09-16T00:14:21.801Z" }, + { url = "https://files.pythonhosted.org/packages/74/6b/d510c0c378cabbf9d0ac7b663af6d00f2e9074073d93b20c85f24aa5c071/propcache-0.5.4-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:4054acf80d40456a0537f2913b349718649d8d6458a14ab7f48d0ce28c30869d", size = 240301, upload-time = "2026-09-16T00:14:23.121Z" }, + { url = "https://files.pythonhosted.org/packages/3d/80/c80f6adaaa1e51f0db2dce8c9b3714d94ec45e358a21f9a1910b10b40a80/propcache-0.5.4-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:40e94adb1e7d39ff28a8bd8d8b8fbd1df6b9f40976dbe379134f1ce058e532dd", size = 230785, upload-time = "2026-09-16T00:14:24.458Z" }, + { url = "https://files.pythonhosted.org/packages/d9/e2/c32a7df3f39caa7f11b2eb37ea5b6960a6946f2bc7c4b8ff97bbdf6d6b6e/propcache-0.5.4-cp311-cp311-win32.whl", hash = "sha256:9f86f7259efe2c951f43e57d471c9b41daa5bfc7db9f67189059cf1ae6d77fd9", size = 42747, upload-time = "2026-09-16T00:14:25.715Z" }, + { url = "https://files.pythonhosted.org/packages/0a/8a/3db6a3543d8101263b4c52978b6276a04ead2caff2c5ab880d934f47bd89/propcache-0.5.4-cp311-cp311-win_amd64.whl", hash = "sha256:e904d4d01f36bd6e197590be1533c44e06058771e0746dd073a8ebb3ef880858", size = 46268, upload-time = "2026-09-16T00:14:26.996Z" }, + { url = "https://files.pythonhosted.org/packages/41/07/5222e2665bbf6e45847492ecbf3b9f3e4975a0ae300e5fd465df7d48ce55/propcache-0.5.4-cp311-cp311-win_arm64.whl", hash = "sha256:d42a9a856a4a6e2f6c10f1318c07e7daa498d6593abe745c71dae4521a26ca39", size = 43547, upload-time = "2026-09-16T00:14:28.143Z" }, + { url = "https://files.pythonhosted.org/packages/71/cd/348d58f142aebc4873345c6b31087629182ca6e0f2b3caeaa528cf882eba/propcache-0.5.4-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:b28f41fa3b8c6900457f858ec5b03998f3a6d535fbc1bb2edec5961ea05ec429", size = 87285, upload-time = "2026-09-16T00:14:29.362Z" }, + { url = "https://files.pythonhosted.org/packages/df/f4/f3ffaee281b276da854ac1d7a6a506d26cbc62ea2e623756f1d0a4a1ba1a/propcache-0.5.4-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:dcbf346a318a5e30063f547630b02bb787ce2f45b6368d5da143660b6a3835d8", size = 50984, upload-time = "2026-09-16T00:14:30.473Z" }, + { url = "https://files.pythonhosted.org/packages/25/88/1d7df7201750b37765ef2b23bc1c526c028dadde80afa0f57a118fc01182/propcache-0.5.4-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:87a3caecf8095e48dc72f84bfa42e23a848cf410cc9cc13031fba4869b706a21", size = 52460, upload-time = "2026-09-16T00:14:31.692Z" }, + { url = "https://files.pythonhosted.org/packages/83/4f/48865bd02a16ee5236bc46166b2946f37b93e07b0eae355dac0be0b216ca/propcache-0.5.4-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:60a64cbccaa11b7760ce705a14ada17ba459e7ca9f23ba587eb013821032d7ef", size = 251768, upload-time = "2026-09-16T00:14:32.908Z" }, + { url = "https://files.pythonhosted.org/packages/b0/19/3742a5eed62317b03b4002ee865dc9fd720308bdd0da1f29a5786c630311/propcache-0.5.4-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:a74bfa37147cc08fb29df10bd9c16f40fa7f860cd3a6d2fff853323a94f6e17f", size = 257723, upload-time = "2026-09-16T00:14:34.267Z" }, + { url = "https://files.pythonhosted.org/packages/cb/d5/ee6350fb0be9122bb6c67082a876d34b90d980d100c106af4b81023e04f4/propcache-0.5.4-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:a4d7a54719b67338a305dca2ce6aafe366817df94ddfd4b5514374356f5ca546", size = 265597, upload-time = "2026-09-16T00:14:35.56Z" }, + { url = "https://files.pythonhosted.org/packages/85/9f/83a07b6ec0e043c050cfdd35fb0cf1b7897b91d554d6eea293740309afe7/propcache-0.5.4-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:2814ecd8e818f487bee4b0f921bc4d1c176cc5fc71ac0f072d0fa67eda4ac14b", size = 250424, upload-time = "2026-09-16T00:14:36.894Z" }, + { url = "https://files.pythonhosted.org/packages/33/2c/a763a8251f50fba042af0fb1f02bfec4b31381e40aff760db2be7b2e1f84/propcache-0.5.4-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:6af4693716bfb03f1752ef1b30faa593db2c01d5272e9b8564a1549452a979ab", size = 216748, upload-time = "2026-09-16T00:14:38.369Z" }, + { url = "https://files.pythonhosted.org/packages/6a/e2/4d11bea8fd6a777149c6c20645f873952eab5de3a2497aa11648ec9ab6ab/propcache-0.5.4-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:4fbc1a15dc8cd1689508758d626b372b1f09d28d9577667feaf9e6bfcd8efcbc", size = 246533, upload-time = "2026-09-16T00:14:39.82Z" }, + { url = "https://files.pythonhosted.org/packages/9f/36/6683597de4907e70c717e3588c541202c66086a72ff3db58be49de66e72c/propcache-0.5.4-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:cdee8205a44d0be91bbac4c41b95d86641b72dfc7aef1279400e4fda3f26a937", size = 238173, upload-time = "2026-09-16T00:14:41.259Z" }, + { url = "https://files.pythonhosted.org/packages/85/84/cb08d79f1762daafeb2b030c470cd0c725c97b8ad67412457c6f35c53e9d/propcache-0.5.4-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:9a2a8a50a93dee0268a860a07fa3b4bd968f8ce4dbd794957da772f395368526", size = 251128, upload-time = "2026-09-16T00:14:42.652Z" }, + { url = "https://files.pythonhosted.org/packages/c2/0d/41b848036db6621370c1f2e5471a7da8149c730f8552a5257567721f4576/propcache-0.5.4-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:7ffafcbfc7b549ab940047e505c831eabac5e67de53e1bc174adbc5285c55944", size = 214821, upload-time = "2026-09-16T00:14:44.112Z" }, + { url = "https://files.pythonhosted.org/packages/f1/b7/adfae4bf9c63bccf12e2d9690a175c6579047a6eec3b5a6a5f51428c15e2/propcache-0.5.4-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:d1f5a500bfcbb2c0ab85e98a0dcd70f5899d34efe365a0187700369a79603031", size = 254793, upload-time = "2026-09-16T00:14:45.429Z" }, + { url = "https://files.pythonhosted.org/packages/51/6f/eeca9647245d5f92e87d53e5f14335bb42fce1a7e6842c8045b364eded8b/propcache-0.5.4-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:8a235f73d6e020855dc29dff012d920c02ee0feab8d73a24185a7569f4be1161", size = 247134, upload-time = "2026-09-16T00:14:46.976Z" }, + { url = "https://files.pythonhosted.org/packages/5d/a9/424e38838793d37160b4379c702f61c74c598fc6cd17204adbe3c554f7a8/propcache-0.5.4-cp312-cp312-win32.whl", hash = "sha256:b3083bfe87f95c756e610bd8025f26cbd1cd4aaa03a422f2d65efb7a97cd53d8", size = 43073, upload-time = "2026-09-16T00:14:48.338Z" }, + { url = "https://files.pythonhosted.org/packages/58/7b/6e8ef26f6d510a7916064fec68d55fcbfbdf7eb01e377480d66a122152d8/propcache-0.5.4-cp312-cp312-win_amd64.whl", hash = "sha256:98914de2c4d7f0f9f4a8c6ea4bf05841f4175796941e3ef7d47eb718f22311fb", size = 46190, upload-time = "2026-09-16T00:14:49.99Z" }, + { url = "https://files.pythonhosted.org/packages/08/b9/72028c5b56ced97f456de6aefa79435ca64d7f77af78ea8cf3c76fc5195f/propcache-0.5.4-cp312-cp312-win_arm64.whl", hash = "sha256:8876b39961e33d912afe3c1bee18ee564fdad0206f873cc15d522756b7f50737", size = 43075, upload-time = "2026-09-16T00:14:51.155Z" }, + { url = "https://files.pythonhosted.org/packages/78/4c/3b1365d58a667689e067e13d055fcd92bdf8d9a2fca3d9201b47ed5b3631/propcache-0.5.4-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:36c0d9db44b523ef93d03341b1c42d69ff01d673c053d1b1c6c3a363bcaa39ba", size = 85290, upload-time = "2026-09-16T00:14:52.342Z" }, + { url = "https://files.pythonhosted.org/packages/8f/61/5f9c29c3aa67c30238c4eadf95149b1d983a48f69b86b0cff927a7d6df13/propcache-0.5.4-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:e1d52a05dc417279f7e5c7618c5dfbbc29923aaf9bc0a5c1802ddcebf54c61a0", size = 50027, upload-time = "2026-09-16T00:14:53.67Z" }, + { url = "https://files.pythonhosted.org/packages/25/7d/c1ab1ef09e9d4d835be5d58c0a32a1e1de8397abaa4e502a9d4141328cad/propcache-0.5.4-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:44149f46500a0a41b95b4d99c2e586a77319539730607b9892974a092788b111", size = 51425, upload-time = "2026-09-16T00:14:54.826Z" }, + { url = "https://files.pythonhosted.org/packages/73/36/0093091ebb270fcd1bc1f6e095f93b2e0ed7f1011c28837dc2dbe5f96b99/propcache-0.5.4-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:dbab5f5ff6897c81f355d079010cdae85b02e5a0b518b5251523b8ad8ae9ac3c", size = 233595, upload-time = "2026-09-16T00:14:56.09Z" }, + { url = "https://files.pythonhosted.org/packages/ae/8f/0de9d4c8e05ce0be71b436919a216bd7fc5cc6e2691c0602295efb22b9ed/propcache-0.5.4-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:c3e98c55bde2bcf7db3c70d1aed7ae9aa8aebbf19a250c66645cde44cdb8b867", size = 240318, upload-time = "2026-09-16T00:14:57.674Z" }, + { url = "https://files.pythonhosted.org/packages/7d/71/2b35e91455209b85ee98f7859583e0814fab57d3af0f2381aaee34c37304/propcache-0.5.4-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:db3ae52ccc150dbc84704e9d642743897f3e1c54742ff34cacb661e52e3818a9", size = 246649, upload-time = "2026-09-16T00:14:59.352Z" }, + { url = "https://files.pythonhosted.org/packages/ed/74/08e6c1faf26ee2732023a3828787ba535557122774f4a386b1f715cbd8e0/propcache-0.5.4-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:f85915e00dcb1cd9f2f890ead064ed40a27df06f0db65be427b29482ae357572", size = 234316, upload-time = "2026-09-16T00:15:00.696Z" }, + { url = "https://files.pythonhosted.org/packages/5c/9a/08385733c9321c9bb78039d3ff31045e4fca962d9665023c4eb70f998819/propcache-0.5.4-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:c2ba30a89035b57b73e00475de948521602f543d79ce01db10b04b36c4c76fc8", size = 204666, upload-time = "2026-09-16T00:15:02.019Z" }, + { url = "https://files.pythonhosted.org/packages/1d/f4/e87bc7629af9a14a752b218764a78742d73c2c563ac58315da6841f0cbe4/propcache-0.5.4-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:ae58f361bd5dae942717c65d3413b478c70aea9c462599e7b9adad3731db3894", size = 225900, upload-time = "2026-09-16T00:15:03.394Z" }, + { url = "https://files.pythonhosted.org/packages/d9/6d/11014938d3fe9bea2ea2dcf930f26ed565bfb2f5be3c756362ea48c92636/propcache-0.5.4-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:96f7c5c15656040ddcbc51e56dc59b58aa25999d743c126abd425b9766ab43e9", size = 219988, upload-time = "2026-09-16T00:15:04.811Z" }, + { url = "https://files.pythonhosted.org/packages/dc/72/fbf17c589f92c0b3bbf6709a425661f8ef2ed0d46b38985a7d7b5a0f6b91/propcache-0.5.4-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:7cc528e760a8af06f2b13e9b9f362cd90c7c718ea61228a96dbd31ba16ed7f47", size = 233611, upload-time = "2026-09-16T00:15:06.498Z" }, + { url = "https://files.pythonhosted.org/packages/55/7e/dbd637572a279692e5518d117274a9331bf5faac59f191d30e82521a3ec7/propcache-0.5.4-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:425f8cc86ab5018b4b8d4a23bc8e74d964bd3d757c3702e301aa79be76c53f6c", size = 204333, upload-time = "2026-09-16T00:15:07.961Z" }, + { url = "https://files.pythonhosted.org/packages/ba/5a/f99c92068f1e0f5c886899ce0e4a619db376ca98c5279d93f95bd86906af/propcache-0.5.4-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:a5793c7698a53f56f4a1889a4737c7eeb1b7ad0842fa6b1abca22913ff79c8c1", size = 235177, upload-time = "2026-09-16T00:15:09.334Z" }, + { url = "https://files.pythonhosted.org/packages/ee/28/95456fabd2daf6be89049a13fbf03341756014d2959c83d12957d4c49694/propcache-0.5.4-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:c02c0e570c5c7e077b0181a9f3cdb7d4c3617d1cda6b5c95bd5d34022923d82c", size = 228982, upload-time = "2026-09-16T00:15:10.729Z" }, + { url = "https://files.pythonhosted.org/packages/b1/bb/df90f62c9cf7c93ea235f6f9405143bba802914607317266dd81fc8d737e/propcache-0.5.4-cp313-cp313-win32.whl", hash = "sha256:3e413d7a4a9b4866b7a761d6060d434b64d23cd35122eda3b026a0bbe8196b25", size = 42611, upload-time = "2026-09-16T00:15:12.111Z" }, + { url = "https://files.pythonhosted.org/packages/01/bc/e0a7b84af04ec02d73a48aa71f091e1e4a2107e3074b7ce12195b66901f4/propcache-0.5.4-cp313-cp313-win_amd64.whl", hash = "sha256:0c889f6fa84957bc7e8b4eab71fd16a0455068d5045e3aa40c733071d2b2fd77", size = 45342, upload-time = "2026-09-16T00:15:13.519Z" }, + { url = "https://files.pythonhosted.org/packages/9a/70/50b031cafe72a5c1878b903ee87303f71313345566bf3d6ec202e5ddc9ec/propcache-0.5.4-cp313-cp313-win_arm64.whl", hash = "sha256:69fc35c0779522da366c563e5faf203ffc1f8ff0021d5b1337fa4efa5be73177", size = 42408, upload-time = "2026-09-16T00:15:14.788Z" }, + { url = "https://files.pythonhosted.org/packages/f5/cd/785c64ed382f3f04201870267b02783f63b4678c2acfddc177a3ebcc2727/propcache-0.5.4-py3-none-any.whl", hash = "sha256:62c60aec739ed00124573cce1178138fd690c7676352d67a37328c1cf51d7468", size = 16338, upload-time = "2026-09-16T00:17:13.106Z" }, +] + [[package]] name = "protobuf" version = "7.35.0" @@ -794,6 +1244,34 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/73/e8/2bdf3ca2090f68bb3d75b44da7bbc71843b19c9f2b9cb9b0f4ab7a5a4329/pyyaml-6.0.3-cp313-cp313-win_arm64.whl", hash = "sha256:5498cd1645aa724a7c71c8f378eb29ebe23da2fc0d7a08071d89469bf1d2defb", size = 140246, upload-time = "2025-09-25T21:32:34.663Z" }, ] +[[package]] +name = "requests" +version = "2.34.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "certifi" }, + { name = "charset-normalizer" }, + { name = "idna" }, + { name = "urllib3" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/ac/c3/e2a2b89f2d3e2179abd6d00ebd70bff6273f37fb3e0cc209f48b39d00cbf/requests-2.34.2.tar.gz", hash = "sha256:f288924cae4e29463698d6d60bc6a4da69c89185ad1e0bcc4104f584e960b9ed", size = 142856, upload-time = "2026-05-14T19:25:27.735Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a0/f4/c67b0b3f1b9245e8d266f0f112c500d50e5b4e83cb6f3b71b6528104182a/requests-2.34.2-py3-none-any.whl", hash = "sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0", size = 73075, upload-time = "2026-05-14T19:25:26.443Z" }, +] + +[[package]] +name = "requests-oauthlib" +version = "2.0.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "oauthlib" }, + { name = "requests" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/42/f2/05f29bc3913aea15eb670be136045bf5c5bbf4b99ecb839da9b422bb2c85/requests-oauthlib-2.0.0.tar.gz", hash = "sha256:b3dffaebd884d8cd778494369603a9e7b58d29111bf6b41bdc2dcd87203af4e9", size = 55650, upload-time = "2024-03-22T20:32:29.939Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/3b/5d/63d4ae3b9daea098d5d6f5da83984853c1bbacd5dc826764b249fe119d24/requests_oauthlib-2.0.0-py2.py3-none-any.whl", hash = "sha256:7dd8a5c40426b779b0868c404bdef9768deccf22749cde15852df527e6269b36", size = 24179, upload-time = "2024-03-22T20:32:28.055Z" }, +] + [[package]] name = "six" version = "1.17.0" @@ -850,3 +1328,86 @@ sdist = { url = "https://files.pythonhosted.org/packages/ba/19/1b9b0e29f30c6d35c wheels = [ { url = "https://files.pythonhosted.org/packages/ce/e4/dccd7f47c4b64213ac01ef921a1337ee6e30e8c6466046018326977efd95/tzdata-2026.2-py2.py3-none-any.whl", hash = "sha256:bbe9af844f658da81a5f95019480da3a89415801f6cc966806612cc7169bffe7", size = 349321, upload-time = "2026-04-24T15:22:05.876Z" }, ] + +[[package]] +name = "urllib3" +version = "2.8.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/e3/05/b17359e1cefb4f909b5e40b1b90a496d987258916dbbf88e842c729f510e/urllib3-2.8.0.tar.gz", hash = "sha256:63bf2ead4c879426ebf22ef2a781eeb4aa3b4ae798a0435506f8687fd5bb9b63", size = 458972, upload-time = "2026-09-15T19:29:36.253Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/92/9d/c4e665119135114480843e7ab388fa94d8480650450e6f8e26b70d323a4c/urllib3-2.8.0-py3-none-any.whl", hash = "sha256:0cf3cae568d36aa9576b28dfb35f11328f1cb974ca7647d9475ebb86c75ac6e3", size = 135717, upload-time = "2026-09-15T19:29:34.577Z" }, +] + +[[package]] +name = "websocket-client" +version = "1.9.2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/d8/cb/a5abcc2891249f393827c650c6296660ce40374ac22d99ab9aea41f9d2a2/websocket_client-1.9.2.tar.gz", hash = "sha256:0fcb57545848be86992e128218fd96dd87a6769ffdb1a968dff79632b85604d0", size = 84110, upload-time = "2026-08-31T14:08:40.964Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/d5/d2/cc4dc1271e464942db7ee278baae2daa99ee77cb2af744025c04da585a3e/websocket_client-1.9.2-py3-none-any.whl", hash = "sha256:e1a673830a9c7bfa47b1cd3d5e4178f4c9651d80a4eab02c9c23a1c3ec6250ce", size = 95786, upload-time = "2026-08-31T14:08:39.899Z" }, +] + +[[package]] +name = "yarl" +version = "1.25.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "idna" }, + { name = "multidict" }, + { name = "propcache" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/75/16/e8be8e2fb175bbf41a0680381a319f1199fae256588241a2ac8677eafb49/yarl-1.25.1.tar.gz", hash = "sha256:03dd38de09bc213e9a8b29761eec33ee1d5318dac0e49d8af36e4d27830e23a7", size = 246245, upload-time = "2026-09-15T19:35:02.264Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/83/b3/2cea721d495ca57f8f414aa4867ee263f486b274e07463158cf52f02ba7f/yarl-1.25.1-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:9d693bf4bf534e9ba3ae2780cfd577f5135629f7b5ac653490859d0b77864865", size = 143794, upload-time = "2026-09-15T19:30:26.946Z" }, + { url = "https://files.pythonhosted.org/packages/55/e6/cd145cff8e5cf60b8b3c41fbecfa2a45028a8dec3fbc52bec03595ff3d3b/yarl-1.25.1-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:ab2054c5531af2a9ba7b69b8ec91e4f884420e83a8c5e579b013084cb57e5e5d", size = 103993, upload-time = "2026-09-15T19:30:28.661Z" }, + { url = "https://files.pythonhosted.org/packages/ce/7c/fbb40fe2d53747c40aa36a9e2bd2178a202f942bda0b670f3306d4aefbce/yarl-1.25.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:564fdc7085d2245ab84f88882fdb1d6ac0723124bff6ded35bfb1c00f812630d", size = 104010, upload-time = "2026-09-15T19:30:31.069Z" }, + { url = "https://files.pythonhosted.org/packages/f0/0b/5a516f70641092283f57cf3670bdb75e7327bcc0dcb5038697e4dfbfd569/yarl-1.25.1-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:acae6b45d1ace09b6ba3876da43b88366ef368f73b988c7f57e14231753d4420", size = 116444, upload-time = "2026-09-15T19:30:32.894Z" }, + { url = "https://files.pythonhosted.org/packages/b6/8a/d1a627f827b0a404ae0f5647cab0534959081cde13b0756a783469fb3b5e/yarl-1.25.1-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:1fb2a01ba8cd9c5d2c5dc1ec35e0fc951d04b4f037541d4ac090c993ce58b3d7", size = 107565, upload-time = "2026-09-15T19:30:35.188Z" }, + { url = "https://files.pythonhosted.org/packages/aa/cf/c8e0aaec886840a6c4480fe44eaec7cd4319f79a563b79321473be79c56f/yarl-1.25.1-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:e92b6bcc741b86d67606c40d3cb9c7cc8e6c737f81e31f4a94efc204456c92e3", size = 125006, upload-time = "2026-09-15T19:30:37.1Z" }, + { url = "https://files.pythonhosted.org/packages/50/26/0cce366d54a93cdc8342965dc7663385e4db161625cc1e8b18e786753d24/yarl-1.25.1-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:72c34ac7ad4314c19362d5ce27626dcc8429bd30bbf8c179f4234078851f9492", size = 128717, upload-time = "2026-09-15T19:30:38.904Z" }, + { url = "https://files.pythonhosted.org/packages/95/0c/a71501bbc1a674ff72c4d6c2b75f4d9a5af819f5244c3a7558080a8802c5/yarl-1.25.1-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d5add7b4ca7afeea91d52e4d4e4db3b1fe9885b71f07054560d8c4296b7441a2", size = 117728, upload-time = "2026-09-15T19:30:40.809Z" }, + { url = "https://files.pythonhosted.org/packages/e7/6f/c3267ca01defeed9ed9c4ff9b17bd54915432c405945233265b707475d1c/yarl-1.25.1-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:def538065f9e4d4cf1ae164bd59aba00dfa84f03923e0de4c3788f252d6bcd17", size = 116224, upload-time = "2026-09-15T19:30:42.818Z" }, + { url = "https://files.pythonhosted.org/packages/8f/69/fad57ee52d648431718ee0f1f68966a99c1352c3924688ffbdcc9d3fe51a/yarl-1.25.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:0a191bfdb30a79b98e5d175d75285f9fcb78bf0e46ba5efda042e1c72071a0de", size = 116299, upload-time = "2026-09-15T19:30:44.966Z" }, + { url = "https://files.pythonhosted.org/packages/2a/d4/6a8c1e29f33338687ca278cba0a8fbf6525a322c2c02a9a500ccbe041152/yarl-1.25.1-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:71f42c5b9a948c113bbdebfa544598321431d064ff959d32e99b1feb61d68345", size = 108625, upload-time = "2026-09-15T19:30:46.874Z" }, + { url = "https://files.pythonhosted.org/packages/e8/d2/3a35ae791c9cb6522c106923ff25c3230d999091e5e65511bca23bbd9914/yarl-1.25.1-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:72849d892954be4d09e569b8b831ac39ce58417fedc767d4308a0fe542018a40", size = 124515, upload-time = "2026-09-15T19:30:49.082Z" }, + { url = "https://files.pythonhosted.org/packages/be/fd/2b022109a6b4af0f7dc371cf7500af380b0d4f034010243e1b0ce218dc93/yarl-1.25.1-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:efb01a106f971cb3752856bca2318bbdf7f01bd8823779c461586cbe5ffd5258", size = 115711, upload-time = "2026-09-15T19:30:51.208Z" }, + { url = "https://files.pythonhosted.org/packages/18/59/f7586271136c3ddb0126bbfe661844699369b76b555fe38c4efe86870b2e/yarl-1.25.1-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:a1daf47cd95a7c3a63456336bc5aaa8c86dd3a47d07ed3d0e76132ae4666a5a1", size = 122751, upload-time = "2026-09-15T19:30:53.535Z" }, + { url = "https://files.pythonhosted.org/packages/a0/0b/07f7a2d881f7e16c385b47fc1753382600847cad305dcc7fa0c25828acf8/yarl-1.25.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:9489e6abf47ba37f332075a91444c7cfedb03e6ce99fbb2f116bfe1ce810da3b", size = 117983, upload-time = "2026-09-15T19:30:55.277Z" }, + { url = "https://files.pythonhosted.org/packages/aa/9d/8cdceec66a9b940700cb45931741403f045b162afd79bcb93c41cadd0972/yarl-1.25.1-cp311-cp311-win_amd64.whl", hash = "sha256:d7306dee25b8a0e737363f347362b875094b4dc4e367311470656ae420fdbf8e", size = 102894, upload-time = "2026-09-15T19:30:57.606Z" }, + { url = "https://files.pythonhosted.org/packages/17/f1/7ec357db1d3ad2863542d71e8fe64a126bbec17b134cd7d30951196809a5/yarl-1.25.1-cp311-cp311-win_arm64.whl", hash = "sha256:abb1384477f5901d436b5d2e5465954de46ea6098f59163d243660b5c4461d35", size = 98609, upload-time = "2026-09-15T19:30:59.705Z" }, + { url = "https://files.pythonhosted.org/packages/75/b3/cd32ac66ae622b854c2df0ac52106dda220d361b65a64fde7d5b3684aa3f/yarl-1.25.1-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:94d7aa6debf92a1dd14cb5280b083a764169a13cfb23a452111160274ed989f4", size = 144798, upload-time = "2026-09-15T19:31:01.821Z" }, + { url = "https://files.pythonhosted.org/packages/61/fb/a2c52a8007c2051ba74662afb112ecf3d00346af4c25e33df9d80fd14fb8/yarl-1.25.1-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:83d4a37e4b95da4d8bda930d6d35b75b4cdadbacbb4980cae290ea3100b5d51d", size = 104583, upload-time = "2026-09-15T19:31:04.05Z" }, + { url = "https://files.pythonhosted.org/packages/be/dd/ee38aec8e09fdf957e50d4085453fbe202f56c6c3b4cf07b81cdb4f09ee9/yarl-1.25.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:e029648f9c951db30e98a7d7ec90835db88ec4b32820efe2a9bdc2287e032eb6", size = 104325, upload-time = "2026-09-15T19:31:06.338Z" }, + { url = "https://files.pythonhosted.org/packages/1e/b3/058dbfb1857b484c9cf9cc135659f50b85ce66e03c99e44dc2f7b6161f55/yarl-1.25.1-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:4d781294bb815ecb5ea57ff6bbf8038e0a31a95fdf3e1788f66e0dc100d64b58", size = 115358, upload-time = "2026-09-15T19:31:08.593Z" }, + { url = "https://files.pythonhosted.org/packages/db/39/29693446cf0cf6b15a0e2f75a5d40f93c56819b05b0622196f45e95b5cc0/yarl-1.25.1-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:e12c538e00e7c1b286a07061046b90e8124e6a9793efae2c70db6a4aad07faad", size = 107658, upload-time = "2026-09-15T19:31:10.802Z" }, + { url = "https://files.pythonhosted.org/packages/86/b3/3c4dd7e1af43b931fba95e0a722737f2ea94a6d199c802585282831d7abd/yarl-1.25.1-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:7e4de3ac4adbad3d0bc7c6f4360a7dbff5de2f15e3b723be3198074e17fd9c40", size = 122660, upload-time = "2026-09-15T19:31:12.84Z" }, + { url = "https://files.pythonhosted.org/packages/bd/b5/1b60dbc3cfc9c5712b15148c206748f2bc93953ffdbe25ea75b63dfc89c9/yarl-1.25.1-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:419f392a1da624877975709e3864dfe833af6cc7671b39318086d456e288380c", size = 126506, upload-time = "2026-09-15T19:31:15.088Z" }, + { url = "https://files.pythonhosted.org/packages/bc/7b/ca212cbe170ac8b96e45317ecbcf9c3c3ecf0cdec98d5b088a9c4088929b/yarl-1.25.1-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:c6f117789d22dce188e5754e8bc65b7e6ebf8cb73963b9fa761f672a5883769d", size = 117050, upload-time = "2026-09-15T19:31:17.241Z" }, + { url = "https://files.pythonhosted.org/packages/cb/c3/72b4938cdbe619ad71ac156182faef4908846b84dc3ca4dbb4c4e6f84014/yarl-1.25.1-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:80e47012e730da131c9f059c80936783f9659aae22dc31c03c0595590d11ed54", size = 114174, upload-time = "2026-09-15T19:31:19.294Z" }, + { url = "https://files.pythonhosted.org/packages/e8/43/268717870f9ba0cc9701a95181587f6dc8c5f387aab4aeecc83158f38a79/yarl-1.25.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:e80f557716fd765439577131e526b8942ffc2c07bdbc5e39fa62f660ba1e963f", size = 114944, upload-time = "2026-09-15T19:31:21.414Z" }, + { url = "https://files.pythonhosted.org/packages/da/84/baa5bf504d51fe062c4bcaf62936da97fffb43285978d0b39984824231fd/yarl-1.25.1-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:f61964f235a43738bfac50da46fc4254943a7eea3051aeb0b6fc7c992c29fadc", size = 108263, upload-time = "2026-09-15T19:31:23.388Z" }, + { url = "https://files.pythonhosted.org/packages/a4/28/779a2ed9e0152a601a27039bed9aead3f0b79797a67e2c44bfa444622dd8/yarl-1.25.1-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:e546fe1d4a93ebc2910f0d768baff19faa09843ab3f2036a67ed6e69fae4419d", size = 122184, upload-time = "2026-09-15T19:31:25.343Z" }, + { url = "https://files.pythonhosted.org/packages/f8/1f/118e9e5b8f07694d63fd3222e801d7782270003f1a222aa798df3f8d5933/yarl-1.25.1-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:cce0727fd5ac04d372fa9bbfde9febc2bcf209aadfcf0468e45dec72719895d1", size = 114001, upload-time = "2026-09-15T19:31:27.465Z" }, + { url = "https://files.pythonhosted.org/packages/0f/ae/a4cf1cf372313734b17996d4007f9f73596e7a178b9485802e5494ecf484/yarl-1.25.1-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:af4ea5b37403ef4e30f3927eaed540db942bde01d8d3ff083527c0704d1c9c68", size = 120565, upload-time = "2026-09-15T19:31:29.47Z" }, + { url = "https://files.pythonhosted.org/packages/05/79/ad94f93ca731bc9e44d321833ab96b82a4f9f5f63cf773f81a4aeea5ecc1/yarl-1.25.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:68782fdb4027b8d1eee25ec35e9a6db05e863b899eb0310b3a33b6c3fef55707", size = 117060, upload-time = "2026-09-15T19:31:31.367Z" }, + { url = "https://files.pythonhosted.org/packages/bb/cc/51a7b4abf4ac593b8e7eb3794b28e5a35ae26eed8bc04787628d215af82f/yarl-1.25.1-cp312-cp312-win_amd64.whl", hash = "sha256:7d575b54cb3863ef9bc290ea4b009999d55dc237326131e4853cf33e888fee03", size = 102593, upload-time = "2026-09-15T19:31:33.329Z" }, + { url = "https://files.pythonhosted.org/packages/9d/21/0941a6b93a58b59a1ec75e5333bf06929b671309c43c0cd201c172d9c39f/yarl-1.25.1-cp312-cp312-win_arm64.whl", hash = "sha256:bc3ac7bf569f6b64dad04dd7808c7872dae8a97df657856eac05e9b7e3614a85", size = 97697, upload-time = "2026-09-15T19:31:35.855Z" }, + { url = "https://files.pythonhosted.org/packages/7b/ed/2f3129bbcc9a5c8ba12cc2b29d8060a3bab9c8043c456cfd4b5ca3188890/yarl-1.25.1-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:25868beca8b6765f8f7d0e11fe6dd7c66dd4b0793b9500286d20cc92352126a5", size = 143623, upload-time = "2026-09-15T19:31:37.966Z" }, + { url = "https://files.pythonhosted.org/packages/17/e1/f1bc3390fdca352826676b531d0712736f156919090206700421d46b2c37/yarl-1.25.1-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:10b2fd95332f0d716d5eee3c9fb2ce8eada19082de7fee83d32e37992fd75c26", size = 104011, upload-time = "2026-09-15T19:31:40.25Z" }, + { url = "https://files.pythonhosted.org/packages/a8/aa/50acc5c3e5da04172ae3c281c75405af4d2ca911e16120ab0563f4dffb66/yarl-1.25.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:0f12afda4eea8c8994a76d4df1875c765194f5fbe8a9d197929ea303caee29ec", size = 103677, upload-time = "2026-09-15T19:31:42.46Z" }, + { url = "https://files.pythonhosted.org/packages/30/d2/7d1e0ab9f8390e1fbcede5a6dbf70d23c96ad09b8c5567f3a514d1ddb0e2/yarl-1.25.1-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:14b79a30a93a3ce2e8832603fd0ab780ada281b0ba5110b519a634f2d7d7d1fc", size = 115392, upload-time = "2026-09-15T19:31:44.371Z" }, + { url = "https://files.pythonhosted.org/packages/71/e1/5ba1e3a2a22139213655e760919038e8ed7e2d4a99826d0bbddb3beb96e5/yarl-1.25.1-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:4bd6340d20ae2c7ca719b87b426e808e90743b676d05d4c26c4fb5ca71f41184", size = 107493, upload-time = "2026-09-15T19:31:46.273Z" }, + { url = "https://files.pythonhosted.org/packages/f5/53/780653d5e0f73831f467cf13548912e5eec97f21dc49fc8daf21da027df4/yarl-1.25.1-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:126a2533570c554719ca40a1288fdee1700b6bc82e7131aa69fa85252d92e651", size = 122537, upload-time = "2026-09-15T19:31:48.654Z" }, + { url = "https://files.pythonhosted.org/packages/03/92/d54fa70236c6036271c9c9c09fd978df5cbe3ef49ef6c46e9b833476d215/yarl-1.25.1-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:a3faadac7d812ddac258feb57b9846b60c1b437c4f4b9ad42595c6f6fe4390df", size = 126170, upload-time = "2026-09-15T19:31:50.872Z" }, + { url = "https://files.pythonhosted.org/packages/0e/b7/a82a49bf88340b837ef6972b508a1604ae377b9e6904b46b10cf5f1cf925/yarl-1.25.1-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:be80550d9bfe83d9b62398a37081a90434e6df2d978ec345c3d2820de6beddab", size = 117012, upload-time = "2026-09-15T19:31:53.189Z" }, + { url = "https://files.pythonhosted.org/packages/ef/78/5d684b411e3f3602464ee9b538db48205038f8605872985f61efb809ced0/yarl-1.25.1-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:e07595c7d6f4db270ceede356a1bd1c07a34f1c26f958d1ed0cd7b48e0d2bba3", size = 114950, upload-time = "2026-09-15T19:31:55.694Z" }, + { url = "https://files.pythonhosted.org/packages/2f/11/51d82b852c64f7fad0fc7a7ff3031517204887e874c722bbca839c0b23ac/yarl-1.25.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:eb96ed1ae6c7d072d60840c0434aef07a2df611812810807fbc54263a6053e9a", size = 115428, upload-time = "2026-09-15T19:31:57.966Z" }, + { url = "https://files.pythonhosted.org/packages/e4/49/9d1978049bf646b9ea918313926453c6901b71c92f097467777d47d36a88/yarl-1.25.1-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:3feb99222553a8cbedfa52c2f59dd84c3f50d5b582c728d522caf8d72769a54b", size = 108428, upload-time = "2026-09-15T19:32:00.048Z" }, + { url = "https://files.pythonhosted.org/packages/43/35/7b8f1ebb45d7ec3dda7d1909bf44f458de41ef91e2937f107733582a5166/yarl-1.25.1-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:a2ed0ba415ccdf08f14bf544cb78346d0f76086707ffee24921a2c84dbf1305a", size = 121961, upload-time = "2026-09-15T19:32:02.436Z" }, + { url = "https://files.pythonhosted.org/packages/63/d6/d8b689ab7ca26edeb85f6ff28812aac7a25376eefc1780e303a7bfbaceff/yarl-1.25.1-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:2b49375d22299b0a834c2bca72f39aaecc270d96fb24c30424899676f487b22a", size = 114961, upload-time = "2026-09-15T19:32:04.456Z" }, + { url = "https://files.pythonhosted.org/packages/cf/d5/1a1798ea4dc6b7ee3260010a27907ebc697c95dae99817d817ed446d24aa/yarl-1.25.1-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:ef74070ac553c59eb4f04258722066d6c6135b7baa03b2e9f2da65c096e96d98", size = 120036, upload-time = "2026-09-15T19:32:06.5Z" }, + { url = "https://files.pythonhosted.org/packages/91/8d/b1b35ed7903da6669b1d367cb2c09436acd4ff508029b4f39a0c0c2058fc/yarl-1.25.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:0a66db89ea473abeac4b70523cafd94db3772380e565f9d28af7a179b7af71fa", size = 117276, upload-time = "2026-09-15T19:32:09.401Z" }, + { url = "https://files.pythonhosted.org/packages/3a/8f/4db01cef62caff0d7a4593ed694fb8a41a27a11158cab80d290221f13e57/yarl-1.25.1-cp313-cp313-win_amd64.whl", hash = "sha256:1f51020b2eb8a003c84925638ec63c21a750a4bddd3a22ec8eac6a742dadf1b9", size = 101945, upload-time = "2026-09-15T19:32:11.545Z" }, + { url = "https://files.pythonhosted.org/packages/c0/5e/3ce00497c5c0babb74d4130c10c3828ccd215b4819d12020c42429f991ac/yarl-1.25.1-cp313-cp313-win_arm64.whl", hash = "sha256:b10dd0557ba422715b5206b3743192135a6022acca8baec51aa127d0a75db8fe", size = 97270, upload-time = "2026-09-15T19:32:14.127Z" }, + { url = "https://files.pythonhosted.org/packages/54/22/318c7980066769c6bcd9221ed2248294f5698811da099013098c670565ed/yarl-1.25.1-py3-none-any.whl", hash = "sha256:681c758b0490f9e96b78e5fa8e8dc6e648e9185bb6eaebe73183c33ea0c445f3", size = 63617, upload-time = "2026-09-15T19:34:59.616Z" }, +]