From c51ec18532179d76cb322e0035e656737a0863e6 Mon Sep 17 00:00:00 2001 From: Kristina Pathak Date: Tue, 6 Oct 2026 14:35:17 -0700 Subject: [PATCH 1/6] refactor(llm-routing)!: rename the Spark recipe to a generic recipes tool The GLM recipe is not specific to DGX Spark hardware, and more recipes may follow. Move deploy/helm/llm-routing/spark to recipes/ and spark.py to recipe.py, and drop Spark from user-facing names. Behavior is unchanged; the existing tests pass with the new names. BREAKING CHANGE: SPARK_CONTEXT is now LLM_ROUTING_CONTEXT, the default namespace and clusterId are llm-routing-poc, and the stack values sparkRecipeSource and sparkRecipeChartsSha256 are now recipeSource and recipeChartsSha256. Installations made with the Spark recipe must be reinstalled, or upgraded with a coordinated stack update, before image update and rollback work with this tool. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: Kristina Pathak --- NOTICE | 2 +- .../llm-gateway-stack/values.yaml | 2 +- deploy/helm/llm-routing/AGENTS.md | 4 +- deploy/helm/llm-routing/README.md | 86 +++---- .../llm-routing/{spark => recipes}/.gitignore | 0 deploy/helm/llm-routing/recipes/AGENTS.md | 9 + .../{spark => recipes}/BUILDING.md | 18 +- .../llm-routing/{spark => recipes}/CLAUDE.md | 0 .../llm-routing/{spark => recipes}/NOTICE | 0 .../llm-routing/{spark => recipes}/README.md | 0 .../{spark => recipes}/backend.defaults.json | 4 +- .../charts/gguf-backend/Chart.yaml | 0 .../gguf-backend/files/artifact-server.py | 0 .../charts/gguf-backend/files/build.py | 0 .../charts/gguf-backend/files/chain-check.py | 0 .../charts/gguf-backend/files/download.py | 0 .../gguf-backend/files/fetch-runtime.py | 0 .../charts/gguf-backend/files/preflight.py | 0 .../charts/gguf-backend/files/qualify.py | 0 .../gguf-backend/files/rpc-chain-check.cpp | 0 .../gguf-backend/files/rpc-gpu-check.cpp | 0 .../charts/gguf-backend/files/rpc-health.py | 0 .../gguf-backend/files/runtime_guard.py | 0 .../charts/gguf-backend/files/serve.py | 0 .../charts/gguf-backend/templates/build.yaml | 0 .../charts/gguf-backend/templates/chain.yaml | 0 .../gguf-backend/templates/download.yaml | 0 .../gguf-backend/templates/preflight.yaml | 0 .../gguf-backend/templates/qualify.yaml | 0 .../gguf-backend/templates/runtime.yaml | 0 .../charts/gguf-backend/templates/serve.yaml | 0 .../gguf-backend/tests/test_download.py | 0 .../gguf-backend/tests/test_rpc_health.py | 0 .../gguf-backend/tests/test_runtime_guard.py | 0 .../charts/gguf-backend/values.yaml | 0 .../charts/image-loader/Chart.yaml | 0 .../charts/image-loader/templates/jobs.yaml | 0 .../charts/image-loader/values.yaml | 0 .../llm-routing/{spark => recipes}/client.py | 0 .../{spark => recipes}/cluster_setup.py | 2 +- .../{spark => recipes}/config.example.json | 20 +- .../{spark => recipes}/console_output.py | 2 +- .../{spark => recipes}/gateway_access.py | 2 +- .../{spark => recipes}/model.lock.json | 0 .../{spark => recipes}/operator.Dockerfile | 0 .../{spark/spark.py => recipes/recipe.py} | 20 +- .../tests/sample-backend/Chart.yaml | 0 .../sample-backend/templates/backend.yaml | 0 .../sample-backend/templates/endpoint.yaml | 0 .../tests/sample-backend/values.yaml | 0 .../tests/test_attach_reuse.py | 30 +-- .../{spark => recipes}/tests/test_chat.py | 24 +- .../{spark => recipes}/tests/test_cli.py | 212 +++++++++--------- .../{spark => recipes}/tests/test_client.py | 2 +- .../tests/test_cluster_setup.py | 0 .../tests/test_console_output.py | 10 +- .../tests/test_discovery.py | 18 +- .../tests/test_gateway_access.py | 0 .../tests/test_image_import.py | 18 +- .../tests/test_load_resume.py | 18 +- .../{spark => recipes}/tests/test_recipe.py | 170 +++++++------- .../tests/test_reinitialization.py | 0 deploy/helm/llm-routing/spark/AGENTS.md | 9 - 63 files changed, 341 insertions(+), 341 deletions(-) rename deploy/helm/llm-routing/{spark => recipes}/.gitignore (100%) create mode 100644 deploy/helm/llm-routing/recipes/AGENTS.md rename deploy/helm/llm-routing/{spark => recipes}/BUILDING.md (82%) rename deploy/helm/llm-routing/{spark => recipes}/CLAUDE.md (100%) rename deploy/helm/llm-routing/{spark => recipes}/NOTICE (100%) rename deploy/helm/llm-routing/{spark => recipes}/README.md (100%) rename deploy/helm/llm-routing/{spark => recipes}/backend.defaults.json (98%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/Chart.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/files/artifact-server.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/files/build.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/files/chain-check.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/files/download.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/files/fetch-runtime.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/files/preflight.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/files/qualify.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/files/rpc-chain-check.cpp (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/files/rpc-gpu-check.cpp (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/files/rpc-health.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/files/runtime_guard.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/files/serve.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/templates/build.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/templates/chain.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/templates/download.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/templates/preflight.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/templates/qualify.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/templates/runtime.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/templates/serve.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/tests/test_download.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/tests/test_rpc_health.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/tests/test_runtime_guard.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/gguf-backend/values.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/image-loader/Chart.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/image-loader/templates/jobs.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/charts/image-loader/values.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/client.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/cluster_setup.py (99%) rename deploy/helm/llm-routing/{spark => recipes}/config.example.json (69%) rename deploy/helm/llm-routing/{spark => recipes}/console_output.py (99%) rename deploy/helm/llm-routing/{spark => recipes}/gateway_access.py (99%) rename deploy/helm/llm-routing/{spark => recipes}/model.lock.json (100%) rename deploy/helm/llm-routing/{spark => recipes}/operator.Dockerfile (100%) rename deploy/helm/llm-routing/{spark/spark.py => recipes/recipe.py} (98%) rename deploy/helm/llm-routing/{spark => recipes}/tests/sample-backend/Chart.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/tests/sample-backend/templates/backend.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/tests/sample-backend/templates/endpoint.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/tests/sample-backend/values.yaml (100%) rename deploy/helm/llm-routing/{spark => recipes}/tests/test_attach_reuse.py (87%) rename deploy/helm/llm-routing/{spark => recipes}/tests/test_chat.py (88%) rename deploy/helm/llm-routing/{spark => recipes}/tests/test_cli.py (67%) rename deploy/helm/llm-routing/{spark => recipes}/tests/test_client.py (99%) rename deploy/helm/llm-routing/{spark => recipes}/tests/test_cluster_setup.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/tests/test_console_output.py (96%) rename deploy/helm/llm-routing/{spark => recipes}/tests/test_discovery.py (93%) rename deploy/helm/llm-routing/{spark => recipes}/tests/test_gateway_access.py (100%) rename deploy/helm/llm-routing/{spark => recipes}/tests/test_image_import.py (96%) rename deploy/helm/llm-routing/{spark => recipes}/tests/test_load_resume.py (95%) rename deploy/helm/llm-routing/{spark => recipes}/tests/test_recipe.py (86%) rename deploy/helm/llm-routing/{spark => recipes}/tests/test_reinitialization.py (100%) delete mode 100644 deploy/helm/llm-routing/spark/AGENTS.md diff --git a/NOTICE b/NOTICE index ad6530d3a..cce2ffddf 100644 --- a/NOTICE +++ b/NOTICE @@ -9,7 +9,7 @@ regenerate this file by running: The following third-party licenses are included in this repository: deploy/helm/container-cache/NOTICE - deploy/helm/llm-routing/spark/NOTICE + deploy/helm/llm-routing/recipes/NOTICE deploy/helm/nats/NOTICE deploy/helm/openbao/NOTICE infra/openbao/plugins/vault-plugin-secrets-jwt/NOTICE diff --git a/deploy/helm/llm-gateway-stack/llm-gateway-stack/values.yaml b/deploy/helm/llm-gateway-stack/llm-gateway-stack/values.yaml index f800c3c35..35c8d8581 100644 --- a/deploy/helm/llm-gateway-stack/llm-gateway-stack/values.yaml +++ b/deploy/helm/llm-gateway-stack/llm-gateway-stack/values.yaml @@ -47,7 +47,7 @@ clusterCredential: # sha256: apiKeys: [] -# Optional plaintext UI key, supplied from a private file by the Spark recipe. +# Optional plaintext UI key, supplied from a private file by the routing recipe. # Creates Secret demo-ui-api-key (field api-key) in this release's namespace. # apiKeys must also contain its matching hash under id demo-ui. demoUiApiKey: "" diff --git a/deploy/helm/llm-routing/AGENTS.md b/deploy/helm/llm-routing/AGENTS.md index cc951fb85..2b0848e78 100644 --- a/deploy/helm/llm-routing/AGENTS.md +++ b/deploy/helm/llm-routing/AGENTS.md @@ -1,7 +1,7 @@ # LLM routing stack -This directory deploys the LLM routing stack and GLM on DGX Spark. Read `README.md` and `spark/AGENTS.md`. Edit runtime code in its owning source directories in the current checkout. Builds include local edits. Commit IDs are informational. Image updates compare routing chart contents with the fingerprint recorded in the installed stack. +This directory deploys the LLM routing stack and GLM on DGX Spark. Read `README.md` and `recipes/AGENTS.md`. Edit runtime code in its owning source directories in the current checkout. Builds include local edits. Commit IDs are informational. Image updates compare routing chart contents with the fingerprint recorded in the installed stack. Run the Python tests and offline Helm render documented in the README. Always specify the Kubernetes context. Packaging validation must not change a live model deployment. -Keep mocks under `spark/tests` and out of the default deployment. Keep credentials, private targets, kubeconfigs and evidence in an external work directory. Do not publish private configuration overlays. +Keep mocks under `recipes/tests` and out of the default deployment. Keep credentials, private targets, kubeconfigs and evidence in an external work directory. Do not publish private configuration overlays. diff --git a/deploy/helm/llm-routing/README.md b/deploy/helm/llm-routing/README.md index dc5a3ac68..a9bbb8b90 100644 --- a/deploy/helm/llm-routing/README.md +++ b/deploy/helm/llm-routing/README.md @@ -4,7 +4,7 @@ Deploy the LLM Gateway Stack (LLM API Gateway and request router) and Pylon Oper ## Overview -The `spark.py` installer coordinates the combined gateway/router chart, the Pylon Operator chart and the GLM backend chart. The backend chart builds llama.cpp, downloads and verifies the GGUF model files, runs GLM across the two GPUs, and creates an `InferenceEndpoint` for Pylon to register. +The `recipe.py` installer coordinates the combined gateway/router chart, the Pylon Operator chart and the GLM backend chart. The backend chart builds llama.cpp, downloads and verifies the GGUF model files, runs GLM across the two GPUs, and creates an `InferenceEndpoint` for Pylon to register. Application images and charts use your checkout, including local edits. @@ -14,7 +14,7 @@ Application images and charts use your checkout, including local edits. - Kubernetes and storage: an existing ARM64 Kubernetes cluster containing these nodes, with pod networking, `cluster.local` DNS, NetworkPolicy enforcement and node-compatible `ReadWriteOnce` persistent storage. Reserve 400 GiB on the leader and 160 GiB on the worker. - GPU enablement: model nodes with a working NVIDIA driver, NVIDIA Container Toolkit configured for the container runtime, and a device plugin advertising `nvidia.com/gpu`. Use an installed `RuntimeClass` matching `runtimeClass` in the config (`nvidia` in the example). - Access: kubeconfig for the target cluster. For a kubeconfig with multiple contexts, pass `--context ` to the commands below. Initial installation requires creating namespaces, custom resource definitions (CRDs), role-based access control (RBAC) resources and namespaced Helm resources. -- Workstation: Python 3.11+, Git, Helm 3.14+ or Helm 4, and kubectl. Provide access to GitHub, model downloads and container images. The [image build guide](spark/BUILDING.md) covers build tools and distribution credentials. +- Workstation: Python 3.11+, Git, Helm 3.14+ or Helm 4, and kubectl. Provide access to GitHub, model downloads and container images. The [image build guide](recipes/BUILDING.md) covers build tools and distribution credentials. ## Installation @@ -27,8 +27,8 @@ Use your existing kubeconfig. 1. From the repository root, open the recipe directory and create the configuration. ```bash - cd deploy/helm/llm-routing/spark - python3 spark.py init + cd deploy/helm/llm-routing/recipes + python3 recipe.py init ``` Review the selected nodes and generated configuration. `init` discovers idle GPUs, storage and image preload settings for K3s. The configuration is saved at the printed path in a private work directory. @@ -36,37 +36,37 @@ Use your existing kubeconfig. 2. Render the manifests and inventory the cluster. ```bash - python3 spark.py render - python3 spark.py inventory + python3 recipe.py render + python3 recipe.py inventory ``` `render` lints the Helm charts and generates manifests. `inventory` checks node readiness, GPU availability and cluster prerequisites. Continue after both commands pass. ### Build and distribute the application images -Build and distribute `gateway`, `router`, `pylon` and `operator` using the [image build guide](spark/BUILDING.md). +Build and distribute `gateway`, `router`, `pylon` and `operator` using the [image build guide](recipes/BUILDING.md). ### Deploy in order Run each command in order and continue after it succeeds. ```bash -python3 spark.py preflight -python3 spark.py stack -python3 spark.py build-runtime -python3 spark.py qualify -python3 spark.py download -python3 spark.py load -python3 spark.py verify-direct -python3 spark.py register -python3 spark.py verify-gateway +python3 recipe.py preflight +python3 recipe.py stack +python3 recipe.py build-runtime +python3 recipe.py qualify +python3 recipe.py download +python3 recipe.py load +python3 recipe.py verify-direct +python3 recipe.py register +python3 recipe.py verify-gateway ``` 1. `preflight`: Check GPU calculations and available memory on both model nodes. The reference GPU environment uses NVIDIA driver `580.178.04` and CUDA 13. Rerun `preflight` and `qualify` after changing these versions. 2. `stack`: Install the gateway, router and Pylon Operator. This generates the default caller API key and self-signed certificates. To supply your own, complete [Optional configuration](#optional-configuration) before running `stack`. 3. `build-runtime`: Build llama.cpp with CUDA and remote procedure call (RPC) support. -4. `qualify`: Test calculations and data transfer across both GPUs. (After fixing a failed qualification Job, run `python3 spark.py qualify --retry`.) -5. `download`: Download the six GLM files and verify their sizes and SHA256 checksums. See the [model and runtime licenses](spark/NOTICE). +4. `qualify`: Test calculations and data transfer across both GPUs. (After fixing a failed qualification Job, run `python3 recipe.py qualify --retry`.) +5. `download`: Download the six GLM files and verify their sizes and SHA256 checksums. See the [model and runtime licenses](recipes/NOTICE). 6. `load`: Load GLM across both GPUs and wait for the model server. 7. `verify-direct`: Test model answers and streaming directly. 8. `register`: Register GLM with Pylon and wait for readiness. @@ -80,12 +80,12 @@ Pinned runtime: ## Verification -Run these commands from `deploy/helm/llm-routing/spark` after installation or [attachment to an existing stack](#update-an-existing-installation). +Run these commands from `deploy/helm/llm-routing/recipes` after installation or [attachment to an existing stack](#update-an-existing-installation). ### Inspect the deployment ```bash -context="$(python3 spark.py context)" && +context="$(python3 recipe.py context)" && kubectl --context "$context" get nodes -o wide && kubectl --context "$context" get deployments,pods,services,inferenceendpoints --all-namespaces -o wide ``` @@ -95,8 +95,8 @@ Check the node placement, ready replicas and model endpoint status. ### Send a chat or streaming request ```bash -python3 spark.py chat 'What is 17 multiplied by 19? Give one short sentence.' -python3 spark.py chat 'Explain what a GPU does in two sentences.' --stream +python3 recipe.py chat 'What is 17 multiplied by 19? Give one short sentence.' +python3 recipe.py chat 'Explain what a GPU does in two sentences.' --stream ``` ## Maintenance @@ -106,12 +106,12 @@ python3 spark.py chat 'Explain what a GPU does in two sentences.' --stream If this workstation has not used the running installation before, [attach to it first](#update-an-existing-installation). 1. Edit the service in your checkout: `src/invocation-plane-services/llm-api-gateway` for gateway or `src/libraries/rust/stargate` for router. -2. [Build and distribute that component](spark/BUILDING.md#rebuild-gateway-or-router) with a fresh tag. Run the next commands in the same terminal. +2. [Build and distribute that component](recipes/BUILDING.md#rebuild-gateway-or-router) with a fresh tag. Run the next commands in the same terminal. 3. Update the selected image and verify gateway requests. ```bash - python3 spark.py update --component "$COMPONENT" --tag "$NEW_TAG" - python3 spark.py verify-gateway + python3 recipe.py update --component "$COMPONENT" --tag "$NEW_TAG" + python3 recipe.py verify-gateway ``` 4. Save the printed update record path for rollback. @@ -124,8 +124,8 @@ To roll back the image update: 2. Restore the recorded tag and verify requests. ```bash - python3 spark.py rollback --result /path/to/saved-update.json - python3 spark.py verify-gateway + python3 recipe.py rollback --result /path/to/saved-update.json + python3 recipe.py verify-gateway ``` Use the record from the latest update when rolling back. @@ -137,14 +137,14 @@ Use your existing kubeconfig. 1. From the repository root, open the recipe directory. ```bash - cd deploy/helm/llm-routing/spark + cd deploy/helm/llm-routing/recipes ``` 2. Discover the installation and verify gateway requests. ```bash - python3 spark.py attach-existing - python3 spark.py verify-gateway + python3 recipe.py attach-existing + python3 recipe.py verify-gateway ``` Add `--namespace ` to attachment when the cluster has multiple installations or your access is limited to one namespace. @@ -153,7 +153,7 @@ Continue with [Update only gateway or router](#update-only-gateway-or-router). ### Uninstall -Run the entire block, including parentheses, from `deploy/helm/llm-routing/spark` in the same configured terminal used for installation. The context lookup uses the recipe's normal selection. If you passed `--context`, `--config` or `--work-dir` during installation, pass the same options before `context` in the lookup below. +Run the entire block, including parentheses, from `deploy/helm/llm-routing/recipes` in the same configured terminal used for installation. The context lookup uses the recipe's normal selection. If you passed `--context`, `--config` or `--work-dir` during installation, pass the same options before `context` in the lookup below. The namespace and release names below are the default K3s recipe values. If you changed `namespace`, `releasePrefix` or `releases` in your saved configuration, replace these names to match. The model chain release is the GLM release name plus `-chain`, and the image-import release is the release prefix plus `-images`. @@ -162,9 +162,9 @@ The block skips absent releases, including the optional image importer, and stop ```bash ( set -eu - context="$(python3 spark.py context)" + context="$(python3 recipe.py context)" : "${context:?Context lookup returned an empty value}" - namespace=llm-spark-poc + namespace=llm-routing-poc : "${namespace:?Set the namespace from your saved configuration}" helm --kube-context "$context" -n "$namespace" uninstall llm-poc-glm --ignore-not-found --wait --timeout 3m @@ -181,9 +181,9 @@ The model/artifact and RPC-cache PVCs, downloaded models, namespace, InferenceEn After uninstalling the demo releases, run these commands from the recipe directory with the same configuration and context selection used for installation: ```bash -python3 spark.py init -python3 spark.py render -python3 spark.py inventory +python3 recipe.py init +python3 recipe.py render +python3 recipe.py inventory ``` `init` checks that the demo is uninstalled, reuses the saved placement, image references and credentials, and archives stale progress under `before-reinit-*` in the work directory. Continue with [Deploy in order](#deploy-in-order), starting at `preflight`. @@ -196,7 +196,7 @@ The stack records a SHA-256 fingerprint of its routing chart files. Image update 1. Make runtime, API and chart changes in the same checkout, including generated files and regression tests. 2. Render the manifests and run the [local regression checks](#local-validation). -3. For an existing installation, preserve its installed Helm values and credentials when upgrading the stack. Copy only `sparkRecipeChartsSha256` from the private work directory's `render/stack-values.json` into those preserved values. The remaining render values are offline test data. Review the rendered changes before applying the upgrade. Fresh installations record this automatically during `stack`. +3. For an existing installation, preserve its installed Helm values and credentials when upgrading the stack. Copy only `recipeChartsSha256` from the private work directory's `render/stack-values.json` into those preserved values. The remaining render values are offline test data. Review the rendered changes before applying the upgrade. Fresh installations record this automatically during `stack`. 4. Build and deploy gateway and router together when their API contract changes. 5. Rerun gateway verification, recovery and image update/rollback checks. @@ -208,7 +208,7 @@ This optional resilience check restarts the RPC worker, interrupts model service 2. Run the explicit recovery check. ```bash - python3 spark.py recover --confirm-model-interruption + python3 recipe.py recover --confirm-model-interruption ``` 3. Save and review the results from the target cluster. @@ -228,7 +228,7 @@ Both model persistent volume claims (PVCs) remain after uninstall. ### Alternative container runtimes and external configuration -For a non-K3s cluster or custom container runtime, prepare an external copy of [config.example.json](spark/config.example.json) before installation. Set the context, node placement, storage, runtime and image settings for your cluster. Use that file instead of `init`, and pass `--config /path/to/config.json` to each recipe command, starting with `render` and `inventory`. +For a non-K3s cluster or custom container runtime, prepare an external copy of [config.example.json](recipes/config.example.json) before installation. Set the context, node placement, storage, runtime and image settings for your cluster. Use that file instead of `init`, and pass `--config /path/to/config.json` to each recipe command, starting with `render` and `inventory`. ### Runtime image mirror @@ -260,18 +260,18 @@ To use existing certificates, complete these steps before running `stack`: ## Troubleshooting -If a command fails, follow the next check and diagnostic log path printed by the CLI. Detailed tool output is saved in private `evidence/*.log` files inside the work directory. Use `python3 spark.py paths` to locate that directory. +If a command fails, follow the next check and diagnostic log path printed by the CLI. Detailed tool output is saved in private `evidence/*.log` files inside the work directory. Use `python3 recipe.py paths` to locate that directory. ### Gateway check failures If `verify-gateway` fails, inspect its results in `evidence/gateway.json` under the local work directory. The check uses local port 18443. Stop a previous port-forward if it occupies that port. -If the command reports incomplete key cleanup, run `python3 spark.py cleanup-key`. +If the command reports incomplete key cleanup, run `python3 recipe.py cleanup-key`. After resolving the problem, rerun the check from the recipe directory: ```bash -python3 spark.py verify-gateway +python3 recipe.py verify-gateway ``` ## Local validation @@ -281,6 +281,6 @@ From the recipe directory, run the runner/client tests, runtime chart tests and ```bash python3 -m unittest discover -s tests -v python3 -m unittest discover -s charts/gguf-backend/tests -v -python3 spark.py render +python3 recipe.py render git diff --check ``` diff --git a/deploy/helm/llm-routing/spark/.gitignore b/deploy/helm/llm-routing/recipes/.gitignore similarity index 100% rename from deploy/helm/llm-routing/spark/.gitignore rename to deploy/helm/llm-routing/recipes/.gitignore diff --git a/deploy/helm/llm-routing/recipes/AGENTS.md b/deploy/helm/llm-routing/recipes/AGENTS.md new file mode 100644 index 000000000..b9fe6459f --- /dev/null +++ b/deploy/helm/llm-routing/recipes/AGENTS.md @@ -0,0 +1,9 @@ +# Spark GLM recipe + +This entry point uses the checkout containing `recipe.py`. Builds include local edits and record the current commit ID for debugging. Image updates and rollback require the routing chart fingerprint recorded at installation. Helm dependencies are built automatically. `--source-dir` selects another existing checkout. Keep test mocks under `tests` and out of the normal model deployment. + +Keep environment-specific values, credentials, kubeconfigs, generated TLS material and runtime evidence outside this repository. Pass an explicit context on every Kubernetes and Helm command. Do not change a live cluster while testing packaging. + +Run `python3 -m unittest discover -s tests -v` and the chart tests under `charts/gguf-backend/tests`. Run `python3 recipe.py --config config.example.json --work-dir /tmp/recipe-render render` for offline chart validation. A render does not establish a fresh-cluster deployment. + +Land runtime and chart fixes in their owning source directories with generated API files and tests. Never hand-edit generated CRDs or deepcopy code. diff --git a/deploy/helm/llm-routing/spark/BUILDING.md b/deploy/helm/llm-routing/recipes/BUILDING.md similarity index 82% rename from deploy/helm/llm-routing/spark/BUILDING.md rename to deploy/helm/llm-routing/recipes/BUILDING.md index 9d8acf801..b189aa987 100644 --- a/deploy/helm/llm-routing/spark/BUILDING.md +++ b/deploy/helm/llm-routing/recipes/BUILDING.md @@ -1,6 +1,6 @@ # Build application images -Build gateway, router, Pylon and operator images for `linux/arm64` from your checkout, including local edits. Run these commands from `deploy/helm/llm-routing/spark`. +Build gateway, router, Pylon and operator images for `linux/arm64` from your checkout, including local edits. Run these commands from `deploy/helm/llm-routing/recipes`. ## Requirements @@ -14,7 +14,7 @@ Router and Pylon builds use Cargo profile `integration`. 1. Build all four images. ```bash - python3 spark.py build-images + python3 recipe.py build-images ``` 2. Choose the distribution method. @@ -22,13 +22,13 @@ Router and Pylon builds use Cargo profile `integration`. - Registry: set `images.pullPolicy` to `IfNotPresent`, authenticate Docker with registry write credentials, then push. Every node where Pylon can schedule needs registry pull access. ```bash - python3 spark.py push-images + python3 recipe.py push-images ``` - Node preload: set `images.pullPolicy` to `Never`, export the images, then [import the archive](#import-an-archive). ```bash - python3 spark.py export-images + python3 recipe.py export-images ``` 3. Continue with [Deploy in order](../README.md#deploy-in-order). @@ -40,7 +40,7 @@ Router and Pylon builds use Cargo profile `integration`. ```bash COMPONENT=gateway # Or router. NEW_TAG=dev-$(date -u +%Y%m%d%H%M%S) - python3 spark.py build-images --component "$COMPONENT" --tag "$NEW_TAG" + python3 recipe.py build-images --component "$COMPONENT" --tag "$NEW_TAG" ``` 2. Distribute the image using your existing method. @@ -48,13 +48,13 @@ Router and Pylon builds use Cargo profile `integration`. - Registry: ```bash - python3 spark.py push-images --component "$COMPONENT" --tag "$NEW_TAG" + python3 recipe.py push-images --component "$COMPONENT" --tag "$NEW_TAG" ``` - Node preload: export the image, then [import the archive](#import-an-archive). ```bash - python3 spark.py export-images --component "$COMPONENT" --tag "$NEW_TAG" + python3 recipe.py export-images --component "$COMPONENT" --tag "$NEW_TAG" ``` 3. Keep `COMPONENT` and `NEW_TAG` set and continue with [Update only gateway or router](../README.md#update-only-gateway-or-router). @@ -66,13 +66,13 @@ After `export-images`, upload and import the archive through Kubernetes. Keep ea For all four images: ```bash -python3 spark.py import-images --allow-containerd-import +python3 recipe.py import-images --allow-containerd-import ``` For a gateway/router rebuild, import only that component on the control node: ```bash -python3 spark.py import-images \ +python3 recipe.py import-images \ --component "$COMPONENT" --tag "$NEW_TAG" --allow-containerd-import ``` diff --git a/deploy/helm/llm-routing/spark/CLAUDE.md b/deploy/helm/llm-routing/recipes/CLAUDE.md similarity index 100% rename from deploy/helm/llm-routing/spark/CLAUDE.md rename to deploy/helm/llm-routing/recipes/CLAUDE.md diff --git a/deploy/helm/llm-routing/spark/NOTICE b/deploy/helm/llm-routing/recipes/NOTICE similarity index 100% rename from deploy/helm/llm-routing/spark/NOTICE rename to deploy/helm/llm-routing/recipes/NOTICE diff --git a/deploy/helm/llm-routing/spark/README.md b/deploy/helm/llm-routing/recipes/README.md similarity index 100% rename from deploy/helm/llm-routing/spark/README.md rename to deploy/helm/llm-routing/recipes/README.md diff --git a/deploy/helm/llm-routing/spark/backend.defaults.json b/deploy/helm/llm-routing/recipes/backend.defaults.json similarity index 98% rename from deploy/helm/llm-routing/spark/backend.defaults.json rename to deploy/helm/llm-routing/recipes/backend.defaults.json index 655c67ed8..fa4db2264 100644 --- a/deploy/helm/llm-routing/spark/backend.defaults.json +++ b/deploy/helm/llm-routing/recipes/backend.defaults.json @@ -5,11 +5,11 @@ "targets": [ { "id": "leader", - "node": "spark-leader" + "node": "model-0" }, { "id": "worker", - "node": "spark-worker" + "node": "model-1" } ], "build": { diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/Chart.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/Chart.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/Chart.yaml rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/Chart.yaml diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/files/artifact-server.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/artifact-server.py similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/files/artifact-server.py rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/files/artifact-server.py diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/files/build.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/build.py similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/files/build.py rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/files/build.py diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/files/chain-check.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/chain-check.py similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/files/chain-check.py rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/files/chain-check.py diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/files/download.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/download.py similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/files/download.py rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/files/download.py diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/files/fetch-runtime.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/fetch-runtime.py similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/files/fetch-runtime.py rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/files/fetch-runtime.py diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/files/preflight.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/preflight.py similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/files/preflight.py rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/files/preflight.py diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/files/qualify.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/qualify.py similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/files/qualify.py rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/files/qualify.py diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/files/rpc-chain-check.cpp b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/rpc-chain-check.cpp similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/files/rpc-chain-check.cpp rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/files/rpc-chain-check.cpp diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/files/rpc-gpu-check.cpp b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/rpc-gpu-check.cpp similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/files/rpc-gpu-check.cpp rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/files/rpc-gpu-check.cpp diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/files/rpc-health.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/rpc-health.py similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/files/rpc-health.py rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/files/rpc-health.py diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/files/runtime_guard.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/runtime_guard.py similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/files/runtime_guard.py rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/files/runtime_guard.py diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/files/serve.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/serve.py similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/files/serve.py rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/files/serve.py diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/templates/build.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/build.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/templates/build.yaml rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/build.yaml diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/templates/chain.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/chain.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/templates/chain.yaml rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/chain.yaml diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/templates/download.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/download.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/templates/download.yaml rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/download.yaml diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/templates/preflight.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/preflight.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/templates/preflight.yaml rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/preflight.yaml diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/templates/qualify.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/qualify.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/templates/qualify.yaml rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/qualify.yaml diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/templates/runtime.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/runtime.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/templates/runtime.yaml rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/runtime.yaml diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/templates/serve.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/serve.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/templates/serve.yaml rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/serve.yaml diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/tests/test_download.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/tests/test_download.py similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/tests/test_download.py rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/tests/test_download.py diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/tests/test_rpc_health.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/tests/test_rpc_health.py similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/tests/test_rpc_health.py rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/tests/test_rpc_health.py diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/tests/test_runtime_guard.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/tests/test_runtime_guard.py similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/tests/test_runtime_guard.py rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/tests/test_runtime_guard.py diff --git a/deploy/helm/llm-routing/spark/charts/gguf-backend/values.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/values.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/charts/gguf-backend/values.yaml rename to deploy/helm/llm-routing/recipes/charts/gguf-backend/values.yaml diff --git a/deploy/helm/llm-routing/spark/charts/image-loader/Chart.yaml b/deploy/helm/llm-routing/recipes/charts/image-loader/Chart.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/charts/image-loader/Chart.yaml rename to deploy/helm/llm-routing/recipes/charts/image-loader/Chart.yaml diff --git a/deploy/helm/llm-routing/spark/charts/image-loader/templates/jobs.yaml b/deploy/helm/llm-routing/recipes/charts/image-loader/templates/jobs.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/charts/image-loader/templates/jobs.yaml rename to deploy/helm/llm-routing/recipes/charts/image-loader/templates/jobs.yaml diff --git a/deploy/helm/llm-routing/spark/charts/image-loader/values.yaml b/deploy/helm/llm-routing/recipes/charts/image-loader/values.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/charts/image-loader/values.yaml rename to deploy/helm/llm-routing/recipes/charts/image-loader/values.yaml diff --git a/deploy/helm/llm-routing/spark/client.py b/deploy/helm/llm-routing/recipes/client.py similarity index 100% rename from deploy/helm/llm-routing/spark/client.py rename to deploy/helm/llm-routing/recipes/client.py diff --git a/deploy/helm/llm-routing/spark/cluster_setup.py b/deploy/helm/llm-routing/recipes/cluster_setup.py similarity index 99% rename from deploy/helm/llm-routing/spark/cluster_setup.py rename to deploy/helm/llm-routing/recipes/cluster_setup.py index 6b6b5e41c..066516165 100644 --- a/deploy/helm/llm-routing/spark/cluster_setup.py +++ b/deploy/helm/llm-routing/recipes/cluster_setup.py @@ -1,6 +1,6 @@ # SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""Discover fresh Spark installation settings without changing the cluster.""" +"""Discover fresh recipe installation settings without changing the cluster.""" import datetime import json import pathlib diff --git a/deploy/helm/llm-routing/spark/config.example.json b/deploy/helm/llm-routing/recipes/config.example.json similarity index 69% rename from deploy/helm/llm-routing/spark/config.example.json rename to deploy/helm/llm-routing/recipes/config.example.json index 00899b104..c06af7fc2 100644 --- a/deploy/helm/llm-routing/spark/config.example.json +++ b/deploy/helm/llm-routing/recipes/config.example.json @@ -1,12 +1,12 @@ { - "context": "spark-demo", - "namespace": "llm-spark-poc", + "context": "llm-routing-demo", + "namespace": "llm-routing-poc", "releasePrefix": "llm-poc", - "clusterId": "spark-poc", + "clusterId": "llm-routing-poc", "nodes": { - "control": "spark-control", - "leader": "spark-leader", - "worker": "spark-worker" + "control": "control", + "leader": "model-0", + "worker": "model-1" }, "storageClass": "local-path", "runtimeClass": "nvidia", @@ -26,12 +26,12 @@ "apiKeyFile": null, "containerd": { "socketPath": "/run/k3s/containerd/containerd.sock", - "archiveNode": "spark-control", + "archiveNode": "control", "runAsUser": 1000, "nodeNames": [ - "spark-control", - "spark-leader", - "spark-worker" + "control", + "model-0", + "model-1" ] } } diff --git a/deploy/helm/llm-routing/spark/console_output.py b/deploy/helm/llm-routing/recipes/console_output.py similarity index 99% rename from deploy/helm/llm-routing/spark/console_output.py rename to deploy/helm/llm-routing/recipes/console_output.py index 5840bd254..fea90b6e0 100644 --- a/deploy/helm/llm-routing/spark/console_output.py +++ b/deploy/helm/llm-routing/recipes/console_output.py @@ -11,7 +11,7 @@ import tempfile import traceback -_ACTIVE = contextvars.ContextVar('spark_console', default=None) +_ACTIVE = contextvars.ContextVar('recipe_console', default=None) PAYLOAD_PHASES = {'chat', 'paths', 'context'} NEXT_CHECK = { 'init': 'Check the saved configuration and retained-resource ownership', diff --git a/deploy/helm/llm-routing/spark/gateway_access.py b/deploy/helm/llm-routing/recipes/gateway_access.py similarity index 99% rename from deploy/helm/llm-routing/spark/gateway_access.py rename to deploy/helm/llm-routing/recipes/gateway_access.py index 3728e39f6..73e9bf0a7 100644 --- a/deploy/helm/llm-routing/spark/gateway_access.py +++ b/deploy/helm/llm-routing/recipes/gateway_access.py @@ -214,7 +214,7 @@ def create(self): name, uid, original = self.discover() token = secrets.token_urlsafe(48) self.record = {'binding': self.binding, 'secret': name, 'secretUid': uid, 'originalData': original, - 'entry': {'id': 'spark-test-' + secrets.token_hex(16), + 'entry': {'id': 'recipe-test-' + secrets.token_hex(16), 'sha256': hashlib.sha256(token.encode()).hexdigest()}, 'createdAt': int(time.time())} _save(self.key_path, token + '\n') diff --git a/deploy/helm/llm-routing/spark/model.lock.json b/deploy/helm/llm-routing/recipes/model.lock.json similarity index 100% rename from deploy/helm/llm-routing/spark/model.lock.json rename to deploy/helm/llm-routing/recipes/model.lock.json diff --git a/deploy/helm/llm-routing/spark/operator.Dockerfile b/deploy/helm/llm-routing/recipes/operator.Dockerfile similarity index 100% rename from deploy/helm/llm-routing/spark/operator.Dockerfile rename to deploy/helm/llm-routing/recipes/operator.Dockerfile diff --git a/deploy/helm/llm-routing/spark/spark.py b/deploy/helm/llm-routing/recipes/recipe.py similarity index 98% rename from deploy/helm/llm-routing/spark/spark.py rename to deploy/helm/llm-routing/recipes/recipe.py index ff90fdc5d..da1e341a1 100644 --- a/deploy/helm/llm-routing/spark/spark.py +++ b/deploy/helm/llm-routing/recipes/recipe.py @@ -1,7 +1,7 @@ #!/usr/bin/env python3 # SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""Explicit phases for the two-GPU Spark recipe. See README.md first.""" +"""Explicit phases for the LLM routing recipes. See README.md first.""" import argparse import base64 import contextlib @@ -72,7 +72,7 @@ def validate(c): def default_work_dir(context): require(isinstance(context, str) and bool(context.strip()), - 'Set SPARK_CONTEXT or --context. The current kubectl context is not selected automatically.') + 'Set LLM_ROUTING_CONTEXT or --context. The current kubectl context is not selected automatically.') xdg = os.environ.get('XDG_STATE_HOME') if xdg: root = pathlib.Path(xdg) @@ -103,7 +103,7 @@ def kubeconfig_context(): if not names: raise ContextSelectionError('No Kubernetes context found. Set KUBECONFIG or pass --context NAME.') if len(names) != 1: - raise ContextSelectionError('Multiple Kubernetes contexts found. Pass --context NAME or set SPARK_CONTEXT.') + raise ContextSelectionError('Multiple Kubernetes contexts found. Pass --context NAME or set LLM_ROUTING_CONTEXT.') return names.pop() @@ -118,7 +118,7 @@ def cli_settings(args): work = args.work_dir.expanduser().resolve() if args.work_dir else None if config is None and work and not args.config and (work/'config.json').exists(): config = json.loads((work/'config.json').read_text()) - context = args.context or os.environ.get('SPARK_CONTEXT') or (config.get('context') if config else None) + context = args.context or os.environ.get('LLM_ROUTING_CONTEXT') or (config.get('context') if config else None) if not context: context = kubeconfig_context() work = work or default_work_dir(context) @@ -130,7 +130,7 @@ def cli_settings(args): require(config.get('context') == context, 'Selected context differs from the saved deployment configuration.') require(not args.namespace or config.get('namespace') == args.namespace, 'Selected namespace differs from the saved deployment. Use a separate --work-dir for another installation.') - require(isinstance(context, str) and bool(context.strip()), 'Set --context NAME or SPARK_CONTEXT.') + require(isinstance(context, str) and bool(context.strip()), 'Set --context NAME or LLM_ROUTING_CONTEXT.') return context, work, config_path, config @@ -363,7 +363,7 @@ def operator_values(self): def stack_values(self, token_hash, key_hash, ui_key=None): values = {'clusterId': self.c['clusterId'], 'clusterCredential': {'sha256': token_hash}, 'apiKeys': [{'id': 'poc-client', 'sha256': key_hash}], 'tls': copy.deepcopy(self.c['tls']), - 'sparkRecipeSource': self.source_identity(), 'sparkRecipeChartsSha256': self.chart_digest()} + 'recipeSource': self.source_identity(), 'recipeChartsSha256': self.chart_digest()} if ui_key is not None: values['apiKeys'].append({'id': 'demo-ui', 'sha256': hashlib.sha256(ui_key.encode()).hexdigest()}) values['demoUiApiKey'] = ui_key @@ -460,10 +460,10 @@ def attach_existing(self): save(self.work/'ca.crt', ca) self.stamp('attachedExisting') self.stamp('inventory', {'nodes': {n['metadata']['name']: n['metadata']['uid'] for n in nodes}}) - self.stamp('stack', {'apiKeyFile': str(key) if key else None, 'source': values.get('sparkRecipeSource')}) + self.stamp('stack', {'apiKeyFile': str(key) if key else None, 'source': values.get('recipeSource')}) self.stamp('serve') print('Existing installation inspected. Run verify-gateway next.') - if values.get('sparkRecipeChartsSha256') != self.chart_digest(): + if values.get('recipeChartsSha256') != self.chart_digest(): console_output.warn('Image updates require matching recorded charts. Use a coordinated stack installation first.') def bound_cluster(self): @@ -923,7 +923,7 @@ def update(self, component, tag): chart, service = {'gateway': ('llm-api-gateway', 'llmApiGateway'), 'router': ('llm-request-router', 'llmRequestRouter')}[component] require(re.fullmatch(r'[A-Za-z0-9_][A-Za-z0-9_.-]{0,127}', tag) is not None, 'Invalid image tag.') values = json.loads(output(self.hm+['get', 'values', self.stack, '-o', 'json'])) - require(values.get('sparkRecipeChartsSha256') == self.chart_digest(), + require(values.get('recipeChartsSha256') == self.chart_digest(), 'Installed chart fingerprint is missing or differs from this checkout. Use a coordinated stack installation before image updates.') current = values[chart][service]['image'] require(current['registry']+'/'+current['repository'] == self.repository(component), 'Live image repository differs from config.') @@ -1015,7 +1015,7 @@ def render(self): def main(argv=None, console=None): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('--config', type=pathlib.Path) - parser.add_argument('--context', help='Kubernetes context; defaults to SPARK_CONTEXT, saved settings, or the sole kubeconfig context.') + parser.add_argument('--context', help='Kubernetes context; defaults to LLM_ROUTING_CONTEXT, saved settings, or the sole kubeconfig context.') parser.add_argument('--namespace', help='Select the namespace when the cluster has multiple installations.') parser.add_argument('--work-dir', type=pathlib.Path, help='Private local state directory; defaults to a per-context directory.') parser.add_argument('--source-dir', type=pathlib.Path, help='Existing source checkout; defaults to the checkout containing this script.') diff --git a/deploy/helm/llm-routing/spark/tests/sample-backend/Chart.yaml b/deploy/helm/llm-routing/recipes/tests/sample-backend/Chart.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/tests/sample-backend/Chart.yaml rename to deploy/helm/llm-routing/recipes/tests/sample-backend/Chart.yaml diff --git a/deploy/helm/llm-routing/spark/tests/sample-backend/templates/backend.yaml b/deploy/helm/llm-routing/recipes/tests/sample-backend/templates/backend.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/tests/sample-backend/templates/backend.yaml rename to deploy/helm/llm-routing/recipes/tests/sample-backend/templates/backend.yaml diff --git a/deploy/helm/llm-routing/spark/tests/sample-backend/templates/endpoint.yaml b/deploy/helm/llm-routing/recipes/tests/sample-backend/templates/endpoint.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/tests/sample-backend/templates/endpoint.yaml rename to deploy/helm/llm-routing/recipes/tests/sample-backend/templates/endpoint.yaml diff --git a/deploy/helm/llm-routing/spark/tests/sample-backend/values.yaml b/deploy/helm/llm-routing/recipes/tests/sample-backend/values.yaml similarity index 100% rename from deploy/helm/llm-routing/spark/tests/sample-backend/values.yaml rename to deploy/helm/llm-routing/recipes/tests/sample-backend/values.yaml diff --git a/deploy/helm/llm-routing/spark/tests/test_attach_reuse.py b/deploy/helm/llm-routing/recipes/tests/test_attach_reuse.py similarity index 87% rename from deploy/helm/llm-routing/spark/tests/test_attach_reuse.py rename to deploy/helm/llm-routing/recipes/tests/test_attach_reuse.py index 1b6d29aca..1ff30265d 100644 --- a/deploy/helm/llm-routing/spark/tests/test_attach_reuse.py +++ b/deploy/helm/llm-routing/recipes/tests/test_attach_reuse.py @@ -11,14 +11,14 @@ from unittest.mock import patch HERE = pathlib.Path(__file__).resolve().parents[1] -spec = importlib.util.spec_from_file_location('spark_attach_reuse', HERE/'spark.py') -spark = importlib.util.module_from_spec(spec) -spec.loader.exec_module(spark) +spec = importlib.util.spec_from_file_location('recipe_attach_reuse', HERE/'recipe.py') +tool = importlib.util.module_from_spec(spec) +spec.loader.exec_module(tool) class AttachReuseTests(unittest.TestCase): def setUp(self): - self.tmp = tempfile.TemporaryDirectory(prefix='spark-attach-reuse-') + self.tmp = tempfile.TemporaryDirectory(prefix='recipe-attach-reuse-') self.addCleanup(self.tmp.cleanup) self.count = 0 @@ -28,8 +28,8 @@ def installation(self, explicit=False): if explicit: config['releases'] = {'stack': 'custom-front', 'operator': 'custom-operator', 'glm': 'custom-model'} config['images']['repositories'] = { - name: 'registry.example.com/custom/'+name for name in spark.COMPONENTS} - recipe = spark.Recipe(config, pathlib.Path(self.tmp.name)/str(self.count)) + name: 'registry.example.com/custom/'+name for name in tool.COMPONENTS} + recipe = tool.Recipe(config, pathlib.Path(self.tmp.name)/str(self.count)) key = recipe.work/'api-key' key.write_text('test-only-key\n') recipe.state = { @@ -40,11 +40,11 @@ def installation(self, explicit=False): 'serve': True, 'registered': True, 'direct': True, 'gateway': True, 'lastUpdate': {'component': 'gateway', 'previousTag': 'original', 'newTag': 'changed'}, } - spark.save(recipe.work/'config.json', config) - spark.save(recipe.state_path, recipe.state) + tool.save(recipe.work/'config.json', config) + tool.save(recipe.state_path, recipe.state) live = copy.deepcopy(config) live['releases'] = {'stack': recipe.stack, 'operator': recipe.operator, 'glm': recipe.glm} - live['images']['repositories'] = {name: recipe.repository(name) for name in spark.COMPONENTS} + live['images']['repositories'] = {name: recipe.repository(name) for name in tool.COMPONENTS} live['apiKeyFile'] = None return recipe, live @@ -54,9 +54,9 @@ def attempt(self, recipe, live, *, rejected=False, discovery_error=None): before_files = {str(p.relative_to(recipe.work)): p.read_bytes() for p in recipe.work.rglob('*') if p.is_file()} with patch.object(recipe, 'bound_cluster') as bound, \ - patch.object(spark, 'discover_config', return_value=copy.deepcopy(live), side_effect=discovery_error) as discover, \ - patch.object(spark, 'save') as save, patch.object(spark, 'run') as run, \ - patch.object(spark, 'output') as output, contextlib.redirect_stdout(io.StringIO()): + patch.object(tool, 'discover_config', return_value=copy.deepcopy(live), side_effect=discovery_error) as discover, \ + patch.object(tool, 'save') as save, patch.object(tool, 'run') as run, \ + patch.object(tool, 'output') as output, contextlib.redirect_stdout(io.StringIO()): if rejected: with self.assertRaises(RuntimeError): recipe.attach_existing() @@ -133,7 +133,7 @@ def test_effective_release_mismatch_is_rejected(self): self.attempt(recipe, live, rejected=True) def test_each_component_repository_is_checked(self): - for component in spark.COMPONENTS: + for component in tool.COMPONENTS: with self.subTest(component=component): recipe, live = self.installation() live['images']['repositories'][component] = 'registry.example.com/foreign/'+component @@ -150,8 +150,8 @@ def test_changed_node_uid_is_rejected_before_any_rediscovery(self): before = recipe.state_path.read_bytes() nodes = {'items': [{'metadata': {'name': name, 'uid': name+'-replacement'}} for name in recipe.c['nodes'].values()]} - with patch.object(spark, 'output', return_value=json.dumps(nodes)) as output, \ - patch.object(spark, 'discover_config') as discover, patch.object(spark, 'save') as save: + with patch.object(tool, 'output', return_value=json.dumps(nodes)) as output, \ + patch.object(tool, 'discover_config') as discover, patch.object(tool, 'save') as save: with self.assertRaisesRegex(RuntimeError, 'identities changed|another cluster'): recipe.attach_existing() output.assert_called_once_with(recipe.kc+['get', 'nodes', '-o', 'json']) diff --git a/deploy/helm/llm-routing/spark/tests/test_chat.py b/deploy/helm/llm-routing/recipes/tests/test_chat.py similarity index 88% rename from deploy/helm/llm-routing/spark/tests/test_chat.py rename to deploy/helm/llm-routing/recipes/tests/test_chat.py index e013c0613..69674bff7 100644 --- a/deploy/helm/llm-routing/spark/tests/test_chat.py +++ b/deploy/helm/llm-routing/recipes/tests/test_chat.py @@ -10,9 +10,9 @@ from unittest.mock import patch HERE = pathlib.Path(__file__).resolve().parents[1] -spec = importlib.util.spec_from_file_location('chat_recipe', HERE/'spark.py') -spark = importlib.util.module_from_spec(spec) -spec.loader.exec_module(spark) +spec = importlib.util.spec_from_file_location('chat_recipe', HERE/'recipe.py') +tool = importlib.util.module_from_spec(spec) +spec.loader.exec_module(tool) class ChatTests(unittest.TestCase): @@ -20,7 +20,7 @@ def setUp(self): self.tmp = tempfile.TemporaryDirectory() self.addCleanup(self.tmp.cleanup) config = json.loads((HERE/'config.example.json').read_text()) - self.recipe = spark.Recipe(config, self.tmp.name) + self.recipe = tool.Recipe(config, self.tmp.name) self.recipe.state = {'stack': {'apiKeyFile': None}, 'gateway': False} self.events = [] @@ -51,8 +51,8 @@ def test_custom_and_streaming_chat_use_temporary_key_then_revoke(self): before = copy.deepcopy(self.recipe.state) with self.subTest(streaming=streaming), patch.object(self.recipe, 'bound_cluster') as bound, \ patch.object(self.recipe, 'forward', self.forward), \ - patch.object(spark.gateway_access, 'temporary_gateway_key', self.key), \ - patch.object(spark, 'run') as run: + patch.object(tool.gateway_access, 'temporary_gateway_key', self.key), \ + patch.object(tool, 'run') as run: self.recipe.chat('How many planets are in the solar system?', streaming, 18443) bound.assert_called_once() command = run.call_args.args[0] @@ -69,7 +69,7 @@ def test_custom_and_streaming_chat_use_temporary_key_then_revoke(self): def test_existing_installer_key_is_supported_without_mutating_credentials(self): self.recipe.state['stack']['apiKeyFile'] = '/private/existing-key' with patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'forward', self.forward), \ - patch.object(spark.gateway_access, 'temporary_gateway_key') as temporary, patch.object(spark, 'run') as run: + patch.object(tool.gateway_access, 'temporary_gateway_key') as temporary, patch.object(tool, 'run') as run: self.recipe.chat('Hello', False, 18443) temporary.assert_not_called() command = run.call_args.args[0] @@ -81,8 +81,8 @@ def test_request_failure_and_interrupt_revoke_before_closing_tunnel(self): self.events = [] with self.subTest(error=type(error).__name__), patch.object(self.recipe, 'bound_cluster'), \ patch.object(self.recipe, 'forward', self.forward), \ - patch.object(spark.gateway_access, 'temporary_gateway_key', self.key), \ - patch.object(spark, 'run', side_effect=error), self.assertRaises(type(error)): + patch.object(tool.gateway_access, 'temporary_gateway_key', self.key), \ + patch.object(tool, 'run', side_effect=error), self.assertRaises(type(error)): self.recipe.chat('Hello', False, 18443) self.assertEqual(self.events, ['connected', 'registered', 'revoked', 'disconnected']) self.assertFalse(self.recipe.state['gateway']) @@ -90,7 +90,7 @@ def test_request_failure_and_interrupt_revoke_before_closing_tunnel(self): def test_changed_cluster_rejects_chat_before_connecting_or_issuing_key(self): with patch.object(self.recipe, 'bound_cluster', side_effect=RuntimeError('cluster changed')), \ patch.object(self.recipe, 'forward') as forward, \ - patch.object(spark.gateway_access, 'temporary_gateway_key') as key, \ + patch.object(tool.gateway_access, 'temporary_gateway_key') as key, \ self.assertRaisesRegex(RuntimeError, 'cluster changed'): self.recipe.chat('Hello', False, 18443) forward.assert_not_called() @@ -107,7 +107,7 @@ def test_prompts_are_literal_arguments_not_options_or_shell(self): self.recipe.state['stack']['apiKeyFile'] = '/private/existing-key' prompt = '--stream $(echo example) `echo literal`' with patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'forward', self.forward), \ - patch.object(spark, 'run') as run: + patch.object(tool, 'run') as run: self.recipe.chat(prompt, False, 18443) self.assertEqual(run.call_args.args[0][-2:], ['--', prompt]) self.assertNotIn('shell', run.call_args.kwargs) @@ -115,7 +115,7 @@ def test_prompts_are_literal_arguments_not_options_or_shell(self): def test_omitted_prompt_uses_client_default(self): self.recipe.state['stack']['apiKeyFile'] = '/private/existing-key' with patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'forward', self.forward), \ - patch.object(spark, 'run') as run: + patch.object(tool, 'run') as run: self.recipe.chat(None, False, 18443) self.assertEqual(run.call_args.args[0][-2:], ['--api-key-file', '/private/existing-key']) diff --git a/deploy/helm/llm-routing/spark/tests/test_cli.py b/deploy/helm/llm-routing/recipes/tests/test_cli.py similarity index 67% rename from deploy/helm/llm-routing/spark/tests/test_cli.py rename to deploy/helm/llm-routing/recipes/tests/test_cli.py index 752941b26..ba0f4dd0d 100644 --- a/deploy/helm/llm-routing/spark/tests/test_cli.py +++ b/deploy/helm/llm-routing/recipes/tests/test_cli.py @@ -13,9 +13,9 @@ from unittest.mock import patch HERE = pathlib.Path(__file__).resolve().parents[1] -spec = importlib.util.spec_from_file_location('cli_recipe', HERE/'spark.py') -spark = importlib.util.module_from_spec(spec) -spec.loader.exec_module(spark) +spec = importlib.util.spec_from_file_location('cli_recipe', HERE/'recipe.py') +tool = importlib.util.module_from_spec(spec) +spec.loader.exec_module(tool) class CliTests(unittest.TestCase): @@ -25,17 +25,17 @@ def setUp(self): self.root = pathlib.Path(self.tmp.name).resolve() self.home = self.root/'home' self.home.mkdir() - self.env = patch.dict(os.environ, {'HOME': str(self.home), 'SPARK_CONTEXT': 'team-context', 'XDG_STATE_HOME': ''}) + self.env = patch.dict(os.environ, {'HOME': str(self.home), 'LLM_ROUTING_CONTEXT': 'team-context', 'XDG_STATE_HOME': ''}) self.env.start() self.addCleanup(self.env.stop) self.config = json.loads((HERE/'config.example.json').read_text()) self.config['context'] = 'team-context' self.config['releases'] = {'stack': 'test-stack', 'operator': 'test-operator', 'glm': 'test-glm'} - self.work = spark.default_work_dir('team-context') + self.work = tool.default_work_dir('team-context') def write_config(self, config=None, path=None): path = path or self.work/'config.json' - spark.save(path, config or self.config) + tool.save(path, config or self.config) return path def test_default_state_is_private_path_outside_checkout(self): @@ -45,26 +45,26 @@ def test_default_state_is_private_path_outside_checkout(self): self.assertFalse(self.work.is_relative_to(HERE.parents[3])) def test_context_names_cannot_escape_storage_and_contexts_are_isolated(self): - other = spark.default_work_dir('../../another context') + other = tool.default_work_dir('../../another context') self.assertNotEqual(other, self.work) self.assertEqual(other.parent, self.work.parent) self.assertRegex(other.name, r'^[0-9a-f]{20}$') def test_absolute_xdg_state_home_is_used(self): with patch.dict(os.environ, {'XDG_STATE_HOME': str(self.root/'state')}): - result = spark.default_work_dir('team-context') + result = tool.default_work_dir('team-context') self.assertEqual(result.parent, self.root/'state/nvcf/llm-routing') def test_relative_xdg_is_rejected(self): with patch.dict(os.environ, {'XDG_STATE_HOME': 'relative'}), self.assertRaisesRegex(RuntimeError, 'absolute'): - spark.main(['paths']) + tool.main(['paths']) self.assertFalse(self.work.exists()) def test_empty_kubeconfig_fails_with_actionable_error_without_traceback(self): stream = io.StringIO() - with patch.dict(os.environ, {'SPARK_CONTEXT': ''}), patch.object(spark, 'output', return_value='{}') as output, \ + with patch.dict(os.environ, {'LLM_ROUTING_CONTEXT': ''}), patch.object(tool, 'output', return_value='{}') as output, \ redirect_stderr(stream), self.assertRaises(SystemExit) as result: - spark.main(['attach-existing']) + tool.main(['attach-existing']) self.assertEqual(result.exception.code, 2) self.assertEqual(output.call_args.args[0], ['kubectl', 'config', 'view', '-o', 'json']) self.assertIn('No Kubernetes context found. Set KUBECONFIG', stream.getvalue()) @@ -73,13 +73,13 @@ def test_empty_kubeconfig_fails_with_actionable_error_without_traceback(self): def test_sole_kubeconfig_context_supports_attach_and_verify_without_environment_context(self): local = {'contexts': [{'name': 'team-context'}], 'current-context': 'unrelated'} - with patch.dict(os.environ, {'SPARK_CONTEXT': '', 'KUBECONFIG': str(self.root/'team-kubeconfig')}), \ - patch.object(spark, 'output', return_value=json.dumps(local)) as output, \ - patch.object(spark, 'discover_config', return_value=self.config) as discover, \ - patch.object(spark.Recipe, 'attach_existing'), patch.object(spark.Recipe, 'verify') as verify, \ + with patch.dict(os.environ, {'LLM_ROUTING_CONTEXT': '', 'KUBECONFIG': str(self.root/'team-kubeconfig')}), \ + patch.object(tool, 'output', return_value=json.dumps(local)) as output, \ + patch.object(tool, 'discover_config', return_value=self.config) as discover, \ + patch.object(tool.Recipe, 'attach_existing'), patch.object(tool.Recipe, 'verify') as verify, \ redirect_stdout(io.StringIO()): - spark.main(['attach-existing']) - spark.main(['verify-gateway']) + tool.main(['attach-existing']) + tool.main(['verify-gateway']) discover.assert_called_once_with('team-context', None) verify.assert_called_once_with(True, 18443) self.assertEqual(output.call_count, 2) @@ -90,10 +90,10 @@ def test_sole_kubeconfig_context_supports_attach_and_verify_without_environment_ def test_multiple_contexts_never_choose_current_context_even_with_explicit_kubeconfig(self): local = {'contexts': [{'name': 'team-context'}, {'name': 'unrelated'}], 'current-context': 'unrelated'} for kubeconfig in ('', str(self.root/'team-kubeconfig')): - with self.subTest(kubeconfig=kubeconfig), patch.dict(os.environ, {'SPARK_CONTEXT': '', 'KUBECONFIG': kubeconfig}), \ - patch.object(spark, 'output', return_value=json.dumps(local)), patch.object(spark, 'discover_config') as discover, \ + with self.subTest(kubeconfig=kubeconfig), patch.dict(os.environ, {'LLM_ROUTING_CONTEXT': '', 'KUBECONFIG': kubeconfig}), \ + patch.object(tool, 'output', return_value=json.dumps(local)), patch.object(tool, 'discover_config') as discover, \ redirect_stderr(io.StringIO()) as errors, self.assertRaises(SystemExit) as result: - spark.main(['attach-existing']) + tool.main(['attach-existing']) self.assertEqual(result.exception.code, 2) self.assertIn('Multiple Kubernetes contexts found. Pass --context NAME', errors.getvalue()) self.assertNotIn('Traceback', errors.getvalue()) @@ -110,57 +110,57 @@ def test_explicit_context_sources_precede_kubeconfig_discovery(self): ('', ['--work-dir', str(self.work), 'paths']), ] for context, arguments in scenarios: - with self.subTest(arguments=arguments), patch.dict(os.environ, {'SPARK_CONTEXT': context}), \ - patch.object(spark, 'output') as output, redirect_stdout(io.StringIO()): - spark.main(arguments) + with self.subTest(arguments=arguments), patch.dict(os.environ, {'LLM_ROUTING_CONTEXT': context}), \ + patch.object(tool, 'output') as output, redirect_stdout(io.StringIO()): + tool.main(arguments) output.assert_not_called() def test_context_discovery_does_not_print_credentials(self): local = {'contexts': [{'name': 'team-context'}], 'users': [{'name': 'user', 'user': {'token': 'sensitive-fixture'}}]} - with patch.dict(os.environ, {'SPARK_CONTEXT': ''}), patch.object(spark, 'output', return_value=json.dumps(local)) as output, \ + with patch.dict(os.environ, {'LLM_ROUTING_CONTEXT': ''}), patch.object(tool, 'output', return_value=json.dumps(local)) as output, \ redirect_stdout(io.StringIO()) as stream: - spark.main(['paths']) + tool.main(['paths']) self.assertNotIn('sensitive-fixture', stream.getvalue()) self.assertNotIn('--raw', output.call_args.args[0]) self.assertFalse(self.work.exists()) def test_unreadable_kubeconfig_reports_only_setup_error(self): for failure in (FileNotFoundError('kubectl missing'), ValueError('sensitive parse text')): - with self.subTest(failure=failure), patch.dict(os.environ, {'SPARK_CONTEXT': ''}), \ - patch.object(spark, 'output', side_effect=failure), redirect_stderr(io.StringIO()) as stream, \ + with self.subTest(failure=failure), patch.dict(os.environ, {'LLM_ROUTING_CONTEXT': ''}), \ + patch.object(tool, 'output', side_effect=failure), redirect_stderr(io.StringIO()) as stream, \ self.assertRaises(SystemExit) as result: - spark.main(['paths']) + tool.main(['paths']) self.assertEqual(result.exception.code, 2) self.assertEqual(stream.getvalue(), 'error: Could not read kubeconfig. Set KUBECONFIG or pass --context NAME.\n') def test_discovered_context_still_rejects_saved_identity_mismatch(self): self.write_config(dict(self.config, context='other-context')) local = {'contexts': [{'name': 'team-context'}]} - with patch.dict(os.environ, {'SPARK_CONTEXT': ''}), patch.object(spark, 'output', return_value=json.dumps(local)), \ - patch.object(spark.Recipe, 'verify') as verify, self.assertRaisesRegex(RuntimeError, 'Selected context differs'): - spark.main(['verify-gateway']) + with patch.dict(os.environ, {'LLM_ROUTING_CONTEXT': ''}), patch.object(tool, 'output', return_value=json.dumps(local)), \ + patch.object(tool.Recipe, 'verify') as verify, self.assertRaisesRegex(RuntimeError, 'Selected context differs'): + tool.main(['verify-gateway']) verify.assert_not_called() def test_context_prints_only_selected_name_without_writing_state(self): - with redirect_stdout(io.StringIO()) as result, patch.object(spark, 'Recipe') as recipe, patch.object(spark, 'output') as output: - spark.main(['context']) + with redirect_stdout(io.StringIO()) as result, patch.object(tool, 'Recipe') as recipe, patch.object(tool, 'output') as output: + tool.main(['context']) self.assertEqual(result.getvalue(), 'team-context\n') recipe.assert_not_called() output.assert_not_called() self.assertFalse(self.work.exists()) def test_context_can_resolve_sole_kubeconfig_name_without_export(self): - with patch.dict(os.environ, {'SPARK_CONTEXT': ''}), \ - patch.object(spark, 'output', return_value='{"contexts": [{"name": "team-context"}]}'), \ + with patch.dict(os.environ, {'LLM_ROUTING_CONTEXT': ''}), \ + patch.object(tool, 'output', return_value='{"contexts": [{"name": "team-context"}]}'), \ redirect_stdout(io.StringIO()) as result: - spark.main(['context']) + tool.main(['context']) self.assertEqual(result.getvalue(), 'team-context\n') self.assertFalse(self.work.exists()) def test_paths_is_read_only_and_has_no_credentials(self): stream = io.StringIO() - with redirect_stdout(stream), patch.object(spark, 'Recipe') as recipe, patch.object(spark, 'output') as output: - spark.main(['paths']) + with redirect_stdout(stream), patch.object(tool, 'Recipe') as recipe, patch.object(tool, 'output') as output: + tool.main(['paths']) paths = json.loads(stream.getvalue()) self.assertEqual(paths, {'workDir': str(self.work), 'config': str(self.work/'config.json'), 'source': str(HERE.parents[3])}) recipe.assert_not_called() @@ -169,8 +169,8 @@ def test_paths_is_read_only_and_has_no_credentials(self): def test_explicit_source_directory_overrides_the_recipe_checkout(self): source = self.root/'another-checkout' - with redirect_stdout(io.StringIO()) as stream, patch.object(spark, 'output') as output: - spark.main(['paths', '--source-dir', str(source)]) + with redirect_stdout(io.StringIO()) as stream, patch.object(tool, 'output') as output: + tool.main(['paths', '--source-dir', str(source)]) self.assertEqual(json.loads(stream.getvalue())['source'], str(source)) output.assert_not_called() self.assertFalse(self.work.exists()) @@ -183,11 +183,11 @@ def attach(recipe): recipe.stamp('attachedExisting') def verify(recipe, gateway, port): seen.append(('verify', recipe.work, gateway, port)) - with patch.object(spark, 'discover_config', return_value=self.config) as discover, \ - patch.object(spark.Recipe, 'attach_existing', attach), patch.object(spark.Recipe, 'verify', verify), \ - patch.object(spark.Recipe, 'prepare') as prepare, redirect_stdout(io.StringIO()): - spark.main(['attach-existing']) - spark.main(['verify-gateway']) + with patch.object(tool, 'discover_config', return_value=self.config) as discover, \ + patch.object(tool.Recipe, 'attach_existing', attach), patch.object(tool.Recipe, 'verify', verify), \ + patch.object(tool.Recipe, 'prepare') as prepare, redirect_stdout(io.StringIO()): + tool.main(['attach-existing']) + tool.main(['verify-gateway']) discover.assert_called_once_with('team-context', None) prepare.assert_not_called() self.assertEqual(seen, [('attach', self.work), ('verify', self.work, True, 18443)]) @@ -197,82 +197,82 @@ def verify(recipe, gateway, port): def test_repeated_attach_reuses_config_instead_of_rediscovery(self): self.write_config() - with patch.object(spark, 'discover_config') as discover, patch.object(spark.Recipe, 'attach_existing') as attach, redirect_stdout(io.StringIO()): - spark.main(['attach-existing']) + with patch.object(tool, 'discover_config') as discover, patch.object(tool.Recipe, 'attach_existing') as attach, redirect_stdout(io.StringIO()): + tool.main(['attach-existing']) discover.assert_not_called() attach.assert_called_once() def test_unavailable_docker_reports_one_actionable_line_without_traceback(self): self.write_config() - with patch.object(spark.Recipe, 'source_check'), \ - patch.object(spark, 'run', side_effect=FileNotFoundError('missing Docker socket')) as run, \ + with patch.object(tool.Recipe, 'source_check'), \ + patch.object(tool, 'run', side_effect=FileNotFoundError('missing Docker socket')) as run, \ redirect_stderr(io.StringIO()) as errors, self.assertRaises(SystemExit) as result: - spark.main(['build-images', '--component', 'gateway']) + tool.main(['build-images', '--component', 'gateway']) self.assertEqual(result.exception.code, 2) self.assertEqual(errors.getvalue(), 'error: Start Docker, then rerun build-images.\n') self.assertEqual(run.call_count, 1) def test_prepare_uses_default_saved_config(self): self.write_config() - with patch.object(spark.Recipe, 'prepare') as prepare: - spark.main(['prepare']) + with patch.object(tool.Recipe, 'prepare') as prepare: + tool.main(['prepare']) prepare.assert_called_once() def test_missing_default_config_does_not_silently_attach_for_verification(self): - with patch.object(spark, 'discover_config') as discover, self.assertRaisesRegex(RuntimeError, 'attach-existing first'): - spark.main(['verify-gateway']) + with patch.object(tool, 'discover_config') as discover, self.assertRaisesRegex(RuntimeError, 'attach-existing first'): + tool.main(['verify-gateway']) discover.assert_not_called() self.assertFalse(self.work.exists()) def test_selected_context_must_match_saved_config(self): other = dict(self.config, context='other') self.write_config(other) - with patch.object(spark.Recipe, 'verify') as verify, self.assertRaisesRegex(RuntimeError, 'Selected context differs'): - spark.main(['verify-gateway']) + with patch.object(tool.Recipe, 'verify') as verify, self.assertRaisesRegex(RuntimeError, 'Selected context differs'): + tool.main(['verify-gateway']) verify.assert_not_called() def test_namespace_mismatch_cannot_retarget_saved_installation(self): self.write_config() - with patch.object(spark, 'discover_config') as discover, self.assertRaisesRegex(RuntimeError, 'Selected namespace differs'): - spark.main(['attach-existing', '--namespace', 'another']) + with patch.object(tool, 'discover_config') as discover, self.assertRaisesRegex(RuntimeError, 'Selected namespace differs'): + tool.main(['attach-existing', '--namespace', 'another']) discover.assert_not_called() self.assertEqual(json.loads((self.work/'config.json').read_text()), self.config) def test_legacy_config_and_work_flags_need_no_context_environment(self): path = self.write_config(path=self.root/'legacy.json') - with patch.dict(os.environ, {'SPARK_CONTEXT': ''}), patch.object(spark.Recipe, 'prepare') as prepare: - spark.main(['--config', str(path), '--work-dir', str(self.root/'legacy-work'), 'prepare']) + with patch.dict(os.environ, {'LLM_ROUTING_CONTEXT': ''}), patch.object(tool.Recipe, 'prepare') as prepare: + tool.main(['--config', str(path), '--work-dir', str(self.root/'legacy-work'), 'prepare']) prepare.assert_called_once() def test_explicit_work_reuses_its_saved_context_when_no_context_env(self): self.write_config() - with patch.dict(os.environ, {'SPARK_CONTEXT': ''}), patch.object(spark.Recipe, 'verify') as verify: - spark.main(['--work-dir', str(self.work), 'verify-gateway']) + with patch.dict(os.environ, {'LLM_ROUTING_CONTEXT': ''}), patch.object(tool.Recipe, 'verify') as verify: + tool.main(['--work-dir', str(self.work), 'verify-gateway']) verify.assert_called_once_with(True, 18443) def test_explicit_context_overrides_environment_but_must_match_config(self): path = self.write_config(path=self.root/'explicit.json') - with patch.dict(os.environ, {'SPARK_CONTEXT': 'another'}), patch.object(spark.Recipe, 'prepare') as prepare: - spark.main(['--context', 'team-context', '--config', str(path), 'prepare']) + with patch.dict(os.environ, {'LLM_ROUTING_CONTEXT': 'another'}), patch.object(tool.Recipe, 'prepare') as prepare: + tool.main(['--context', 'team-context', '--config', str(path), 'prepare']) prepare.assert_called_once() with self.assertRaisesRegex(RuntimeError, 'Selected context differs'): - spark.main(['--context', 'another', '--config', str(path), 'prepare']) + tool.main(['--context', 'another', '--config', str(path), 'prepare']) def test_missing_config_with_existing_state_is_not_overwritten(self): - spark.save(self.work/'state.json', {'identity': 'original'}) - with patch.object(spark, 'discover_config') as discover, self.assertRaisesRegex(RuntimeError, 'missing its configuration'): - spark.main(['attach-existing']) + tool.save(self.work/'state.json', {'identity': 'original'}) + with patch.object(tool, 'discover_config') as discover, self.assertRaisesRegex(RuntimeError, 'missing its configuration'): + tool.main(['attach-existing']) discover.assert_not_called() self.assertEqual(json.loads((self.work/'state.json').read_text()), {'identity': 'original'}) def test_export_uses_configured_repository_and_atomic_private_archive(self): self.config['images']['repositories'] = {'gateway': 'registry.example.com/own/gateway'} - recipe = spark.Recipe(self.config, self.work) + recipe = tool.Recipe(self.config, self.work) def save_image(command): self.assertEqual(command[:3], ['docker', 'save', '--output']) self.assertEqual(command[-1], 'registry.example.com/own/gateway:edited') pathlib.Path(command[3]).write_text('image archive') - with patch.object(recipe, 'source_check') as source, patch.object(spark, 'run', side_effect=save_image), redirect_stdout(io.StringIO()): + with patch.object(recipe, 'source_check') as source, patch.object(tool, 'run', side_effect=save_image), redirect_stdout(io.StringIO()): recipe.export_images('gateway', 'edited') source.assert_called_once() self.assertEqual((self.work/'arm64-images.tar').read_text(), 'image archive') @@ -280,9 +280,9 @@ def save_image(command): self.assertEqual(list(self.work.glob('.arm64-images-*')), []) def test_failed_export_preserves_previous_archive(self): - recipe = spark.Recipe(self.config, self.work) + recipe = tool.Recipe(self.config, self.work) (self.work/'arm64-images.tar').write_text('original') - with patch.object(recipe, 'source_check'), patch.object(spark, 'run', side_effect=RuntimeError('docker failed')): + with patch.object(recipe, 'source_check'), patch.object(tool, 'run', side_effect=RuntimeError('docker failed')): with self.assertRaisesRegex(RuntimeError, 'docker failed'): recipe.export_images('router', 'edited') self.assertEqual((self.work/'arm64-images.tar').read_text(), 'original') @@ -290,34 +290,34 @@ def test_failed_export_preserves_previous_archive(self): def test_import_defaults_to_work_archive_and_keeps_permission_gate(self): self.write_config() - with patch.object(spark.Recipe, 'import_images') as importer: - spark.main(['import-images', '--component', 'gateway', '--tag', 'edited']) + with patch.object(tool.Recipe, 'import_images') as importer: + tool.main(['import-images', '--component', 'gateway', '--tag', 'edited']) importer.assert_called_once_with(self.work/'arm64-images.tar', False, 'gateway', 'edited') explicit = self.root/'custom.tar' - with patch.object(spark.Recipe, 'import_images') as importer: - spark.main(['import-images', '--archive', str(explicit), '--allow-containerd-import']) + with patch.object(tool.Recipe, 'import_images') as importer: + tool.main(['import-images', '--archive', str(explicit), '--allow-containerd-import']) importer.assert_called_once_with(explicit, True, None, None) def test_chat_uses_saved_config_with_either_stream_option_order(self): self.write_config() for arguments in (['chat', 'Name eight planets.', '--stream'], ['chat', '--stream', 'Name eight planets.']): - with self.subTest(arguments=arguments), patch.object(spark.Recipe, 'chat') as chat: - spark.main(arguments) + with self.subTest(arguments=arguments), patch.object(tool.Recipe, 'chat') as chat: + tool.main(arguments) chat.assert_called_once_with('Name eight planets.', True, 18443) - with patch.object(spark.Recipe, 'chat') as chat: - spark.main(['chat']) + with patch.object(tool.Recipe, 'chat') as chat: + tool.main(['chat']) chat.assert_called_once_with(None, False, 18443) def test_chat_options_cannot_be_silently_ignored_by_other_phases(self): for arguments in (['verify-gateway', '--stream'], ['paths', 'accidental prompt']): - with self.subTest(arguments=arguments), patch.object(spark, 'Recipe') as recipe, self.assertRaisesRegex(RuntimeError, 'only for chat'): - spark.main(arguments) + with self.subTest(arguments=arguments), patch.object(tool, 'Recipe') as recipe, self.assertRaisesRegex(RuntimeError, 'only for chat'): + tool.main(arguments) recipe.assert_not_called() def test_init_creates_private_discovered_configuration(self): - with redirect_stdout(io.StringIO()), patch.object(spark, 'Recipe') as recipe, \ - patch.object(spark.cluster_setup, 'discover_config', return_value=self.config) as discover: - spark.main(['init']) + with redirect_stdout(io.StringIO()), patch.object(tool, 'Recipe') as recipe, \ + patch.object(tool.cluster_setup, 'discover_config', return_value=self.config) as discover: + tool.main(['init']) recipe.assert_not_called() discover.assert_called_once_with('team-context', None) expected = self.config @@ -326,26 +326,26 @@ def test_init_creates_private_discovered_configuration(self): self.assertEqual(path.stat().st_mode & 0o777, 0o600) self.assertEqual(self.work.stat().st_mode & 0o777, 0o700) self.assertFalse((self.work/'state.json').exists()) - with patch.object(spark.Recipe, 'prepare') as prepare: - spark.main(['prepare']) + with patch.object(tool.Recipe, 'prepare') as prepare: + tool.main(['prepare']) prepare.assert_called_once() def test_init_refuses_state_without_configuration(self): - spark.save(self.work/'state.json', {'identity': 'original'}) + tool.save(self.work/'state.json', {'identity': 'original'}) with self.assertRaisesRegex(RuntimeError, 'does not overwrite'): - spark.main(['init']) + tool.main(['init']) self.assertFalse((self.work/'config.json').exists()) self.assertEqual(json.loads((self.work/'state.json').read_text()), {'identity': 'original'}) def test_init_archives_stale_progress_and_preserves_config_and_credentials(self): self.write_config() - recipe = spark.Recipe(self.config, self.work) + recipe = tool.Recipe(self.config, self.work) original = {'identity': recipe.identity, 'serve': True, 'stack': True, 'attachedExisting': True} - spark.save(self.work/'state.json', original) - spark.save(self.work/'api-key', 'keep-me') + tool.save(self.work/'state.json', original) + tool.save(self.work/'api-key', 'keep-me') before = (self.work/'config.json').read_bytes() - with patch.object(spark.cluster_setup, 'validate_reinitialization') as inspect, redirect_stdout(io.StringIO()): - spark.main(['init']) + with patch.object(tool.cluster_setup, 'validate_reinitialization') as inspect, redirect_stdout(io.StringIO()): + tool.main(['init']) inspect.assert_called_once_with(self.config) self.assertEqual((self.work/'config.json').read_bytes(), before) self.assertEqual((self.work/'api-key').read_text(), 'keep-me') @@ -354,33 +354,33 @@ def test_init_archives_stale_progress_and_preserves_config_and_credentials(self) self.assertEqual(json.loads((archive/'state.json').read_text()), original) self.assertEqual(json.loads((archive/'config.json').read_text()), self.config) self.assertEqual(archive.stat().st_mode & 0o777, 0o700) - self.assertEqual(spark.Recipe(self.config, self.work).state, {}) + self.assertEqual(tool.Recipe(self.config, self.work).state, {}) def test_init_reuses_config_when_progress_was_already_archived(self): self.write_config() - with patch.object(spark.cluster_setup, 'validate_reinitialization') as inspect, redirect_stdout(io.StringIO()): - spark.main(['--context', 'team-context', 'init']) + with patch.object(tool.cluster_setup, 'validate_reinitialization') as inspect, redirect_stdout(io.StringIO()): + tool.main(['--context', 'team-context', 'init']) inspect.assert_called_once_with(self.config) self.assertFalse((self.work/'state.json').exists()) self.assertEqual(list(self.work.glob('before-reinit-*')), []) def test_rejected_reinit_preserves_progress(self): self.write_config() - recipe = spark.Recipe(self.config, self.work) + recipe = tool.Recipe(self.config, self.work) original = {'identity': recipe.identity, 'serve': True} - spark.save(self.work/'state.json', original) - with patch.object(spark.cluster_setup, 'validate_reinitialization', - side_effect=spark.cluster_setup.ClusterSetupError('Live installation')), \ + tool.save(self.work/'state.json', original) + with patch.object(tool.cluster_setup, 'validate_reinitialization', + side_effect=tool.cluster_setup.ClusterSetupError('Live installation')), \ redirect_stderr(io.StringIO()), self.assertRaises(SystemExit): - spark.main(['init']) + tool.main(['init']) self.assertEqual(json.loads((self.work/'state.json').read_text()), original) self.assertEqual(list(self.work.glob('before-reinit-*')), []) def test_init_supports_new_explicit_config_path_and_namespace(self): path = self.root/'custom'/'config.json' - with redirect_stdout(io.StringIO()), patch.object(spark.cluster_setup, 'discover_config', \ + with redirect_stdout(io.StringIO()), patch.object(tool.cluster_setup, 'discover_config', \ return_value=dict(self.config, namespace='my-stack')) as discover: - spark.main(['--config', str(path), '--namespace', 'my-stack', 'init']) + tool.main(['--config', str(path), '--namespace', 'my-stack', 'init']) discover.assert_called_once_with('team-context', 'my-stack') config = json.loads(path.read_text()) self.assertEqual(config['context'], 'team-context') @@ -388,12 +388,12 @@ def test_init_supports_new_explicit_config_path_and_namespace(self): self.assertFalse((self.work/'config.json').exists()) def test_repeated_attach_checks_existing_node_bindings_before_refresh(self): - recipe = spark.Recipe(self.config, self.work) + recipe = tool.Recipe(self.config, self.work) recipe.state = {'identity': recipe.identity, 'attachedExisting': True, 'inventory': {'nodes': {name: name+'-old' for name in self.config['nodes'].values()}}} original = copy.deepcopy(recipe.state) live = {'items': [{'metadata': {'name': name, 'uid': name+'-new'}} for name in self.config['nodes'].values()]} - with patch.object(spark, 'output', return_value=json.dumps(live)) as output, \ + with patch.object(tool, 'output', return_value=json.dumps(live)) as output, \ self.assertRaisesRegex(RuntimeError, 'node identities changed'): recipe.attach_existing() self.assertEqual(output.call_count, 1) diff --git a/deploy/helm/llm-routing/spark/tests/test_client.py b/deploy/helm/llm-routing/recipes/tests/test_client.py similarity index 99% rename from deploy/helm/llm-routing/spark/tests/test_client.py rename to deploy/helm/llm-routing/recipes/tests/test_client.py index c8c23da68..70daa06d9 100644 --- a/deploy/helm/llm-routing/spark/tests/test_client.py +++ b/deploy/helm/llm-routing/recipes/tests/test_client.py @@ -11,7 +11,7 @@ from unittest.mock import Mock, patch HERE = pathlib.Path(__file__).resolve().parents[1] -spec = importlib.util.spec_from_file_location('spark_client', HERE/'client.py') +spec = importlib.util.spec_from_file_location('recipe_client', HERE/'client.py') client = importlib.util.module_from_spec(spec) spec.loader.exec_module(client) MODEL = json.loads((HERE/'backend.defaults.json').read_text())['model']['servedName'] diff --git a/deploy/helm/llm-routing/spark/tests/test_cluster_setup.py b/deploy/helm/llm-routing/recipes/tests/test_cluster_setup.py similarity index 100% rename from deploy/helm/llm-routing/spark/tests/test_cluster_setup.py rename to deploy/helm/llm-routing/recipes/tests/test_cluster_setup.py diff --git a/deploy/helm/llm-routing/spark/tests/test_console_output.py b/deploy/helm/llm-routing/recipes/tests/test_console_output.py similarity index 96% rename from deploy/helm/llm-routing/spark/tests/test_console_output.py rename to deploy/helm/llm-routing/recipes/tests/test_console_output.py index ad5432fa1..94c742454 100644 --- a/deploy/helm/llm-routing/spark/tests/test_console_output.py +++ b/deploy/helm/llm-routing/recipes/tests/test_console_output.py @@ -13,7 +13,7 @@ HERE = pathlib.Path(__file__).resolve().parents[1] sys.path.insert(0, str(HERE)) import console_output -import spark +import recipe as tool class ConsoleOutputTests(unittest.TestCase): @@ -138,9 +138,9 @@ def fail(): print('node inspection completed') raise RuntimeError('Selected model GPU is occupied: another-model') - with self.captured(), patch.object(spark, 'Recipe') as recipe: + with self.captured(), patch.object(tool, 'Recipe') as recipe: recipe.return_value.inventory.side_effect = fail - status = spark.cli(arguments) + status = tool.cli(arguments) self.assertEqual(status, 1) recipe.return_value.inventory.assert_called_once_with() @@ -155,9 +155,9 @@ def fail(): def test_cli_entrypoint_reports_single_success_after_action_completes(self): arguments = self.cli_arguments() - with self.captured(), patch.object(spark, 'Recipe') as recipe: + with self.captured(), patch.object(tool, 'Recipe') as recipe: recipe.return_value.inventory.side_effect = lambda: print('Inventory passed.') - status = spark.cli(arguments) + status = tool.cli(arguments) self.assertEqual(status, 0) recipe.return_value.inventory.assert_called_once_with() diff --git a/deploy/helm/llm-routing/spark/tests/test_discovery.py b/deploy/helm/llm-routing/recipes/tests/test_discovery.py similarity index 93% rename from deploy/helm/llm-routing/spark/tests/test_discovery.py rename to deploy/helm/llm-routing/recipes/tests/test_discovery.py index c45cf9d02..8da5a7620 100644 --- a/deploy/helm/llm-routing/spark/tests/test_discovery.py +++ b/deploy/helm/llm-routing/recipes/tests/test_discovery.py @@ -8,9 +8,9 @@ from unittest.mock import patch HERE = pathlib.Path(__file__).resolve().parents[1] -spec = importlib.util.spec_from_file_location('discovery_recipe', HERE/'spark.py') -spark = importlib.util.module_from_spec(spec) -spec.loader.exec_module(spark) +spec = importlib.util.spec_from_file_location('discovery_recipe', HERE/'recipe.py') +tool = importlib.util.module_from_spec(spec) +spec.loader.exec_module(tool) class DiscoveryTests(unittest.TestCase): @@ -36,7 +36,7 @@ def owned(name, release, node=None): ('inferenceendpoint', 'glm53-iq2'): endpoint} self.deployments = [self.gateway] self.values = { - 'team-stack': {'sparkRecipeSource': {'revision': 'old-build'}, + 'team-stack': {'recipeSource': {'revision': 'old-build'}, 'clusterId': 'demo', 'apiKeys': [{'id': 'owner', 'sha256': 'DO-NOT-COPY-HASH'}], 'tls': {'privateKey': 'DO-NOT-COPY-TLS'}, 'llm-api-gateway': {'llmApiGateway': {'image': {'registry': 'registry.example.com', 'repository': 'gateway', 'tag': 'current', 'pullPolicy': 'Never'}, @@ -68,8 +68,8 @@ def output(self, command): return json.dumps(self.resources[tuple(command[offset+1:offset+3])]) def discover(self, namespace=None): - with patch.object(spark, 'output', side_effect=self.output), patch.object(spark, 'run') as mutate: - result = spark.discover_config('my-context', namespace) + with patch.object(tool, 'output', side_effect=self.output), patch.object(tool, 'run') as mutate: + result = tool.discover_config('my-context', namespace) mutate.assert_not_called() return result @@ -108,7 +108,7 @@ def test_namespace_scoped_discovery_does_not_list_cluster_deployments(self): def test_source_metadata_is_informational_for_discovery(self): for source in (None, {'revision': 'another-build'}, {'repository': 'old-fork', 'revision': 'old-pin'}): - self.values['team-stack']['sparkRecipeSource'] = source + self.values['team-stack']['recipeSource'] = source self.discover() def test_model_service_foreign_ownership_is_rejected(self): @@ -131,8 +131,8 @@ def test_live_image_drift_is_rejected(self): self.discover() def test_missing_context_never_inspects_default_cluster(self): - with patch.object(spark, 'output') as output, self.assertRaisesRegex(RuntimeError, '--context'): - spark.discover_config(None) + with patch.object(tool, 'output') as output, self.assertRaisesRegex(RuntimeError, '--context'): + tool.discover_config(None) output.assert_not_called() diff --git a/deploy/helm/llm-routing/spark/tests/test_gateway_access.py b/deploy/helm/llm-routing/recipes/tests/test_gateway_access.py similarity index 100% rename from deploy/helm/llm-routing/spark/tests/test_gateway_access.py rename to deploy/helm/llm-routing/recipes/tests/test_gateway_access.py diff --git a/deploy/helm/llm-routing/spark/tests/test_image_import.py b/deploy/helm/llm-routing/recipes/tests/test_image_import.py similarity index 96% rename from deploy/helm/llm-routing/spark/tests/test_image_import.py rename to deploy/helm/llm-routing/recipes/tests/test_image_import.py index f8a23190c..d3c504654 100644 --- a/deploy/helm/llm-routing/spark/tests/test_image_import.py +++ b/deploy/helm/llm-routing/recipes/tests/test_image_import.py @@ -21,9 +21,9 @@ from unittest.mock import patch HERE = pathlib.Path(__file__).resolve().parents[1] -spec = importlib.util.spec_from_file_location('image_import_recipe', HERE/'spark.py') -spark = importlib.util.module_from_spec(spec) -spec.loader.exec_module(spark) +spec = importlib.util.spec_from_file_location('image_import_recipe', HERE/'recipe.py') +tool = importlib.util.module_from_spec(spec) +spec.loader.exec_module(tool) class ImageImportTests(unittest.TestCase): @@ -32,7 +32,7 @@ def setUp(self): self.addCleanup(self.tmp.cleanup) self.root = pathlib.Path(self.tmp.name) self.config = json.loads((HERE/'config.example.json').read_text()) - self.recipe = spark.Recipe(self.config, self.root/'work') + self.recipe = tool.Recipe(self.config, self.root/'work') self.archive = self.root/'images.tar' manifest = json.dumps([{'RepoTags': [self.recipe.image('gateway', 'edited')]}]).encode() with tarfile.open(self.archive, 'w') as tar: @@ -112,9 +112,9 @@ def invoke(self): with ExitStack() as stack: stack.enter_context(patch.object(self.recipe, 'bound_cluster')) helm = stack.enter_context(patch.object(self.recipe, 'helm_apply', side_effect=lambda *args, **kwargs: self.helm_calls.append((args, kwargs)))) - stack.enter_context(patch.object(spark, 'output', side_effect=self.output)) - stack.enter_context(patch.object(spark, 'run', side_effect=self.execute)) - stack.enter_context(patch.object(spark.time, 'sleep')) + stack.enter_context(patch.object(tool, 'output', side_effect=self.output)) + stack.enter_context(patch.object(tool, 'run', side_effect=self.execute)) + stack.enter_context(patch.object(tool.time, 'sleep')) stack.enter_context(redirect_stdout(io.StringIO())) self.recipe.import_images(self.archive, True, 'gateway', 'edited') return helm.call_args @@ -182,7 +182,7 @@ def failed_attempt(self): def test_failed_own_attempt_can_retry_without_waiting_for_job_deadline(self): self.prior_jobs = copy.deepcopy(self.jobs) - spark.save(self.recipe.work/'image-import-attempt.json', self.failed_attempt()) + tool.save(self.recipe.work/'image-import-attempt.json', self.failed_attempt()) self.invoke() self.assertEqual(len(self.helm_calls), 1) self.assertFalse((self.recipe.work/'image-import-attempt.json').exists()) @@ -193,7 +193,7 @@ def test_retry_refuses_different_revision_job_uid_context_or_namespace(self): with self.subTest(field=field): previous = self.failed_attempt() previous[field] = {} if field == 'jobs' else 'different' - spark.save(self.recipe.work/'image-import-attempt.json', previous) + tool.save(self.recipe.work/'image-import-attempt.json', previous) with self.assertRaisesRegex(RuntimeError, 'Another image import|revision changed'): self.invoke() self.assertEqual(self.helm_calls, []) diff --git a/deploy/helm/llm-routing/spark/tests/test_load_resume.py b/deploy/helm/llm-routing/recipes/tests/test_load_resume.py similarity index 95% rename from deploy/helm/llm-routing/spark/tests/test_load_resume.py rename to deploy/helm/llm-routing/recipes/tests/test_load_resume.py index cfa66223d..aadfe52e4 100644 --- a/deploy/helm/llm-routing/spark/tests/test_load_resume.py +++ b/deploy/helm/llm-routing/recipes/tests/test_load_resume.py @@ -10,17 +10,17 @@ from unittest.mock import patch HERE = pathlib.Path(__file__).resolve().parents[1] -spec = importlib.util.spec_from_file_location('spark_load_resume', HERE/'spark.py') -spark = importlib.util.module_from_spec(spec) -spec.loader.exec_module(spark) +spec = importlib.util.spec_from_file_location('recipe_load_resume', HERE/'recipe.py') +tool = importlib.util.module_from_spec(spec) +spec.loader.exec_module(tool) class LoadResumeTests(unittest.TestCase): def setUp(self): - self.tmp = tempfile.TemporaryDirectory(prefix='spark-resume-') + self.tmp = tempfile.TemporaryDirectory(prefix='recipe-resume-') self.addCleanup(self.tmp.cleanup) config = json.loads((HERE/'config.example.json').read_text()) - self.recipe = spark.Recipe(config, self.tmp.name) + self.recipe = tool.Recipe(config, self.tmp.name) self.recipe.state = {'download': True, 'runtimeSha256': 'a'*64} self.values = self.recipe.backend_values('serve') self.release = {'name': self.recipe.glm, 'namespace': config['namespace'], @@ -87,8 +87,8 @@ def command_output(self, command, **kwargs): def attempt(self, error=None): with patch.object(self.recipe, 'bound_cluster') as bound, \ - patch.object(self.recipe, 'helm_apply') as helm, patch.object(spark, 'run') as run, \ - patch.object(spark, 'output', side_effect=self.command_output): + patch.object(self.recipe, 'helm_apply') as helm, patch.object(tool, 'run') as run, \ + patch.object(tool, 'output', side_effect=self.command_output): if error: with self.assertRaisesRegex(RuntimeError, error): self.recipe.backend_phase('serve') @@ -177,7 +177,7 @@ def test_revision_or_resource_replacement_during_check_is_rejected(self): def test_read_timeout_preserves_missing_checkpoint(self): with patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'helm_apply') as helm, \ - patch.object(spark, 'output', side_effect=subprocess.TimeoutExpired(['helm', 'status'], 45)): + patch.object(tool, 'output', side_effect=subprocess.TimeoutExpired(['helm', 'status'], 45)): with self.assertRaises(subprocess.TimeoutExpired): self.recipe.backend_phase('serve') self.assertNotIn('serve', self.recipe.state) @@ -186,7 +186,7 @@ def test_read_timeout_preserves_missing_checkpoint(self): def test_connection_failure_preserves_missing_checkpoint(self): with patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'helm_apply') as helm, \ - patch.object(spark, 'output', side_effect=OSError('connection lost')): + patch.object(tool, 'output', side_effect=OSError('connection lost')): with self.assertRaisesRegex(OSError, 'connection lost'): self.recipe.backend_phase('serve') self.assertNotIn('serve', self.recipe.state) diff --git a/deploy/helm/llm-routing/spark/tests/test_recipe.py b/deploy/helm/llm-routing/recipes/tests/test_recipe.py similarity index 86% rename from deploy/helm/llm-routing/spark/tests/test_recipe.py rename to deploy/helm/llm-routing/recipes/tests/test_recipe.py index 9fc723977..7620621d8 100644 --- a/deploy/helm/llm-routing/spark/tests/test_recipe.py +++ b/deploy/helm/llm-routing/recipes/tests/test_recipe.py @@ -16,23 +16,23 @@ from unittest.mock import Mock, patch HERE = pathlib.Path(__file__).resolve().parents[1] -spec = importlib.util.spec_from_file_location('spark_recipe', HERE/'spark.py') -spark = importlib.util.module_from_spec(spec) -spec.loader.exec_module(spark) +spec = importlib.util.spec_from_file_location('recipe_tool', HERE/'recipe.py') +tool = importlib.util.module_from_spec(spec) +spec.loader.exec_module(tool) class RecipeTests(unittest.TestCase): def setUp(self): - self.tmp = tempfile.TemporaryDirectory(prefix='spark-recipe-test-') + self.tmp = tempfile.TemporaryDirectory(prefix='recipe-test-') self.addCleanup(self.tmp.cleanup) self.config = json.loads((HERE/'config.example.json').read_text()) - self.recipe = spark.Recipe(self.config, self.tmp.name) + self.recipe = tool.Recipe(self.config, self.tmp.name) def source_repository(self, name='seed'): source = pathlib.Path(self.tmp.name).resolve()/name source.mkdir() subprocess.run(['git', 'init', '--quiet', source], check=True) - for path in spark.COMPONENTS.values(): + for path in tool.COMPONENTS.values(): (source/path).mkdir(parents=True, exist_ok=True) for name in ('llm-gateway-stack', 'llm-api-gateway', 'llm-request-router'): chart = source/'deploy/helm'/name/name @@ -43,17 +43,17 @@ def source_repository(self, name='seed'): subprocess.run(['git', '-c', 'user.name=Recipe Test', '-c', 'user.email=recipe@example.com', '-c', 'commit.gpgsign=false', '-c', 'core.hooksPath=/dev/null', 'commit', '--quiet', '-m', 'Initial source'], cwd=source, check=True) - revision = spark.output(['git', 'rev-parse', 'HEAD'], cwd=source).strip() + revision = tool.output(['git', 'rev-parse', 'HEAD'], cwd=source).strip() return source, {'repository': str(source), 'revision': revision} def test_prepare_only_builds_dependencies_in_the_selected_checkout(self): self.recipe.source, lock = self.source_repository() - with patch.object(spark, 'run') as run: + with patch.object(tool, 'run') as run: self.recipe.prepare() self.assertEqual(self.recipe.source_identity(), {'revision': lock['revision']}) - self.assertEqual(spark.output(['git', 'rev-parse', 'HEAD'], cwd=self.recipe.source).strip(), lock['revision']) + self.assertEqual(tool.output(['git', 'rev-parse', 'HEAD'], cwd=self.recipe.source).strip(), lock['revision']) self.assertEqual((self.recipe.source/'service.txt').read_text(), 'committed service\n') - self.assertEqual(spark.output(['git', 'status', '--porcelain'], cwd=self.recipe.source), '') + self.assertEqual(tool.output(['git', 'status', '--porcelain'], cwd=self.recipe.source), '') run.assert_called_once_with(['helm', 'dependency', 'build', '--skip-refresh', self.recipe.source/'deploy/helm/llm-gateway-stack/llm-gateway-stack']) @@ -62,25 +62,25 @@ def test_default_source_is_the_recipe_checkout_and_ignores_stale_work_source(sel stale = self.recipe.work/'source' stale.mkdir() (stale/'unrelated.txt').write_text('leave this old copy alone\n') - recipe_path = seed/'deploy/helm/llm-routing/spark' - with patch.object(spark, 'HERE', recipe_path), \ - patch.object(spark, 'run') as run: - recipe = spark.Recipe(self.config, self.recipe.work) + recipe_path = seed/'deploy/helm/llm-routing/recipes' + with patch.object(tool, 'HERE', recipe_path), \ + patch.object(tool, 'run') as run: + recipe = tool.Recipe(self.config, self.recipe.work) recipe.prepare() recipe.build_images('gateway', 'developer-change') self.assertEqual(recipe.source, seed.resolve()) - self.assertEqual(run.call_args.args[0][-1], str(seed/spark.COMPONENTS['gateway'])) + self.assertEqual(run.call_args.args[0][-1], str(seed/tool.COMPONENTS['gateway'])) self.assertEqual((stale/'unrelated.txt').read_text(), 'leave this old copy alone\n') def test_prepare_and_build_keep_local_edits_at_the_pinned_head(self): self.recipe.source, lock = self.source_repository() edited = self.recipe.source/'service.txt' edited.write_text('local gateway change\n') - with patch.object(spark, 'run') as run: + with patch.object(tool, 'run') as run: self.recipe.prepare() self.recipe.build_images('gateway', 'edited-build') self.assertEqual(edited.read_text(), 'local gateway change\n') - self.assertEqual(spark.output(['git', 'rev-parse', 'HEAD'], cwd=self.recipe.source).strip(), lock['revision']) + self.assertEqual(tool.output(['git', 'rev-parse', 'HEAD'], cwd=self.recipe.source).strip(), lock['revision']) self.assertEqual([call.args[0][0] for call in run.call_args_list], ['helm', 'docker', 'docker']) self.assertIn(self.recipe.image('gateway', 'edited-build'), run.call_args.args[0]) @@ -92,15 +92,15 @@ def test_prepare_and_build_accept_committed_descendants_without_resetting_edits( subprocess.run(['git', '-c', 'user.name=Recipe Test', '-c', 'user.email=recipe@example.com', '-c', 'commit.gpgsign=false', '-c', 'core.hooksPath=/dev/null', 'commit', '--quiet', '-m', 'Developer change'], cwd=self.recipe.source, check=True) - head = spark.output(['git', 'rev-parse', 'HEAD'], cwd=self.recipe.source).strip() + head = tool.output(['git', 'rev-parse', 'HEAD'], cwd=self.recipe.source).strip() edited.write_text('uncommitted follow-up\n') - with patch.object(spark, 'run') as run: + with patch.object(tool, 'run') as run: self.recipe.prepare() self.recipe.build_images('router', 'developer-change') self.assertNotEqual(head, lock['revision']) - self.assertEqual(spark.output(['git', 'rev-parse', 'HEAD'], cwd=self.recipe.source).strip(), head) + self.assertEqual(tool.output(['git', 'rev-parse', 'HEAD'], cwd=self.recipe.source).strip(), head) self.assertEqual(edited.read_text(), 'uncommitted follow-up\n') - self.assertEqual(run.call_args.args[0][-1], str(self.recipe.source/spark.COMPONENTS['router'])) + self.assertEqual(run.call_args.args[0][-1], str(self.recipe.source/tool.COMPONENTS['router'])) self.assertEqual([call.args[0][0] for call in run.call_args_list], ['helm', 'docker', 'docker']) def test_rewritten_history_with_the_same_sources_is_accepted(self): @@ -111,7 +111,7 @@ def test_rewritten_history_with_the_same_sources_is_accepted(self): 'commit', '--quiet', '-m', 'Unrelated history'], cwd=self.recipe.source, check=True) for action in (self.recipe.prepare, self.recipe.build_images): with self.subTest(action=action.__name__), \ - patch.object(spark, 'run') as run: + patch.object(tool, 'run') as run: action() run.assert_called() @@ -119,13 +119,13 @@ def test_source_must_be_the_checkout_root(self): seed, lock = self.source_repository() self.recipe.source = seed/'nested' self.recipe.source.mkdir() - with patch.object(spark, 'run') as run, self.assertRaises(RuntimeError): + with patch.object(tool, 'run') as run, self.assertRaises(RuntimeError): self.recipe.prepare() run.assert_not_called() def test_missing_source_does_not_clone_or_create_a_checkout(self): self.recipe.source = self.recipe.work/'missing' - with patch.object(spark, 'run') as run, self.assertRaises(RuntimeError): + with patch.object(tool, 'run') as run, self.assertRaises(RuntimeError): self.recipe.prepare() run.assert_not_called() self.assertFalse(self.recipe.source.exists()) @@ -140,8 +140,8 @@ def output(command, **kwargs): operations.append(tuple(str(value) for value in command[:2])) return 'kind: List\nitems: []\n' - with patch.object(self.recipe, 'source_check'), patch.object(spark, 'run', side_effect=run), \ - patch.object(spark, 'output', side_effect=output): + with patch.object(self.recipe, 'source_check'), patch.object(tool, 'run', side_effect=run), \ + patch.object(tool, 'output', side_effect=output): self.recipe.render() self.assertEqual(operations[0], ('helm', 'dependency', 'build')) self.assertEqual(sum(operation[:2] == ('helm', 'lint') for operation in operations), 8) @@ -149,8 +149,8 @@ def output(command, **kwargs): self.assertEqual(len(list((self.recipe.work/'render').glob('*.yaml'))), 8) def test_render_dependency_failure_stops_before_rendered_files_or_templates(self): - with patch.object(self.recipe, 'source_check'), patch.object(spark, 'run', side_effect=RuntimeError('dependency build failed')) as run, \ - patch.object(spark, 'output') as output, self.assertRaisesRegex(RuntimeError, 'dependency build failed'): + with patch.object(self.recipe, 'source_check'), patch.object(tool, 'run', side_effect=RuntimeError('dependency build failed')) as run, \ + patch.object(tool, 'output') as output, self.assertRaisesRegex(RuntimeError, 'dependency build failed'): self.recipe.render() self.assertEqual(run.call_args.args[0][:3], ['helm', 'dependency', 'build']) run.assert_called_once() @@ -159,7 +159,7 @@ def test_render_dependency_failure_stops_before_rendered_files_or_templates(self def test_build_images_does_not_prepare_chart_dependencies(self): with patch.object(self.recipe, 'source_check'), patch.object(self.recipe, 'prepare') as prepare, \ - patch.object(spark, 'run') as run: + patch.object(tool, 'run') as run: self.recipe.build_images('gateway', 'test-build') prepare.assert_not_called() self.assertTrue(all(command.args[0][0] == 'docker' for command in run.call_args_list)) @@ -191,13 +191,13 @@ def test_chart_digest_ignores_source_history_and_generated_dependencies(self): def test_missing_sources_or_charts_are_rejected_before_preparation(self): self.recipe.source, _ = self.source_repository() - (self.recipe.source/spark.COMPONENTS['gateway']).rmdir() - with patch.object(spark, 'run') as run, self.assertRaisesRegex(RuntimeError, 'missing required'): + (self.recipe.source/tool.COMPONENTS['gateway']).rmdir() + with patch.object(tool, 'run') as run, self.assertRaisesRegex(RuntimeError, 'missing required'): self.recipe.prepare() run.assert_not_called() - (self.recipe.source/spark.COMPONENTS['gateway']).mkdir() + (self.recipe.source/tool.COMPONENTS['gateway']).mkdir() (self.recipe.source/'deploy/helm/llm-api-gateway/llm-api-gateway/Chart.yaml').unlink() - with patch.object(spark, 'run') as run, self.assertRaisesRegex(RuntimeError, 'Missing routing chart'): + with patch.object(tool, 'run') as run, self.assertRaisesRegex(RuntimeError, 'Missing routing chart'): self.recipe.prepare() run.assert_not_called() @@ -206,7 +206,7 @@ def test_missing_or_unreachable_docker_fails_before_build_with_clear_action(self subprocess.TimeoutExpired(['docker', 'info'], 15)) for error in failures: with self.subTest(error=type(error).__name__), patch.object(self.recipe, 'source_check'), \ - patch.object(spark, 'run', side_effect=error) as run, self.assertRaises(spark.DockerUnavailableError) as result: + patch.object(tool, 'run', side_effect=error) as run, self.assertRaises(tool.DockerUnavailableError) as result: self.recipe.build_images('gateway') self.assertEqual(str(result.exception), 'Start Docker, then rerun build-images.') run.assert_called_once_with(['docker', 'info'], stdout=subprocess.DEVNULL, stderr=subprocess.PIPE, timeout=15) @@ -214,7 +214,7 @@ def test_missing_or_unreachable_docker_fails_before_build_with_clear_action(self def test_docker_probe_uses_existing_environment_and_does_not_hide_build_failures(self): failure = subprocess.CalledProcessError(1, ['docker', 'buildx', 'build']) with patch.dict(os.environ, {'DOCKER_HOST': 'unix:///custom/docker.sock', 'DOCKER_CONFIG': '/custom/config'}), \ - patch.object(self.recipe, 'source_check'), patch.object(spark, 'run', side_effect=[None, failure]) as run, \ + patch.object(self.recipe, 'source_check'), patch.object(tool, 'run', side_effect=[None, failure]) as run, \ self.assertRaises(subprocess.CalledProcessError): self.recipe.build_images('gateway') self.assertEqual(run.call_args_list[0].args[0], ['docker', 'info']) @@ -230,29 +230,29 @@ def test_operator_build_identifies_actual_checkout_and_local_edits(self): subprocess.run(['git', '-c', 'user.name=Recipe Test', '-c', 'user.email=recipe@example.com', '-c', 'commit.gpgsign=false', '-c', 'core.hooksPath=/dev/null', 'commit', '--quiet', '-m', 'Operator change'], cwd=self.recipe.source, check=True) - head = spark.output(['git', 'rev-parse', 'HEAD'], cwd=self.recipe.source).strip() + head = tool.output(['git', 'rev-parse', 'HEAD'], cwd=self.recipe.source).strip() self.assertNotEqual(head, lock['revision']) for dirty in (False, True): if dirty: edited.write_text('uncommitted operator change\n') - with self.subTest(dirty=dirty), patch.object(spark, 'run') as run: + with self.subTest(dirty=dirty), patch.object(tool, 'run') as run: self.recipe.build_images('operator') expected = head + ('-dirty' if dirty else '') self.assertIn('SOURCE_REVISION='+expected, run.call_args.args[0]) - self.assertEqual(run.call_args.args[0][-1], str(self.recipe.source/spark.COMPONENTS['operator'])) + self.assertEqual(run.call_args.args[0][-1], str(self.recipe.source/tool.COMPONENTS['operator'])) def test_duplicate_model_nodes_and_missing_context_are_rejected(self): self.config['nodes']['worker'] = self.config['nodes']['leader'] with self.assertRaisesRegex(RuntimeError, 'distinct GPU'): - spark.validate(self.config) + tool.validate(self.config) self.config['nodes']['worker'] = 'another-node' self.config['context'] = '' with self.assertRaisesRegex(RuntimeError, 'context'): - spark.validate(self.config) + tool.validate(self.config) def test_work_directory_cannot_put_credentials_in_checkout(self): with self.assertRaisesRegex(RuntimeError, 'outside the checkout'): - spark.Recipe(self.config, HERE/'.work') + tool.Recipe(self.config, HERE/'.work') def test_extra_model_configuration_is_rejected_before_any_commands(self): for field, value in [('retainedModels', ['legacy-model']), ('testFixture', True)]: @@ -260,9 +260,9 @@ def test_extra_model_configuration_is_rejected_before_any_commands(self): config = copy.deepcopy(self.config) config[field] = value directory = pathlib.Path(self.tmp.name)/'rejected' - with patch.object(spark, 'run') as run, patch.object(spark, 'output') as output: + with patch.object(tool, 'run') as run, patch.object(tool, 'output') as output: with self.assertRaisesRegex(RuntimeError, field): - spark.Recipe(config, directory) + tool.Recipe(config, directory) run.assert_not_called() output.assert_not_called() self.assertFalse(directory.exists()) @@ -288,8 +288,8 @@ def test_stack_generates_private_key_hash_and_only_glm_stack_components(self): with patch.object(self.recipe, 'source_check'), \ patch.object(self.recipe, 'bound_cluster', side_effect=lambda: operations.append('bound')), \ patch.object(self.recipe, 'helm_apply', side_effect=lambda release, *args: operations.append(release)) as helm, \ - patch.object(spark, 'run', side_effect=lambda *args, **kwargs: operations.append('dependencies')) as run, \ - patch.object(spark, 'output', side_effect=responses): + patch.object(tool, 'run', side_effect=lambda *args, **kwargs: operations.append('dependencies')) as run, \ + patch.object(tool, 'output', side_effect=responses): self.recipe.deploy_stack() self.assertEqual(operations, ['bound', 'dependencies', self.recipe.operator, self.recipe.stack]) self.assertEqual(run.call_args.args[0][:3], ['helm', 'dependency', 'build']) @@ -301,7 +301,7 @@ def test_stack_generates_private_key_hash_and_only_glm_stack_components(self): stack_values = helm.call_args_list[1].args[2] self.assertNotIn(key, json.dumps(stack_values)) self.assertEqual(stack_values['apiKeys'][0]['sha256'], hashlib.sha256(key.encode()).hexdigest()) - self.assertEqual(stack_values['sparkRecipeSource'], self.recipe.source_identity()) + self.assertEqual(stack_values['recipeSource'], self.recipe.source_identity()) self.assertEqual(self.recipe.components(), ['gateway', 'router', 'pylon', 'operator']) self.assertEqual(self.recipe.state['stack']['apiKeyFile'], str(key_path.resolve())) @@ -325,7 +325,7 @@ def test_inventory_rejects_all_gpu_allocations_on_model_nodes(self): 'spec': {'nodeName': node, **allocation}, 'status': {'phase': phase}} responses = [json.dumps({'items': items}) for items in (nodes, [pod], [])] with self.subTest(allocation=allocation, phase=phase, node=node), \ - patch.object(spark, 'output', side_effect=responses), patch.object(spark, 'run') as run: + patch.object(tool, 'output', side_effect=responses), patch.object(tool, 'run') as run: if busy: with self.assertRaisesRegex(RuntimeError, 'GPU is occupied: gpu-consumer'): self.recipe.inventory() @@ -337,7 +337,7 @@ def test_stack_reuses_ui_key_in_existing_release(self): encoded = __import__('base64').b64encode(b'private-cluster-token').decode() responses = [json.dumps({'data': {'cluster-token': encoded}}), json.dumps({'data': {'ca.crt': 'public-ca'}})] with patch.object(self.recipe, 'prepare'), patch.object(self.recipe, 'bound_cluster'), \ - patch.object(self.recipe, 'helm_apply') as helm, patch.object(spark, 'output', side_effect=responses*2): + patch.object(self.recipe, 'helm_apply') as helm, patch.object(tool, 'output', side_effect=responses*2): self.recipe.deploy_stack() first = copy.deepcopy(helm.call_args.args[2]) self.recipe.deploy_stack() @@ -356,8 +356,8 @@ def test_stack_dependency_failure_stops_before_operator_or_key_creation(self): self.recipe.state = {'inventory': {'nodes': {'control': 'node-uid'}}} original = copy.deepcopy(self.recipe.state) with patch.object(self.recipe, 'source_check'), patch.object(self.recipe, 'bound_cluster') as cluster, \ - patch.object(spark, 'run', side_effect=RuntimeError('dependency build failed')) as run, \ - patch.object(self.recipe, 'helm_apply') as helm, patch.object(spark, 'output') as output, \ + patch.object(tool, 'run', side_effect=RuntimeError('dependency build failed')) as run, \ + patch.object(self.recipe, 'helm_apply') as helm, patch.object(tool, 'output') as output, \ self.assertRaisesRegex(RuntimeError, 'dependency build failed'): self.recipe.deploy_stack() cluster.assert_called_once() @@ -381,7 +381,7 @@ def test_direct_and_gateway_verification_use_the_glm_client(self): for gateway in (False, True): with self.subTest(gateway=gateway): with patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'forward') as forward, \ - patch.object(self.recipe, 'prepare') as prepare, patch.object(spark, 'run') as run: + patch.object(self.recipe, 'prepare') as prepare, patch.object(tool, 'run') as run: self.recipe.verify(gateway, 18443) prepare.assert_not_called() command = [str(value) for value in run.call_args.args[0]] @@ -405,26 +405,26 @@ def incomplete_cleanup(*args): raise RuntimeError('revocation failed') with patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'forward'), \ - patch.object(spark.gateway_access, 'temporary_gateway_key', side_effect=incomplete_cleanup), \ - patch.object(spark, 'run') as run: + patch.object(tool.gateway_access, 'temporary_gateway_key', side_effect=incomplete_cleanup), \ + patch.object(tool, 'run') as run: with self.assertRaisesRegex(RuntimeError, 'revocation failed'): self.recipe.verify(True, 18443) run.assert_called_once() self.assertFalse(self.recipe.state['gateway']) - self.assertFalse(spark.Recipe(self.config, self.tmp.name).state['gateway']) + self.assertFalse(tool.Recipe(self.config, self.tmp.name).state['gateway']) def test_failed_verification_invalidates_previous_success(self): self.recipe.state = {'stack': {'apiKeyFile': '/unused'}, 'gateway': True, 'direct': True} for gateway in (False, True): - with self.subTest(gateway=gateway), patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'forward'), patch.object(spark, 'run', side_effect=RuntimeError('verification failed')): + with self.subTest(gateway=gateway), patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'forward'), patch.object(tool, 'run', side_effect=RuntimeError('verification failed')): with self.assertRaisesRegex(RuntimeError, 'verification failed'): self.recipe.verify(gateway, 18443) - resumed = spark.Recipe(self.config, self.tmp.name) + resumed = tool.Recipe(self.config, self.tmp.name) self.assertFalse(resumed.state['gateway' if gateway else 'direct']) def test_attached_installation_requires_glm_verification_before_update(self): self.recipe.state = {'attachedExisting': True} - with patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'source_check'), patch.object(spark, 'run') as run, patch.object(spark, 'output') as output: + with patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'source_check'), patch.object(tool, 'run') as run, patch.object(tool, 'output') as output: with self.assertRaisesRegex(RuntimeError, 'Verify GLM'): self.recipe.update('gateway', 'next-tag') run.assert_not_called() @@ -468,13 +468,13 @@ def check_archive(*args, **kwargs): self.assertFalse(saved['qualify']) self.assertFalse(saved['download']) - with patch.object(self.recipe, 'bound_cluster'), patch.object(spark, 'output', side_effect=responses), patch.object(self.recipe, 'helm_apply', side_effect=check_archive) as helm, patch.object(self.recipe, 'logs', return_value=[{'result': 'PASS'}]) as logs: + with patch.object(self.recipe, 'bound_cluster'), patch.object(tool, 'output', side_effect=responses), patch.object(self.recipe, 'helm_apply', side_effect=check_archive) as helm, patch.object(self.recipe, 'logs', return_value=[{'result': 'PASS'}]) as logs: self.recipe.backend_phase('qualify', retry=True) self.assertEqual(helm.call_args_list[0].args[2]['qualification']['attempt'], 6) self.assertEqual(helm.call_args_list[1].args[2]['chain']['attempt'], 4) self.assertEqual(logs.call_args_list[0].kwargs['job'], self.recipe.glm+'-qualify-6') self.assertEqual(logs.call_args_list[1].kwargs['job'], self.recipe.glm+'-chain-4') - resumed = spark.Recipe(self.config, self.tmp.name) + resumed = tool.Recipe(self.config, self.tmp.name) self.assertTrue(resumed.state['qualify']) self.assertEqual(resumed.backend_values('qualify')['qualification']['attempt'], 6) self.assertEqual(resumed.backend_values('chain')['chain']['attempt'], 4) @@ -486,15 +486,15 @@ def test_retry_handles_legacy_failed_qualification_and_unavailable_pod_logs(self job = self.qualification_job('-qualify-'+str(attempt)) pod = self.qualification_pod('node-interrupted', job, 'Failed') responses = [json.dumps({'items': [job]}), json.dumps({'items': [pod]}), - spark.subprocess.CalledProcessError(1, ['kubectl', 'logs'], output='container logs unavailable')] - with patch.object(self.recipe, 'bound_cluster'), patch.object(spark, 'output', side_effect=responses), patch.object(self.recipe, 'helm_apply', side_effect=RuntimeError('Helm interrupted')) as helm: + tool.subprocess.CalledProcessError(1, ['kubectl', 'logs'], output='container logs unavailable')] + with patch.object(self.recipe, 'bound_cluster'), patch.object(tool, 'output', side_effect=responses), patch.object(self.recipe, 'helm_apply', side_effect=RuntimeError('Helm interrupted')) as helm: with self.assertRaisesRegex(RuntimeError, 'Helm interrupted'): self.recipe.backend_phase('qualify', retry=True) self.assertEqual(helm.call_args.args[2]['qualification']['attempt'], attempt+1) self.assertEqual(helm.call_args.args[2]['chain']['attempt'], defaults['chain']['attempt']+1) archived = list((self.recipe.work/'evidence').glob('qualification-retry-*'))[0] self.assertEqual((archived/'node-interrupted-log-error.txt').read_text(), 'container logs unavailable') - resumed = spark.Recipe(self.config, self.tmp.name) + resumed = tool.Recipe(self.config, self.tmp.name) self.assertEqual(resumed.state['qualificationAttempt'], attempt+1) self.assertFalse(resumed.state['qualify']) @@ -503,7 +503,7 @@ def test_qualification_retry_refuses_active_or_foreign_jobs_before_mutation(self cases = [(self.qualification_job('-qualify-2', 'Running'), 'still active'), (self.qualification_job('-chain-1', release='another-owner'), 'ownership')] for job, error in cases: - with self.subTest(error=error), patch.object(self.recipe, 'bound_cluster'), patch.object(spark, 'output', return_value=json.dumps({'items': [job]})), patch.object(self.recipe, 'helm_apply') as helm: + with self.subTest(error=error), patch.object(self.recipe, 'bound_cluster'), patch.object(tool, 'output', return_value=json.dumps({'items': [job]})), patch.object(self.recipe, 'helm_apply') as helm: with self.assertRaisesRegex(RuntimeError, error): self.recipe.backend_phase('qualify', retry=True) helm.assert_not_called() @@ -516,7 +516,7 @@ def test_current_job_acceptance_ignores_stale_and_failed_pod_pass_records(self): self.qualification_pod('current-failed', current, 'Failed'), self.qualification_pod('current-complete', current)] for current_log, expected in [('no PASS record', []), ('{"result":"PASS","current":true}', [{'result': 'PASS', 'current': True}])]: - with self.subTest(current_log=current_log), patch.object(spark, 'output', side_effect=[json.dumps({'items': pods}), '{"result":"PASS"}', current_log]) as output: + with self.subTest(current_log=current_log), patch.object(tool, 'output', side_effect=[json.dumps({'items': pods}), '{"result":"PASS"}', current_log]) as output: self.assertEqual(self.recipe.logs('qualification', job=current['metadata']['name']), expected) names = [call.args[0][-1] for call in output.call_args_list[1:]] self.assertNotIn('old-pass', names) @@ -540,7 +540,7 @@ def fail_helm(*args, **kwargs): with patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'helm_apply', side_effect=fail_helm): with self.assertRaisesRegex(RuntimeError, 'qualification failed'): self.recipe.backend_phase('qualify') - resumed = spark.Recipe(self.config, self.tmp.name) + resumed = tool.Recipe(self.config, self.tmp.name) self.assertFalse(resumed.state['qualify']) self.assertFalse(resumed.state['download']) with patch.object(resumed, 'bound_cluster'), patch.object(resumed, 'helm_apply') as helm: @@ -551,15 +551,15 @@ def fail_helm(*args, **kwargs): def test_retry_option_rejects_other_phases_before_recipe_creation(self): for phase in ('load', 'download', 'stack', 'recover'): - args = ['spark.py', '--config', '/unused', '--work-dir', '/unused', phase, '--retry'] - with self.subTest(phase=phase), patch.object(spark.sys, 'argv', args), patch.object(spark, 'Recipe') as recipe: + args = ['recipe.py', '--config', '/unused', '--work-dir', '/unused', phase, '--retry'] + with self.subTest(phase=phase), patch.object(tool.sys, 'argv', args), patch.object(tool, 'Recipe') as recipe: with self.assertRaisesRegex(RuntimeError, 'only for qualify'): - spark.main() + tool.main() recipe.assert_not_called() def test_retry_does_not_replace_already_successful_qualification(self): self.recipe.state = {'runtimeSha256': 'a'*64, 'qualify': True} - with patch.object(self.recipe, 'bound_cluster'), patch.object(spark, 'output') as output, patch.object(self.recipe, 'helm_apply') as helm: + with patch.object(self.recipe, 'bound_cluster'), patch.object(tool, 'output') as output, patch.object(self.recipe, 'helm_apply') as helm: with self.assertRaisesRegex(RuntimeError, 'already passed'): self.recipe.backend_phase('qualify', retry=True) output.assert_not_called() @@ -567,13 +567,13 @@ def test_retry_does_not_replace_already_successful_qualification(self): def test_image_update_only_changes_selected_tag_and_preserves_other_pods(self): values = self.recipe.stack_values('a'*64, 'b'*64) - values['sparkRecipeSource'] = {'revision': 'an-older-build'} + values['recipeSource'] = {'revision': 'an-older-build'} pods = [{'metadata': {'name': 'llm-api-gateway-old', 'uid': 'g1'}, 'status': {'phase': 'Running'}}, {'metadata': {'name': 'unrelated-workload', 'uid': 'u1'}, 'status': {'phase': 'Running'}}, {'metadata': {'name': self.recipe.glm+'-leader', 'uid': 'm1'}, 'status': {'phase': 'Running'}}] after = copy.deepcopy(pods) after[0]['metadata']['uid'] = 'g2' - with patch.object(self.recipe, 'source_check') as source, patch.object(self.recipe, 'bound_cluster'), patch.object(spark, 'output', side_effect=[json.dumps(values), json.dumps({'items': pods}), json.dumps({'items': after})]), patch.object(spark, 'run') as run: + with patch.object(self.recipe, 'source_check') as source, patch.object(self.recipe, 'bound_cluster'), patch.object(tool, 'output', side_effect=[json.dumps(values), json.dumps({'items': pods}), json.dumps({'items': after})]), patch.object(tool, 'run') as run: self.recipe.update('gateway', 'next-tag') self.assertEqual(source.call_args_list[0].kwargs, {}) self.assertEqual(source.call_count, 2) @@ -602,8 +602,8 @@ def read(command, **kwargs): return '{"items": []}' with patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'source_check'), \ - patch.object(spark, 'output', side_effect=read), \ - patch.object(spark, 'run', side_effect=RuntimeError('dependency build failed')) as run, \ + patch.object(tool, 'output', side_effect=read), \ + patch.object(tool, 'run', side_effect=RuntimeError('dependency build failed')) as run, \ self.assertRaisesRegex(RuntimeError, 'dependency build failed'): self.recipe.update('gateway', 'next-tag') self.assertEqual(run.call_args.args[0][:3], ['helm', 'dependency', 'build']) @@ -621,7 +621,7 @@ def test_update_live_repository_or_tag_mismatch_stops_before_preparation(self): image['tag'] = 'next-tag' with self.subTest(mismatch=mismatch), patch.object(self.recipe, 'bound_cluster'), \ patch.object(self.recipe, 'source_check'), patch.object(self.recipe, 'prepare') as prepare, \ - patch.object(spark, 'output', return_value=json.dumps(values)), patch.object(spark, 'run') as run, \ + patch.object(tool, 'output', return_value=json.dumps(values)), patch.object(tool, 'run') as run, \ self.assertRaises(RuntimeError): self.recipe.update('gateway', 'next-tag') prepare.assert_not_called() @@ -632,10 +632,10 @@ def test_image_update_rejects_missing_or_changed_charts_before_mutation(self): self.recipe.state = {'attachedExisting': True, 'gateway': True} for digest in (None, '0'*64): values = self.recipe.stack_values('a'*64, 'b'*64) - values['sparkRecipeChartsSha256'] = digest + values['recipeChartsSha256'] = digest with self.subTest(digest=digest), patch.object(self.recipe, 'source_check'), \ - patch.object(self.recipe, 'bound_cluster'), patch.object(spark, 'output', return_value=json.dumps(values)), \ - patch.object(spark, 'run') as run: + patch.object(self.recipe, 'bound_cluster'), patch.object(tool, 'output', return_value=json.dumps(values)), \ + patch.object(tool, 'run') as run: with self.assertRaisesRegex(RuntimeError, 'coordinated stack installation'): self.recipe.update('gateway', 'next-tag') run.assert_not_called() @@ -648,7 +648,7 @@ def test_image_updates_detect_replacement_of_an_unrelated_running_pod(self): after[0]['metadata']['uid'] = 'replacement' for component in ('gateway', 'router'): with self.subTest(component=component): - with patch.object(self.recipe, 'source_check'), patch.object(self.recipe, 'bound_cluster'), patch.object(spark, 'output', side_effect=[json.dumps(values), json.dumps({'items': pods}), json.dumps({'items': after})]), patch.object(spark, 'run'): + with patch.object(self.recipe, 'source_check'), patch.object(self.recipe, 'bound_cluster'), patch.object(tool, 'output', side_effect=[json.dumps(values), json.dumps({'items': pods}), json.dumps({'items': after})]), patch.object(tool, 'run'): with self.assertRaisesRegex(RuntimeError, 'Backend pods changed'): self.recipe.update(component, 'next-tag') records = list((pathlib.Path(self.tmp.name)/'evidence').glob('update-*.json')) @@ -663,7 +663,7 @@ def test_rollback_restores_the_recorded_component_tag(self): 'component': 'router', 'newTag': 'next-tag', 'previousTag': 'old-tag', 'source': {'revision': 'an-older-build'}, 'chartsSha256': self.recipe.chart_digest()} path = pathlib.Path(self.tmp.name)/'rollback.json' path.write_text(json.dumps(record)) - with patch.object(self.recipe, 'bound_cluster'), patch.object(spark, 'output', return_value=json.dumps(values)), patch.object(self.recipe, 'update') as update: + with patch.object(self.recipe, 'bound_cluster'), patch.object(tool, 'output', return_value=json.dumps(values)), patch.object(self.recipe, 'update') as update: self.recipe.rollback(path) update.assert_called_once_with('router', 'old-tag') @@ -676,7 +676,7 @@ def test_rollback_automatically_prepares_before_the_restoring_upgrade(self): path.write_text(json.dumps(record)) reads = [json.dumps(values), json.dumps(values), '{"items": []}', '{"items": []}'] with patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'source_check'), \ - patch.object(spark, 'output', side_effect=reads), patch.object(spark, 'run') as run: + patch.object(tool, 'output', side_effect=reads), patch.object(tool, 'run') as run: self.recipe.rollback(path) self.assertEqual(run.call_count, 2) self.assertEqual(run.call_args_list[0].args[0][:3], ['helm', 'dependency', 'build']) @@ -688,7 +688,7 @@ def test_rollback_refuses_a_subsequent_update(self): 'component': 'gateway', 'newTag': 'different-tag', 'previousTag': 'old-tag', 'source': {'revision': 'an-older-build'}, 'chartsSha256': self.recipe.chart_digest()} path = pathlib.Path(self.tmp.name)/'rollback.json' path.write_text(json.dumps(record)) - with patch.object(self.recipe, 'bound_cluster'), patch.object(spark, 'output', return_value=json.dumps(values)), patch.object(self.recipe, 'update') as update: + with patch.object(self.recipe, 'bound_cluster'), patch.object(tool, 'output', return_value=json.dumps(values)), patch.object(self.recipe, 'update') as update: with self.assertRaisesRegex(RuntimeError, 'Another image update'): self.recipe.rollback(path) update.assert_not_called() @@ -701,7 +701,7 @@ def test_rollback_rejects_missing_or_changed_charts_before_reading_release(self) record['chartsSha256'] = digest path.write_text(json.dumps(record)) with self.subTest(digest=digest), patch.object(self.recipe, 'bound_cluster'), \ - patch.object(spark, 'output') as output, patch.object(self.recipe, 'update') as update: + patch.object(tool, 'output') as output, patch.object(self.recipe, 'update') as update: with self.assertRaisesRegex(RuntimeError, 'chart fingerprint'): self.recipe.rollback(path) output.assert_not_called() @@ -752,7 +752,7 @@ def test_existing_attachment_blocks_fresh_stack_and_recovery(self): def test_explicit_release_and_repository_mapping(self): self.config['releases'] = {'stack': 'custom-front', 'operator': 'custom-operator', 'glm': 'custom-model'} self.config['images']['repositories'] = {'gateway': 'registry.example.com/another/gateway'} - recipe = spark.Recipe(self.config, self.tmp.name) + recipe = tool.Recipe(self.config, self.tmp.name) self.assertEqual(recipe.stack, 'custom-front') self.assertEqual(recipe.glm, 'custom-model') self.assertEqual(recipe.image('gateway', 'new'), 'registry.example.com/another/gateway:new') @@ -762,10 +762,10 @@ def test_existing_attachment_ownership_failure_makes_no_mutation(self): key = pathlib.Path(self.tmp.name)/'key' key.write_text('test-only-key') self.config['apiKeyFile'] = str(key) - recipe = spark.Recipe(self.config, self.tmp.name) + recipe = tool.Recipe(self.config, self.tmp.name) nodes = {'items': [{'metadata': {'name': name, 'uid': name}} for name in self.config['nodes'].values()]} foreign = {'metadata': {'annotations': {'meta.helm.sh/release-name': 'another-owner'}}} - with patch.object(spark, 'output', side_effect=[json.dumps(nodes), json.dumps(foreign)]), patch.object(recipe, 'helm_apply') as helm: + with patch.object(tool, 'output', side_effect=[json.dumps(nodes), json.dumps(foreign)]), patch.object(recipe, 'helm_apply') as helm: with self.assertRaisesRegex(RuntimeError, 'ownership'): recipe.attach_existing() helm.assert_not_called() diff --git a/deploy/helm/llm-routing/spark/tests/test_reinitialization.py b/deploy/helm/llm-routing/recipes/tests/test_reinitialization.py similarity index 100% rename from deploy/helm/llm-routing/spark/tests/test_reinitialization.py rename to deploy/helm/llm-routing/recipes/tests/test_reinitialization.py diff --git a/deploy/helm/llm-routing/spark/AGENTS.md b/deploy/helm/llm-routing/spark/AGENTS.md deleted file mode 100644 index 0220e2aa5..000000000 --- a/deploy/helm/llm-routing/spark/AGENTS.md +++ /dev/null @@ -1,9 +0,0 @@ -# Spark GLM recipe - -This entry point uses the checkout containing `spark.py`. Builds include local edits and record the current commit ID for debugging. Image updates and rollback require the routing chart fingerprint recorded at installation. Helm dependencies are built automatically. `--source-dir` selects another existing checkout. Keep test mocks under `tests` and out of the normal model deployment. - -Keep environment-specific values, credentials, kubeconfigs, generated TLS material and runtime evidence outside this repository. Pass an explicit context on every Kubernetes and Helm command. Do not change a live cluster while testing packaging. - -Run `python3 -m unittest discover -s tests -v` and the chart tests under `charts/gguf-backend/tests`. Run `python3 spark.py --config config.example.json --work-dir /tmp/spark-render render` for offline chart validation. A render does not establish a fresh-cluster deployment. - -Land runtime and chart fixes in their owning source directories with generated API files and tests. Never hand-edit generated CRDs or deepcopy code. From e45fd650c1e411df30cab70be0ca932e0cca97c0 Mon Sep 17 00:00:00 2001 From: Kristina Pathak Date: Tue, 6 Oct 2026 14:57:05 -0700 Subject: [PATCH 2/6] refactor(llm-routing)!: move GLM specifics into a recipe folder Model-specific settings were spread across backend.defaults.json and hard-coded names in recipe.py, client.py and the reinitialization checks. Move them into recipes/glm-5.3/ (recipe.json, model.lock.json, NOTICE) so the shared tool has no GLM knowledge and later recipes can be added as folders. The config gains a recipe key, defaulting to the only recipe present. Recipe server arguments may not set --device, --tensor-split or --rpc; the tool appends placement arguments itself. attach-existing finds the installed recipe from its InferenceEndpoint, and client.py takes the served model name from the recipe. BREAKING CHANGE: the model Helm release is configured as releases.model instead of releases.glm, and client.py requires --model. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: Kristina Pathak --- NOTICE | 1 + deploy/helm/llm-routing/recipes/NOTICE | 6 +- .../llm-routing/recipes/backend.defaults.json | 174 ------------- deploy/helm/llm-routing/recipes/client.py | 4 +- .../helm/llm-routing/recipes/cluster_setup.py | 10 +- .../llm-routing/recipes/config.example.json | 1 + .../helm/llm-routing/recipes/glm-5.3/NOTICE | 3 + .../recipes/{ => glm-5.3}/model.lock.json | 0 .../llm-routing/recipes/glm-5.3/recipe.json | 57 +++++ deploy/helm/llm-routing/recipes/recipe.py | 239 +++++++++++------- .../recipes/tests/test_attach_reuse.py | 4 +- .../llm-routing/recipes/tests/test_cli.py | 6 +- .../llm-routing/recipes/tests/test_client.py | 6 +- .../recipes/tests/test_discovery.py | 6 +- .../recipes/tests/test_load_resume.py | 26 +- .../llm-routing/recipes/tests/test_recipe.py | 22 +- .../recipes/tests/test_recipe_definitions.py | 106 ++++++++ .../recipes/tests/test_reinitialization.py | 2 +- 18 files changed, 364 insertions(+), 309 deletions(-) delete mode 100644 deploy/helm/llm-routing/recipes/backend.defaults.json create mode 100644 deploy/helm/llm-routing/recipes/glm-5.3/NOTICE rename deploy/helm/llm-routing/recipes/{ => glm-5.3}/model.lock.json (100%) create mode 100644 deploy/helm/llm-routing/recipes/glm-5.3/recipe.json create mode 100644 deploy/helm/llm-routing/recipes/tests/test_recipe_definitions.py diff --git a/NOTICE b/NOTICE index cce2ffddf..b8206fbf1 100644 --- a/NOTICE +++ b/NOTICE @@ -9,6 +9,7 @@ regenerate this file by running: The following third-party licenses are included in this repository: deploy/helm/container-cache/NOTICE + deploy/helm/llm-routing/recipes/glm-5.3/NOTICE deploy/helm/llm-routing/recipes/NOTICE deploy/helm/nats/NOTICE deploy/helm/openbao/NOTICE diff --git a/deploy/helm/llm-routing/recipes/NOTICE b/deploy/helm/llm-routing/recipes/NOTICE index 4c7b1cc7c..9d5ba4d17 100644 --- a/deploy/helm/llm-routing/recipes/NOTICE +++ b/deploy/helm/llm-routing/recipes/NOTICE @@ -1,7 +1,7 @@ # External artifacts -This recipe downloads and builds llama.cpp from a pinned public revision. llama.cpp is MIT licensed. Its LICENSE is retained in the generated runtime archive. Source: https://github.com/ggml-org/llama.cpp. +The recipe tool downloads and builds llama.cpp from a pinned public revision. llama.cpp is MIT licensed. Its LICENSE is retained in the generated runtime archive. Source: https://github.com/ggml-org/llama.cpp. -The model weights are external downloads and are not distributed in this repository. The GLM-5.3 model card identifies the custom `glm-5.3` license, outside this repository's standard license allowlist. Review and accept the upstream terms before downloading or redistributing weights. This recipe does not establish license compatibility for a downstream product. Model and terms: https://huggingface.co/unsloth/GLM-5.3-GGUF and https://huggingface.co/zai-org/GLM-5.3. +Model weights are external downloads and are not distributed in this repository. Each recipe folder has a NOTICE for its model terms. -NVIDIA container images retain their upstream terms. Access to a public image reference does not grant credentials or image redistribution rights. No third-party runtime binaries, images or model weights are embedded in this recipe. +NVIDIA container images retain their upstream terms. Access to a public image reference does not grant credentials or image redistribution rights. No third-party runtime binaries, images or model weights are embedded in this tool. diff --git a/deploy/helm/llm-routing/recipes/backend.defaults.json b/deploy/helm/llm-routing/recipes/backend.defaults.json deleted file mode 100644 index fa4db2264..000000000 --- a/deploy/helm/llm-routing/recipes/backend.defaults.json +++ /dev/null @@ -1,174 +0,0 @@ -{ - "phase": "serve", - "image": "nvcr.io/nvidia/vllm@sha256:67c617846180989a71e0f353c8b101e9808dc7c2638b60f890064e3d7b660282", - "runtimeClassName": "nvidia", - "targets": [ - { - "id": "leader", - "node": "model-0" - }, - { - "id": "worker", - "node": "model-1" - } - ], - "build": { - "revision": "f872b591121761ac7b2af18283bd99bdc092a63a", - "cudaArchitectures": "121a-real", - "parallel": 8 - }, - "runtime": { - "sha256": "", - "env": { - "GGML_CUDA_DISABLE_GRAPHS": "1", - "GGML_RPC_NO_RDMA": "1" - } - }, - "qualification": { - "attempt": 2 - }, - "model": { - "lock": { - "time": "2026-09-30T20:59:19.281184+00:00", - "model": "unsloth/GLM-5.3-GGUF", - "revision": "346b3591c7f28d1a23716f97a065ecf12ec14771", - "quantization": "UD-IQ2_M", - "weightFiles": 6, - "weightFileBytes": 238577585701, - "weightGB": 238.577585701, - "weightGiB": 222.1926913606003, - "files": [ - { - "rfilename": "UD-IQ2_M/GLM-5.3-UD-IQ2_M-00001-of-00006.gguf", - "blobId": "3ab26a3bed7c17b175bfa183aef34bd7463ce5b5", - "size": 9428677, - "lfs": { - "sha256": "88ec1b9e924bbfe2f5e210bcedb5146aefca41567cdc021c344b253cb733b21d", - "size": 9428677, - "pointerSize": 132 - } - }, - { - "rfilename": "UD-IQ2_M/GLM-5.3-UD-IQ2_M-00002-of-00006.gguf", - "blobId": "7f6200b94f282872e3e3c92b8de28d967b370cbd", - "size": 49223976800, - "lfs": { - "sha256": "fc036c1c96f30d0b8fa334339cd1f9c68b1a3ab7fbc9440ee92df2a65b2a5e11", - "size": 49223976800, - "pointerSize": 136 - } - }, - { - "rfilename": "UD-IQ2_M/GLM-5.3-UD-IQ2_M-00003-of-00006.gguf", - "blobId": "9e13ecaac451858e1530da77371562abfeaf8ba5", - "size": 49143176640, - "lfs": { - "sha256": "684b069f8e7a6b28ba04dc2a4d63757db4270bdad95ac931cd013394c638b4da", - "size": 49143176640, - "pointerSize": 136 - } - }, - { - "rfilename": "UD-IQ2_M/GLM-5.3-UD-IQ2_M-00004-of-00006.gguf", - "blobId": "ced5ebf4845d9304cd40878453a546ad57f133eb", - "size": 49143176640, - "lfs": { - "sha256": "10c735d4af5d923adff14edb9bfbef63e5dce6187720c5db08a45f630638eccb", - "size": 49143176640, - "pointerSize": 136 - } - }, - { - "rfilename": "UD-IQ2_M/GLM-5.3-UD-IQ2_M-00005-of-00006.gguf", - "blobId": "7a741f81c72712e13043e917664c1e53171ebc99", - "size": 49143176640, - "lfs": { - "sha256": "69ee92423d65605f3b7b07744492d42fd85c0fd898127cb23a44f213823f4cff", - "size": 49143176640, - "pointerSize": 136 - } - }, - { - "rfilename": "UD-IQ2_M/GLM-5.3-UD-IQ2_M-00006-of-00006.gguf", - "blobId": "688b8319e9ad9c7f0086f806b68941a4acbf324a", - "size": 41914650304, - "lfs": { - "sha256": "b8f96f7183eba431e047c52adab9091b3e465819c83c8c42ed94697b63507708", - "size": 41914650304, - "pointerSize": 136 - } - } - ], - "metadataURL": "https://huggingface.co/api/models/unsloth/GLM-5.3-GGUF?blobs=true" - }, - "servedName": "GLM-5.3-UD-IQ2_M", - "firstShard": "UD-IQ2_M/GLM-5.3-UD-IQ2_M-00001-of-00006.gguf", - "endpointName": "glm53-iq2", - "register": false, - "args": [ - "--n-gpu-layers", - "999", - "--device", - "CUDA0,RPC0", - "--split-mode", - "layer", - "--tensor-split", - "1,1", - "--load-mode", - "none", - "--lazy-mode", - "off", - "--fit", - "off", - "--ctx-size", - "2048", - "--parallel", - "1", - "--batch-size", - "128", - "--ubatch-size", - "128", - "--cache-ram", - "0", - "--flash-attn", - "on", - "--jinja", - "--reasoning", - "on", - "--reasoning-effort", - "low", - "--reasoning-format", - "deepseek", - "--chat-template-kwargs", - "{\"clear_thinking\":true}", - "--predict", - "512", - "--threads", - "8", - "--metrics" - ], - "canary": { - "timeoutSeconds": 180, - "intervalSeconds": 60 - } - }, - "rpc": { - "resources": { - "requests": { - "cpu": "2", - "memory": "110Gi", - "nvidia.com/gpu": 1 - }, - "limits": { - "cpu": "8", - "memory": "114Gi", - "nvidia.com/gpu": 1 - } - }, - "cache": { - "enabled": true, - "size": "160Gi", - "storageClassName": "local-path" - } - } -} diff --git a/deploy/helm/llm-routing/recipes/client.py b/deploy/helm/llm-routing/recipes/client.py index 216d2c0b2..7ac3007b2 100644 --- a/deploy/helm/llm-routing/recipes/client.py +++ b/deploy/helm/llm-routing/recipes/client.py @@ -1,7 +1,7 @@ #!/usr/bin/env python3 # SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""Chat with GLM and validate direct or authenticated gateway responses without logging keys.""" +"""Chat with the recipe model and validate direct or authenticated gateway responses without logging keys.""" import argparse import datetime import http.client @@ -168,7 +168,7 @@ def main(): parser.add_argument('--url', default='https://127.0.0.1:18443') parser.add_argument('--ca-file') parser.add_argument('--api-key-file') - parser.add_argument('--model', default='GLM-5.3-UD-IQ2_M') + parser.add_argument('--model', required=True, help='Served model name from the recipe definition.') parser.add_argument('--cluster-id', help='Require a healthy registration from this cluster during gateway discovery checks.') parser.add_argument('--mode', choices=['chat', 'verify', 'auth', 'discovery'], default='chat') parser.add_argument('--stream', action='store_true') diff --git a/deploy/helm/llm-routing/recipes/cluster_setup.py b/deploy/helm/llm-routing/recipes/cluster_setup.py index 066516165..077250bb1 100644 --- a/deploy/helm/llm-routing/recipes/cluster_setup.py +++ b/deploy/helm/llm-routing/recipes/cluster_setup.py @@ -128,12 +128,14 @@ def discover_config(context, namespace=None): return config -def validate_reinitialization(config): - """Inspect an uninstalled demo before allowing local progress to be reset.""" +def validate_reinitialization(config, backend): + """Inspect an uninstalled demo before allowing local progress to be reset. + + backend is the model Helm release name chosen by the recipe tool. + """ context, namespace = config['context'], config['namespace'] prefix = config['releasePrefix'] releases = config.get('releases', {}) - glm = releases.get('glm', prefix + '-glm') operator = releases.get('operator', prefix + '-operator') stack = releases.get('stack', prefix + '-stack') try: @@ -190,7 +192,7 @@ def owned(resource, release): 'Workloads still exist in the saved namespace. Finish uninstalling before init.') claims = items(context, 'persistentvolumeclaims', namespace=namespace) volumes = {item['metadata']['name']: item for item in items(context, 'persistentvolumes')} if claims else {} - expected = {glm + '-artifacts': (glm, 'leader'), glm + '-rpc-cache': (glm, 'worker'), + expected = {backend + '-artifacts': (backend, 'leader'), backend + '-rpc-cache': (backend, 'worker'), prefix + '-monitoring-metrics': (prefix + '-monitoring', 'control')} for claim in claims: name = claim['metadata']['name'] diff --git a/deploy/helm/llm-routing/recipes/config.example.json b/deploy/helm/llm-routing/recipes/config.example.json index c06af7fc2..1814306ee 100644 --- a/deploy/helm/llm-routing/recipes/config.example.json +++ b/deploy/helm/llm-routing/recipes/config.example.json @@ -1,6 +1,7 @@ { "context": "llm-routing-demo", "namespace": "llm-routing-poc", + "recipe": "glm-5.3", "releasePrefix": "llm-poc", "clusterId": "llm-routing-poc", "nodes": { diff --git a/deploy/helm/llm-routing/recipes/glm-5.3/NOTICE b/deploy/helm/llm-routing/recipes/glm-5.3/NOTICE new file mode 100644 index 000000000..44464abe7 --- /dev/null +++ b/deploy/helm/llm-routing/recipes/glm-5.3/NOTICE @@ -0,0 +1,3 @@ +# GLM-5.3 model terms + +The model weights are external downloads and are not distributed in this repository. The GLM-5.3 model card identifies the custom `glm-5.3` license, outside this repository's standard license allowlist. Review and accept the upstream terms before downloading or redistributing weights. This recipe does not establish license compatibility for a downstream product. Model and terms: https://huggingface.co/unsloth/GLM-5.3-GGUF and https://huggingface.co/zai-org/GLM-5.3. diff --git a/deploy/helm/llm-routing/recipes/model.lock.json b/deploy/helm/llm-routing/recipes/glm-5.3/model.lock.json similarity index 100% rename from deploy/helm/llm-routing/recipes/model.lock.json rename to deploy/helm/llm-routing/recipes/glm-5.3/model.lock.json diff --git a/deploy/helm/llm-routing/recipes/glm-5.3/recipe.json b/deploy/helm/llm-routing/recipes/glm-5.3/recipe.json new file mode 100644 index 000000000..e7925a1fe --- /dev/null +++ b/deploy/helm/llm-routing/recipes/glm-5.3/recipe.json @@ -0,0 +1,57 @@ +{ + "name": "glm-5.3", + "description": "GLM-5.3 UD-IQ2_M GGUF served by llama.cpp", + "releaseName": "glm", + "llamaCppRevision": "f872b591121761ac7b2af18283bd99bdc092a63a", + "servedName": "GLM-5.3-UD-IQ2_M", + "endpointName": "glm53-iq2", + "firstShard": "UD-IQ2_M/GLM-5.3-UD-IQ2_M-00001-of-00006.gguf", + "artifactsSize": "400Gi", + "rpcCacheSize": "160Gi", + "canary": { + "timeoutSeconds": 180, + "intervalSeconds": 60 + }, + "serverArgs": [ + "--n-gpu-layers", + "999", + "--split-mode", + "layer", + "--load-mode", + "none", + "--lazy-mode", + "off", + "--fit", + "off", + "--ctx-size", + "2048", + "--parallel", + "1", + "--batch-size", + "128", + "--ubatch-size", + "128", + "--cache-ram", + "0", + "--flash-attn", + "on", + "--jinja", + "--reasoning", + "on", + "--reasoning-effort", + "low", + "--reasoning-format", + "deepseek", + "--chat-template-kwargs", + "{\"clear_thinking\":true}", + "--predict", + "512", + "--threads", + "8", + "--metrics" + ], + "runtimeEnv": { + "GGML_CUDA_DISABLE_GRAPHS": "1", + "GGML_RPC_NO_RDMA": "1" + } +} diff --git a/deploy/helm/llm-routing/recipes/recipe.py b/deploy/helm/llm-routing/recipes/recipe.py index da1e341a1..b092bf9ac 100644 --- a/deploy/helm/llm-routing/recipes/recipe.py +++ b/deploy/helm/llm-routing/recipes/recipe.py @@ -25,7 +25,6 @@ import gateway_access import cluster_setup import console_output -MODEL = json.loads((HERE/'model.lock.json').read_text()) COMPONENTS = {'gateway': 'src/invocation-plane-services/llm-api-gateway', 'router': 'src/libraries/rust/stargate', 'pylon': 'src/libraries/rust/stargate', 'operator': 'src/compute-plane-services/pylon-operator'} @@ -52,21 +51,62 @@ def output(command, **kwargs): return console_output.output(command, **kwargs) +RECIPE_KEYS = ('name', 'releaseName', 'llamaCppRevision', 'servedName', 'endpointName', 'firstShard', + 'artifactsSize', 'rpcCacheSize', 'canary', 'serverArgs', 'runtimeEnv') +# The tool derives these from nodes.model; a recipe that sets them would conflict. +PLACEMENT_FLAGS = ('--device', '-dev', '--tensor-split', '-ts', '--rpc') +QUALIFICATION_ATTEMPT = 2 +NAME = r'[a-z0-9]([-a-z0-9]*[a-z0-9])?' + + +def available_recipes(): + return sorted(path.parent.name for path in HERE.glob('*/recipe.json')) + + +def recipe_name(c): + name = c.get('recipe') + if name is None: + names = available_recipes() + require(len(names) == 1, 'Set recipe in the configuration. Available: ' + ', '.join(names)) + name = names[0] + return name + + +def load_recipe(name): + names = available_recipes() + require(name in names, 'Unknown recipe: ' + str(name) + '. Available: ' + ', '.join(names)) + folder = HERE/name + definition = json.loads((folder/'recipe.json').read_text()) + missing = [key for key in RECIPE_KEYS if key not in definition] + require(not missing, 'Recipe ' + name + ' is missing: ' + ', '.join(missing)) + require(definition['name'] == name, 'Recipe name must match its folder: ' + name) + require(re.fullmatch(NAME, definition['releaseName']) is not None, 'Invalid releaseName in recipe ' + name) + flags = sorted({arg.split('=', 1)[0] for arg in definition['serverArgs']} & set(PLACEMENT_FLAGS)) + require(not flags, 'Recipe ' + name + ' must not set placement arguments (' + ', '.join(flags) + + '). The tool derives them from nodes.model.') + lock = json.loads((folder/'model.lock.json').read_text()) + require(definition['firstShard'] in [item['rfilename'] for item in lock['files']], + 'Recipe ' + name + ' firstShard is not in its model lock.') + definition['lock'] = lock + return definition + + def validate(c): + load_recipe(recipe_name(c)) for key in ('context', 'namespace', 'releasePrefix', 'clusterId', 'storageClass', 'runtimeClass'): require(isinstance(c.get(key), str) and bool(c[key].strip()), key + ' must be explicit.') for key in ('namespace', 'releasePrefix', 'clusterId'): require(re.fullmatch(r'[a-z0-9]([-a-z0-9]*[a-z0-9])?', c[key]) is not None, 'Invalid ' + key) require(len(c['releasePrefix']) <= 30, 'releasePrefix must be at most 30 characters.') require(all(c['nodes'].get(role) for role in ('leader', 'worker', 'control')), 'All three placement roles are required.') - require(c['nodes']['leader'] != c['nodes']['worker'], 'GLM needs exactly two distinct GPU nodes.') + require(c['nodes']['leader'] != c['nodes']['worker'], 'The model needs two distinct GPU nodes.') require(c['images']['pullPolicy'] in ('Never', 'IfNotPresent', 'Always'), 'Invalid pull policy.') require(re.fullmatch(r'[A-Za-z0-9_][A-Za-z0-9_.-]{0,127}', c['images']['tag']) is not None, 'Invalid image tag.') require('/' in c['images']['prefix'] and not c['images']['prefix'].endswith('/'), 'Use registry/path as image prefix.') require(not c['images'].get('pullSecrets'), 'Pylon does not propagate image-pull secrets. Use nodes with registry access or pre-import all application images.') require(c.get('caConfigMap'), 'caConfigMap is required for verified QUIC and client TLS.') - require(not c.get('retainedModels'), 'Verification targets GLM. Remove retainedModels from the configuration.') - require(not c.get('testFixture'), 'The recipe deploys GLM. Remove testFixture from the configuration.') + require(not c.get('retainedModels'), 'Verification targets the recipe model. Remove retainedModels from the configuration.') + require(not c.get('testFixture'), 'The recipe deploys its own model. Remove testFixture from the configuration.') @@ -175,18 +215,23 @@ def placement(obj): require(owner(router) == stack, 'Gateway and router belong to different releases.') control = placement(gateway) require(placement(router) == control, 'This recipe requires gateway and router on the same control node.') - endpoint = get('inferenceendpoint', 'glm53-iq2') - glm = owner(endpoint) - require(endpoint['spec']['service']['name'] == glm and endpoint['spec']['modelName'] == 'GLM-5.3-UD-IQ2_M', - 'Existing endpoint is not the GLM deployment supported by this recipe.') - leader = get('deployment', glm) - worker = get('deployment', glm+'-rpc-worker') - service = get('service', glm) - require(all(owner(d) == glm for d in (leader, worker, service)), 'Unexpected GLM resource ownership.') - backend = values(glm) + endpoints = json.loads(output(kc+['get', 'inferenceendpoints', '-o', 'json']))['items'] + require(len(endpoints) == 1, 'Expected one InferenceEndpoint in namespace '+namespace+'.') + endpoint = endpoints[0] + matches = [name for name in available_recipes() + if (load_recipe(name)['endpointName'], load_recipe(name)['servedName']) + == (endpoint['metadata']['name'], endpoint['spec']['modelName'])] + require(len(matches) == 1, 'Existing endpoint does not match a recipe in this checkout: '+endpoint['metadata']['name']) + model = owner(endpoint) + require(endpoint['spec']['service']['name'] == model, 'Existing endpoint does not serve its own Helm release.') + leader = get('deployment', model) + worker = get('deployment', model+'-rpc-worker') + service = get('service', model) + require(all(owner(d) == model for d in (leader, worker, service)), 'Unexpected model resource ownership.') + backend = values(model) nodes = {'control': control, 'leader': placement(leader), 'worker': placement(worker)} targets = {t['id']: t['node'] for t in backend['targets']} - require(all(targets.get(role) == nodes[role] for role in ('leader', 'worker')), 'GLM placement differs from Helm values.') + require(all(targets.get(role) == nodes[role] for role in ('leader', 'worker')), 'Model placement differs from Helm values.') releases = json.loads(output(hm+['list', '-o', 'json'])) operators = [] for release in releases: @@ -226,7 +271,7 @@ def placement(obj): importers.append((release['name'], config)) require(len(importers) <= 1, 'Multiple image import configurations match the control node.') containerd = None - prefix = glm.removesuffix('-glm')[:30] + prefix = model.removesuffix('-'+load_recipe(matches[0])['releaseName'])[:30] if importers: name, config = importers[0] require(name.endswith('-images'), 'Image importer release must end in -images.') @@ -237,8 +282,8 @@ def placement(obj): if '/' not in image_prefix: image_prefix += '/attached' # TLS private material and existing caller/cluster key hashes are intentionally omitted. - config = {'context': context, 'namespace': namespace, 'releasePrefix': prefix, 'clusterId': v['clusterId'], - 'releases': {'stack': stack, 'operator': operator, 'glm': glm}, 'nodes': nodes, + config = {'context': context, 'namespace': namespace, 'recipe': matches[0], 'releasePrefix': prefix, 'clusterId': v['clusterId'], + 'releases': {'stack': stack, 'operator': operator, 'model': model}, 'nodes': nodes, 'storageClass': backend['artifacts']['storageClassName'], 'runtimeClass': backend['runtimeClassName'], 'images': {'prefix': image_prefix, 'tag': image['tag'], 'pullPolicy': image['pullPolicy'], 'pullSecrets': [], 'repositories': images}, @@ -265,11 +310,12 @@ def __init__(self, config, work, source=None): self.identity = identity self.kc = ['kubectl', '--context', config['context'], '-n', config['namespace']] self.hm = ['helm', '--kube-context', config['context'], '-n', config['namespace']] + self.definition = load_recipe(recipe_name(config)) releases = config.get('releases', {}) - self.glm = releases.get('glm', config['releasePrefix'] + '-glm') + self.backend = releases.get('model', config['releasePrefix'] + '-' + self.definition['releaseName']) self.operator = releases.get('operator', config['releasePrefix'] + '-operator') self.stack = releases.get('stack', config['releasePrefix'] + '-stack') - for name in (self.glm, self.operator, self.stack): + for name in (self.backend, self.operator, self.stack): require(re.fullmatch(r'[a-z0-9]([-a-z0-9]*[a-z0-9])?', name) is not None and len(name) <= 53, 'Invalid release name.') def reinitialize(self, config_path): @@ -277,7 +323,7 @@ def reinitialize(self, config_path): 'A temporary gateway key still needs cleanup. No progress was reset.') if self.state.get('inventory'): self.bound_cluster() - cluster_setup.validate_reinitialization(self.c) + cluster_setup.validate_reinitialization(self.c, self.backend) if self.state_path.exists(): require(json.loads(self.state_path.read_text()) == self.state, 'Progress changed during inspection. Stop concurrent recipe commands and retry init.') @@ -337,20 +383,28 @@ def prepare(self): print('Using source checkout:', self.source) def backend_values(self, phase='serve', register=False, render=False): - values = json.loads((HERE/'backend.defaults.json').read_text()) - values.update(phase=phase, image=self.c['runtimeImage'], runtimeClassName=self.c['runtimeClass']) - values['targets'] = [{'id': r, 'node': self.c['nodes'][r]} for r in ('leader', 'worker')] - values['artifacts'] = {'storageClassName': self.c['storageClass'], 'size': '400Gi'} - values['rpc']['cache'].update(storageClassName=self.c['storageClass'], enabled=phase == 'serve') - if phase != 'serve': - values['rpc']['resources'] = {'requests': {'cpu': '2', 'memory': '2Gi', 'nvidia.com/gpu': 1}, - 'limits': {'cpu': '8', 'memory': '8Gi', 'nvidia.com/gpu': 1}} - values['runtime']['sha256'] = self.state.get('runtimeSha256', 'a'*64 if render else '') - values['model']['lock'] = MODEL - values['model']['register'] = register - values['qualification']['attempt'] = self.state.get('qualificationAttempt', values['qualification']['attempt']) - values.setdefault('chain', {}).update(runtimeRelease=self.glm, artifactClaim=self.glm+'-artifacts', attempt=self.state.get('chainAttempt', 1)) - return values + d = self.definition + if phase == 'serve': + rpc_resources = {'requests': {'cpu': '2', 'memory': '110Gi', 'nvidia.com/gpu': 1}, + 'limits': {'cpu': '8', 'memory': '114Gi', 'nvidia.com/gpu': 1}} + else: + rpc_resources = {'requests': {'cpu': '2', 'memory': '2Gi', 'nvidia.com/gpu': 1}, + 'limits': {'cpu': '8', 'memory': '8Gi', 'nvidia.com/gpu': 1}} + return { + 'phase': phase, 'image': self.c['runtimeImage'], 'runtimeClassName': self.c['runtimeClass'], + 'targets': [{'id': r, 'node': self.c['nodes'][r]} for r in ('leader', 'worker')], + 'artifacts': {'storageClassName': self.c['storageClass'], 'size': d['artifactsSize']}, + 'build': {'revision': d['llamaCppRevision'], 'cudaArchitectures': '121a-real', 'parallel': 8}, + 'runtime': {'sha256': self.state.get('runtimeSha256', 'a'*64 if render else ''), 'env': dict(d['runtimeEnv'])}, + 'qualification': {'attempt': self.state.get('qualificationAttempt', QUALIFICATION_ATTEMPT)}, + 'model': {'lock': d['lock'], 'servedName': d['servedName'], 'firstShard': d['firstShard'], + 'endpointName': d['endpointName'], 'register': register, 'canary': dict(d['canary']), + 'args': d['serverArgs'] + ['--device', 'CUDA0,RPC0', '--tensor-split', '1,1']}, + 'rpc': {'resources': rpc_resources, + 'cache': {'enabled': phase == 'serve', 'size': d['rpcCacheSize'], 'storageClassName': self.c['storageClass']}}, + 'chain': {'runtimeRelease': self.backend, 'artifactClaim': self.backend+'-artifacts', + 'attempt': self.state.get('chainAttempt', 1)}, + } def operator_values(self): return {'fullnameOverride': self.operator, 'clusterId': self.c['clusterId'], @@ -422,12 +476,12 @@ def attach_existing(self): if self.state and not self.state.get('attachedExisting'): self.bound_cluster() require(all(self.state.get(phase) for phase in ('stack', 'serve', 'registered')), - 'Saved installation is incomplete. Finish installing and registering GLM before attaching.') + 'Saved installation is incomplete. Finish installing and registering model before attaching.') live = discover_config(self.c['context'], self.c['namespace']) for field in ('context', 'namespace', 'clusterId', 'nodes', 'runtimeClass', 'runtimeImage', 'storageClass', 'caConfigMap'): require(live[field] == self.c[field], 'Existing installation differs from saved configuration: '+field) - require(live['releases'] == {'stack': self.stack, 'operator': self.operator, 'glm': self.glm}, + require(live['releases'] == {'stack': self.stack, 'operator': self.operator, 'model': self.backend}, 'Existing Helm releases differ from the saved installation.') for component in COMPONENTS: require(live['images']['repositories'][component] == self.repository(component), @@ -438,13 +492,13 @@ def attach_existing(self): return if self.state: self.bound_cluster() - require(set(self.c.get('releases', {})) >= {'stack', 'operator', 'glm'}, 'Set explicit releases.stack, releases.operator and releases.glm.') + require(set(self.c.get('releases', {})) >= {'stack', 'operator', 'model'}, 'Set explicit releases.stack, releases.operator and releases.model.') key = pathlib.Path(self.c['apiKeyFile']).expanduser().resolve(strict=True) if self.c.get('apiKeyFile') else None nodes = json.loads(output(self.kc+['get', 'nodes', '-o', 'json']))['items'] names = {n['metadata']['name'] for n in nodes} require(set(self.c['nodes'].values()) <= names, 'Configured placement nodes do not exist.') owners = {'llm-api-gateway': self.stack, 'llm-request-router': self.stack, - self.operator: self.operator, self.glm: self.glm, self.glm+'-rpc-worker': self.glm} + self.operator: self.operator, self.backend: self.backend, self.backend+'-rpc-worker': self.backend} for deployment, release in owners.items(): obj = json.loads(output(self.kc+['get', 'deployment', deployment, '-o', 'json'])) annotations = obj['metadata'].get('annotations', {}) @@ -454,8 +508,8 @@ def attach_existing(self): for component, chart, service in [('gateway', 'llm-api-gateway', 'llmApiGateway'), ('router', 'llm-request-router', 'llmRequestRouter')]: live = values[chart][service]['image'] require(live['registry']+'/'+live['repository'] == self.repository(component), 'Existing image repository differs: '+component) - endpoint = json.loads(output(self.kc+['get', 'inferenceendpoint', 'glm53-iq2', '-o', 'json'])) - require(endpoint['spec']['service']['name'] == self.glm and endpoint['spec']['modelName'] == 'GLM-5.3-UD-IQ2_M', 'Existing GLM endpoint differs from the recipe.') + endpoint = json.loads(output(self.kc+['get', 'inferenceendpoint', self.definition['endpointName'], '-o', 'json'])) + require(endpoint['spec']['service']['name'] == self.backend and endpoint['spec']['modelName'] == self.definition['servedName'], 'Existing model endpoint differs from the recipe.') ca = json.loads(output(self.kc+['get', 'configmap', self.c['caConfigMap'], '-o', 'json']))['data']['ca.crt'] save(self.work/'ca.crt', ca) self.stamp('attachedExisting') @@ -474,7 +528,7 @@ def bound_cluster(self): require(all(current.get(name) == old.get(name) for name in self.c['nodes'].values()), 'The selected node identities changed or the context points to another cluster.') def logs(self, component, release=None, job=None): - selector = 'app.kubernetes.io/instance='+(release or self.glm)+',app.kubernetes.io/component='+component + selector = 'app.kubernetes.io/instance='+(release or self.backend)+',app.kubernetes.io/component='+component # Job labels live on pod templates in these charts, so select pods. pods = json.loads(output(self.kc+['get', 'pods', '-l', selector, '-o', 'json']))['items'] if job: @@ -498,13 +552,13 @@ def logs(self, component, release=None, job=None): def retry_qualification(self): require(not self.state.get('qualify'), 'Qualification already passed. Retry is for an unsuccessful qualification phase.') - prefixes = {'qualificationAttempt': self.glm+'-qualify-', 'chainAttempt': self.glm+'-chain-'} + prefixes = {'qualificationAttempt': self.backend+'-qualify-', 'chainAttempt': self.backend+'-chain-'} jobs = json.loads(output(self.kc+['get', 'jobs', '-o', 'json']))['items'] jobs = [job for job in jobs if any(re.fullmatch(re.escape(prefix)+r'\d+', job['metadata']['name']) for prefix in prefixes.values())] require(bool(jobs), 'No qualification or chain Jobs to retry.') for job in jobs: name = job['metadata']['name'] - release = self.glm+'-chain' if name.startswith(prefixes['chainAttempt']) else self.glm + release = self.backend+'-chain' if name.startswith(prefixes['chainAttempt']) else self.backend owner = job['metadata'].get('annotations', {}) require(owner.get('meta.helm.sh/release-name') == release and owner.get('meta.helm.sh/release-namespace') == self.c['namespace'], 'Unexpected Job ownership: '+name) require(any(c['type'] in ('Complete', 'Failed') and c['status'] == 'True' for c in job.get('status', {}).get('conditions', [])), 'Job is still active: '+name) @@ -533,51 +587,51 @@ def retry_qualification(self): def resume_load(self): """Recover the local load checkpoint without changing an already deployed model.""" def release(): - item = json.loads(output(self.hm+['status', self.glm, '-o', 'json'], timeout=45)) - require(item.get('name') == self.glm and item.get('namespace') == self.c['namespace'], - 'Unexpected GLM Helm release identity.') + item = json.loads(output(self.hm+['status', self.backend, '-o', 'json'], timeout=45)) + require(item.get('name') == self.backend and item.get('namespace') == self.c['namespace'], + 'Unexpected model Helm release identity.') require(item.get('info', {}).get('status') == 'deployed', - 'GLM Helm release is '+str(item.get('info', {}).get('status'))+ + 'Model Helm release is '+str(item.get('info', {}).get('status'))+ '. Resolve the Helm operation, then rerun load.') return item['version'] revision = release() - values = json.loads(output(self.hm+['get', 'values', self.glm, '--revision', str(revision), '-o', 'json'], timeout=45)) + values = json.loads(output(self.hm+['get', 'values', self.backend, '--revision', str(revision), '-o', 'json'], timeout=45)) if values.get('phase') != 'serve': - require(values.get('phase') == 'download', 'GLM Helm release is not at the completed download or serve phase.') - existing = output(self.kc+['get', 'deployment', self.glm, '--ignore-not-found', '-o', 'json'], timeout=45) + require(values.get('phase') == 'download', 'Model Helm release is not at the completed download or serve phase.') + existing = output(self.kc+['get', 'deployment', self.backend, '--ignore-not-found', '-o', 'json'], timeout=45) require(not existing.strip(), 'A model Deployment already exists outside the expected serve phase.') - require(release() == revision, 'GLM Helm revision changed while checking load. Retry after the operation completes.') + require(release() == revision, 'Model Helm revision changed while checking load. Retry after the operation completes.') return False - require(values == self.backend_values('serve'), 'Deployed GLM values differ from this load configuration.') + require(values == self.backend_values('serve'), 'Deployed model values differ from this load configuration.') expected = { - ('Deployment', self.glm): ('leader', 'llama', 'model-server'), - ('Deployment', self.glm+'-rpc-worker'): ('worker', 'rpc', 'rpc-worker'), - ('Deployment', self.glm+'-artifacts'): ('leader', 'artifacts', 'artifacts'), - ('PersistentVolumeClaim', self.glm+'-artifacts'): None, - ('PersistentVolumeClaim', self.glm+'-rpc-cache'): None, + ('Deployment', self.backend): ('leader', 'llama', 'model-server'), + ('Deployment', self.backend+'-rpc-worker'): ('worker', 'rpc', 'rpc-worker'), + ('Deployment', self.backend+'-artifacts'): ('leader', 'artifacts', 'artifacts'), + ('PersistentVolumeClaim', self.backend+'-artifacts'): None, + ('PersistentVolumeClaim', self.backend+'-rpc-cache'): None, } names = [('deployment/' if kind == 'Deployment' else 'pvc/')+name for kind, name in expected] def ready_resources(): items = json.loads(output(self.kc+['get', *names, '-o', 'json'], timeout=45))['items'] require({(item.get('kind'), item['metadata']['name']) for item in items} == set(expected) - and len(items) == len(expected), 'Missing GLM resources while resuming load.') + and len(items) == len(expected), 'Missing model resources while resuming load.') identities = {} for item in items: meta, spec, status = item['metadata'], item['spec'], item.get('status', {}) name, kind = meta['name'], item['kind'] owner = meta.get('annotations', {}) require(meta.get('namespace') == self.c['namespace'] and not meta.get('deletionTimestamp') - and owner.get('meta.helm.sh/release-name') == self.glm + and owner.get('meta.helm.sh/release-name') == self.backend and owner.get('meta.helm.sh/release-namespace') == self.c['namespace'] and meta.get('labels', {}).get('app.kubernetes.io/managed-by') == 'Helm', - 'Unexpected GLM resource ownership: '+name) - require(meta.get('uid'), 'Missing GLM resource identity: '+name) + 'Unexpected model resource ownership: '+name) + require(meta.get('uid'), 'Missing model resource identity: '+name) if kind == 'PersistentVolumeClaim': require(status.get('phase') == 'Bound' and spec.get('volumeName') and spec.get('storageClassName') == self.c['storageClass'], - 'GLM storage is not bound as configured: '+name) + 'Model storage is not bound as configured: '+name) identities[kind+'/'+name] = [meta['uid'], spec['volumeName']] continue generation = meta.get('generation') @@ -586,48 +640,48 @@ def ready_resources(): and all(status.get(field, 0) == 1 for field in ('replicas', 'updatedReplicas', 'readyReplicas', 'availableReplicas')) and not status.get('unavailableReplicas', 0), - 'GLM Deployment is not ready at its current generation: '+name+'. Wait, then rerun load.') + 'Model Deployment is not ready at its current generation: '+name+'. Wait, then rerun load.') role, container, component = expected[(kind, name)] template = spec['template'] - labels = {'app.kubernetes.io/instance': self.glm, 'app.kubernetes.io/component': component} + labels = {'app.kubernetes.io/instance': self.backend, 'app.kubernetes.io/component': component} require(all(template['metadata'].get('labels', {}).get(key) == value for key, value in labels.items()) and spec.get('selector', {}).get('matchLabels') == labels, - 'Unexpected GLM Deployment selector: '+name) + 'Unexpected model Deployment selector: '+name) pod = template['spec'] require(pod.get('nodeSelector', {}).get('kubernetes.io/hostname') == self.c['nodes'][role], - 'GLM Deployment targets another node: '+name) + 'Model Deployment targets another node: '+name) containers = pod.get('containers', []) require(len(containers) == 1 and containers[0].get('name') == container and containers[0].get('image') == self.c['runtimeImage'], - 'GLM Deployment image differs: '+name) + 'Model Deployment image differs: '+name) volume = 'rpc-cache' if component == 'rpc-worker' else 'artifacts' volumes = {item['name']: item for item in pod.get('volumes', [])} mounts = {item['name']: item for item in containers[0].get('volumeMounts', [])} - require(volumes.get(volume, {}).get('persistentVolumeClaim', {}).get('claimName') == self.glm+'-'+volume + require(volumes.get(volume, {}).get('persistentVolumeClaim', {}).get('claimName') == self.backend+'-'+volume and mounts.get(volume, {}).get('mountPath') == '/'+volume, - 'GLM Deployment storage differs: '+name) + 'Model Deployment storage differs: '+name) if component == 'model-server': env = {item['name']: item.get('value') for item in containers[0].get('env', [])} require(env.get('FIRST_SHARD') == values['model']['firstShard'] and env.get('SERVED_MODEL') == values['model']['servedName'] - and env.get('RPC_ENDPOINT') == self.glm+'-rpc-worker:50052' + and env.get('RPC_ENDPOINT') == self.backend+'-rpc-worker:50052' and json.loads(env.get('SERVER_ARGS') or 'null') == values['model']['args'], - 'GLM model or RPC connection differs: '+name) + 'Model model or RPC connection differs: '+name) if component != 'artifacts': require(pod.get('runtimeClassName') == self.c['runtimeClass'] and template['metadata'].get('annotations', {}).get('checksum/runtime') == self.state['runtimeSha256'] and all(str(containers[0].get('resources', {}).get(field, {}).get('nvidia.com/gpu')) == '1' for field in ('requests', 'limits')), - 'GLM GPU or runtime configuration differs: '+name) + 'Model GPU or runtime configuration differs: '+name) identities[kind+'/'+name] = [meta['uid'], generation] return identities identities = ready_resources() require(ready_resources() == identities and release() == revision, - 'GLM resources or Helm revision changed while resuming load. Retry after the operation completes.') - save(self.work/'evidence'/'load-resume.json', {'release': self.glm, 'revision': revision, 'resources': identities}) + 'Model resources or Helm revision changed while resuming load. Retry after the operation completes.') + save(self.work/'evidence'/'load-resume.json', {'release': self.backend, 'revision': revision, 'resources': identities}) self.stamp('serve') - print('Resumed completed GLM load. Run verify-direct next.') + print('Resumed completed model load. Run verify-direct next.') return True def backend_phase(self, phase, retry=False): @@ -640,7 +694,7 @@ def backend_phase(self, phase, retry=False): if self.resume_load(): return for role in ('leader', 'worker'): - raw = output(self.kc+['exec', 'deploy/'+self.glm+'-rpc-'+role, '-c', 'rpc', '--', 'cat', '/proc/meminfo']) + raw = output(self.kc+['exec', 'deploy/'+self.backend+'-rpc-'+role, '-c', 'rpc', '--', 'cat', '/proc/meminfo']) available = next(int(line.split()[1])*1024 for line in raw.splitlines() if line.startswith('MemAvailable:')) require(available > 113*1024**3, 'Insufficient actual host memory on '+role) if phase == 'qualify': @@ -651,7 +705,7 @@ def backend_phase(self, phase, retry=False): self.stamp('qualify', False) values = self.backend_values(phase) timeout = {'preflight': '30m', 'build': '120m', 'qualify': '20m', 'download': '360m', 'serve': '70m'}[phase] - self.helm_apply(self.glm, HERE/'charts/gguf-backend', values, timeout, jobs=phase != 'serve') + self.helm_apply(self.backend, HERE/'charts/gguf-backend', values, timeout, jobs=phase != 'serve') if phase in ('preflight', 'build', 'download'): records = self.logs(phase) require(bool(records), phase+' did not record PASS.') @@ -661,13 +715,14 @@ def backend_phase(self, phase, retry=False): if phase == 'build': self.stamp('runtimeSha256', records[-1]['runtimeSHA256']) if phase == 'download': - require(records[-1]['verifiedBytes'] == MODEL['weightFileBytes'] and records[-1]['verifiedFiles'] == 6, 'Model download incomplete.') + lock = self.definition['lock'] + require(records[-1]['verifiedBytes'] == lock['weightFileBytes'] and records[-1]['verifiedFiles'] == len(lock['files']), 'Model download incomplete.') elif phase == 'qualify': - records = self.logs('qualification', job=self.glm+'-qualify-'+str(values['qualification']['attempt'])) + records = self.logs('qualification', job=self.backend+'-qualify-'+str(values['qualification']['attempt'])) require(bool(records), 'RPC qualification did not record PASS.') chain = self.backend_values('chain') - self.helm_apply(self.glm+'-chain', HERE/'charts/gguf-backend', chain, '15m', jobs=True) - records = self.logs('chain-check', job=self.glm+'-chain-'+str(chain['chain']['attempt'])) + self.helm_apply(self.backend+'-chain', HERE/'charts/gguf-backend', chain, '15m', jobs=True) + records = self.logs('chain-check', job=self.backend+'-chain-'+str(chain['chain']['attempt'])) require(bool(records), 'RPC chain check did not record PASS.') self.stamp(phase) @@ -699,15 +754,15 @@ def deploy_stack(self): def register(self): require(not self.state.get('attachedExisting'), 'Do not re-register or adopt an attached existing backend.') self.bound_cluster() - require(self.state.get('serve') and self.state.get('direct') and self.state.get('stack'), 'Load, directly verify GLM, and deploy the stack before registration.') - self.helm_apply(self.glm, HERE/'charts/gguf-backend', self.backend_values(register=True), '10m') + require(self.state.get('serve') and self.state.get('direct') and self.state.get('stack'), 'Load, directly verify the model, and deploy the stack before registration.') + self.helm_apply(self.backend, HERE/'charts/gguf-backend', self.backend_values(register=True), '10m') for condition in ('Ready', 'TransportReady', 'Registered'): - run(self.kc+['wait', 'inferenceendpoint/glm53-iq2', '--for=condition='+condition, '--timeout=300s']) + run(self.kc+['wait', 'inferenceendpoint/'+self.definition['endpointName'], '--for=condition='+condition, '--timeout=300s']) self.stamp('registered') @contextlib.contextmanager def forward(self, gateway, port): - service = 'llm-api-gateway' if gateway else self.glm + service = 'llm-api-gateway' if gateway else self.backend remote_port = '8080' if gateway else '8000' with socket.socket() as probe: probe.bind(('127.0.0.1', port)) @@ -734,7 +789,7 @@ def chat(self, prompt, stream, port): require(self.state.get('stack'), 'Attach to or deploy the stack first.') url = 'https://127.0.0.1:' + str(port) command = [sys.executable, str(HERE/'client.py'), '--mode', 'chat', '--url', url, - '--ca-file', str(self.work/'ca.crt')] + '--model', self.definition['servedName'], '--ca-file', str(self.work/'ca.crt')] with self.forward(True, port): existing_key = self.state['stack'].get('apiKeyFile') access = contextlib.nullcontext(existing_key) if existing_key else gateway_access.temporary_gateway_key(self, url) @@ -750,7 +805,7 @@ def verify(self, gateway, port): require(self.state.get('stack'), 'Deploy or attach to the stack first.') url = ('https' if gateway else 'http')+'://127.0.0.1:'+str(port) command = [sys.executable, str(HERE/'client.py'), '--mode', 'verify', '--url', url, - '--output', str(self.work/'evidence'/('gateway.json' if gateway else 'direct.json'))] + '--model', self.definition['servedName'], '--output', str(self.work/'evidence'/('gateway.json' if gateway else 'direct.json'))] if gateway: command += ['--ca-file', str(self.work/'ca.crt'), '--cluster-id', self.c['clusterId']] self.stamp('gateway' if gateway else 'direct', False) @@ -919,7 +974,7 @@ def import_images(self, archive, allow, component=None, tag=None): def update(self, component, tag): self.bound_cluster() self.source_check() - require(not self.state.get('attachedExisting') or self.state.get('gateway'), 'Verify GLM through the attached gateway before its first update.') + require(not self.state.get('attachedExisting') or self.state.get('gateway'), 'Verify the model through the attached gateway before its first update.') chart, service = {'gateway': ('llm-api-gateway', 'llmApiGateway'), 'router': ('llm-request-router', 'llmRequestRouter')}[component] require(re.fullmatch(r'[A-Za-z0-9_][A-Za-z0-9_.-]{0,127}', tag) is not None, 'Invalid image tag.') values = json.loads(output(self.hm+['get', 'values', self.stack, '-o', 'json'])) @@ -970,11 +1025,11 @@ def recovery(self, confirm, port): started = time.monotonic() observation = {'started': datetime.datetime.now(datetime.timezone.utc).isoformat(), 'interruptionObserved': False} try: - self.helm_apply(self.glm, HERE/'charts/gguf-backend', down, wait=False) + self.helm_apply(self.backend, HERE/'charts/gguf-backend', down, wait=False) time.sleep(15) down_pods = json.loads(output(self.kc+['get', 'pods', '-o', 'json'])) save(self.work/'evidence/recovery-down.json', down_pods) - require(not any(p['metadata']['name'].startswith(self.glm+'-rpc-worker-') and p['status']['phase'] == 'Running' for p in down_pods['items']), 'Worker is still running. Restore and inspect the rollout.') + require(not any(p['metadata']['name'].startswith(self.backend+'-rpc-worker-') and p['status']['phase'] == 'Running' for p in down_pods['items']), 'Worker is still running. Restore and inspect the rollout.') try: with self.forward(False, port): connection = http.client.HTTPConnection('127.0.0.1', port, timeout=10) @@ -990,7 +1045,7 @@ def recovery(self, confirm, port): observation['interruptionObserved'] = True observation['failureType'] = type(error).__name__ finally: - self.helm_apply(self.glm, HERE/'charts/gguf-backend', values, '70m') + self.helm_apply(self.backend, HERE/'charts/gguf-backend', values, '70m') observation['restoreSeconds'] = time.monotonic() - started save(self.work/'evidence/recovery.json', observation) require(observation['interruptionObserved'], 'Worker interruption was not demonstrated. Recovery restored the model but this test did not prove the failure path.') @@ -1002,7 +1057,7 @@ def render(self): renders = [('stack', self.source/'deploy/helm/llm-gateway-stack/llm-gateway-stack', self.stack_values('a'*64, 'b'*64, 'offline-demo-ui-placeholder')), ('operator', self.source/'deploy/helm/pylon-operator/pylon-operator', self.operator_values())] for phase in ('preflight', 'build', 'qualify', 'chain', 'download', 'serve'): - renders.append(('glm-'+phase, HERE/'charts/gguf-backend', self.backend_values(phase, register=phase=='serve', render=True))) + renders.append((self.definition['releaseName']+'-'+phase, HERE/'charts/gguf-backend', self.backend_values(phase, register=phase=='serve', render=True))) for name, chart, values in renders: path = self.work/'render'/(name+'-values.json') save(path, values) diff --git a/deploy/helm/llm-routing/recipes/tests/test_attach_reuse.py b/deploy/helm/llm-routing/recipes/tests/test_attach_reuse.py index 1ff30265d..a2c2d42d6 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_attach_reuse.py +++ b/deploy/helm/llm-routing/recipes/tests/test_attach_reuse.py @@ -26,7 +26,7 @@ def installation(self, explicit=False): self.count += 1 config = json.loads((HERE/'config.example.json').read_text()) if explicit: - config['releases'] = {'stack': 'custom-front', 'operator': 'custom-operator', 'glm': 'custom-model'} + config['releases'] = {'stack': 'custom-front', 'operator': 'custom-operator', 'model': 'custom-model'} config['images']['repositories'] = { name: 'registry.example.com/custom/'+name for name in tool.COMPONENTS} recipe = tool.Recipe(config, pathlib.Path(self.tmp.name)/str(self.count)) @@ -43,7 +43,7 @@ def installation(self, explicit=False): tool.save(recipe.work/'config.json', config) tool.save(recipe.state_path, recipe.state) live = copy.deepcopy(config) - live['releases'] = {'stack': recipe.stack, 'operator': recipe.operator, 'glm': recipe.glm} + live['releases'] = {'stack': recipe.stack, 'operator': recipe.operator, 'model': recipe.backend} live['images']['repositories'] = {name: recipe.repository(name) for name in tool.COMPONENTS} live['apiKeyFile'] = None return recipe, live diff --git a/deploy/helm/llm-routing/recipes/tests/test_cli.py b/deploy/helm/llm-routing/recipes/tests/test_cli.py index ba0f4dd0d..76d753db2 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_cli.py +++ b/deploy/helm/llm-routing/recipes/tests/test_cli.py @@ -30,7 +30,7 @@ def setUp(self): self.addCleanup(self.env.stop) self.config = json.loads((HERE/'config.example.json').read_text()) self.config['context'] = 'team-context' - self.config['releases'] = {'stack': 'test-stack', 'operator': 'test-operator', 'glm': 'test-glm'} + self.config['releases'] = {'stack': 'test-stack', 'operator': 'test-operator', 'model': 'test-glm'} self.work = tool.default_work_dir('team-context') def write_config(self, config=None, path=None): @@ -346,7 +346,7 @@ def test_init_archives_stale_progress_and_preserves_config_and_credentials(self) before = (self.work/'config.json').read_bytes() with patch.object(tool.cluster_setup, 'validate_reinitialization') as inspect, redirect_stdout(io.StringIO()): tool.main(['init']) - inspect.assert_called_once_with(self.config) + inspect.assert_called_once_with(self.config, self.config['releases']['model']) self.assertEqual((self.work/'config.json').read_bytes(), before) self.assertEqual((self.work/'api-key').read_text(), 'keep-me') self.assertFalse((self.work/'state.json').exists()) @@ -360,7 +360,7 @@ def test_init_reuses_config_when_progress_was_already_archived(self): self.write_config() with patch.object(tool.cluster_setup, 'validate_reinitialization') as inspect, redirect_stdout(io.StringIO()): tool.main(['--context', 'team-context', 'init']) - inspect.assert_called_once_with(self.config) + inspect.assert_called_once_with(self.config, self.config['releases']['model']) self.assertFalse((self.work/'state.json').exists()) self.assertEqual(list(self.work.glob('before-reinit-*')), []) diff --git a/deploy/helm/llm-routing/recipes/tests/test_client.py b/deploy/helm/llm-routing/recipes/tests/test_client.py index 70daa06d9..644e3c8a2 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_client.py +++ b/deploy/helm/llm-routing/recipes/tests/test_client.py @@ -14,7 +14,7 @@ spec = importlib.util.spec_from_file_location('recipe_client', HERE/'client.py') client = importlib.util.module_from_spec(spec) spec.loader.exec_module(client) -MODEL = json.loads((HERE/'backend.defaults.json').read_text())['model']['servedName'] +MODEL = json.loads((HERE/'glm-5.3/recipe.json').read_text())['servedName'] def discovery_documents(model=MODEL): @@ -181,7 +181,7 @@ def test_verify_cli_checks_glm_chat_streaming_discovery_and_auth(self): instance.auth.return_value = [{'status': status} for status in [401]*6+[200]] instance.discovery.return_value = {'registry': discovery_documents()[2]} stdout = io.StringIO() - with patch('sys.argv', ['client.py', '--mode', 'verify', '--cluster-id', 'expected-cluster']), patch.object(client, 'Client', return_value=instance), contextlib.redirect_stdout(stdout): + with patch('sys.argv', ['client.py', '--mode', 'verify', '--model', MODEL, '--cluster-id', 'expected-cluster']), patch.object(client, 'Client', return_value=instance), contextlib.redirect_stdout(stdout): client.main() calls = instance.completion.call_args_list self.assertEqual([call.args[0] for call in calls], [MODEL]*4) @@ -197,7 +197,7 @@ def test_verify_cli_checks_glm_chat_streaming_discovery_and_auth(self): def test_verify_cli_rejects_an_incorrect_glm_answer(self): instance = Mock(key=None) instance.completion.return_value = {'model': MODEL, 'content': '5'} - with patch('sys.argv', ['client.py', '--mode', 'verify']), patch.object(client, 'Client', return_value=instance): + with patch('sys.argv', ['client.py', '--mode', 'verify', '--model', MODEL]), patch.object(client, 'Client', return_value=instance): with self.assertRaisesRegex(RuntimeError, 'Incorrect answer'): client.main() instance.completion.assert_called_once() diff --git a/deploy/helm/llm-routing/recipes/tests/test_discovery.py b/deploy/helm/llm-routing/recipes/tests/test_discovery.py index 8da5a7620..6e1a6dc11 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_discovery.py +++ b/deploy/helm/llm-routing/recipes/tests/test_discovery.py @@ -64,6 +64,8 @@ def output(self, command): if '--all-namespaces' in command: self.assertLess(command.index('get'), command.index('--all-namespaces')) return json.dumps({'items': self.deployments}) + if 'inferenceendpoints' in command: + return json.dumps({'items': [self.resources[('inferenceendpoint', 'glm53-iq2')]]}) offset = command.index('get') return json.dumps(self.resources[tuple(command[offset+1:offset+3])]) @@ -76,7 +78,7 @@ def discover(self, namespace=None): def test_discovers_distinct_repositories_import_settings_and_omits_credentials(self): config = self.discover() self.assertEqual(config['context'], 'my-context') - self.assertEqual(config['releases'], {'stack': 'team-stack', 'operator': 'operator', 'glm': 'team-glm'}) + self.assertEqual(config['releases'], {'stack': 'team-stack', 'operator': 'operator', 'model': 'team-glm'}) self.assertEqual(config['images']['repositories'], {'gateway': 'registry.example.com/gateway', 'router': 'other.example.com/router', 'operator': 'operator.example.com/operator', 'pylon': 'worker.example.com/pylon'}) self.assertEqual(config['containerd']['runAsUser'], 4567) @@ -113,7 +115,7 @@ def test_source_metadata_is_informational_for_discovery(self): def test_model_service_foreign_ownership_is_rejected(self): self.resources['service', 'team-glm']['metadata']['annotations']['meta.helm.sh/release-name'] = 'someone-else' - with self.assertRaisesRegex(RuntimeError, 'GLM resource ownership'): + with self.assertRaisesRegex(RuntimeError, 'model resource ownership'): self.discover() def test_unrelated_operator_is_ignored(self): diff --git a/deploy/helm/llm-routing/recipes/tests/test_load_resume.py b/deploy/helm/llm-routing/recipes/tests/test_load_resume.py index aadfe52e4..26366c74b 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_load_resume.py +++ b/deploy/helm/llm-routing/recipes/tests/test_load_resume.py @@ -23,13 +23,13 @@ def setUp(self): self.recipe = tool.Recipe(config, self.tmp.name) self.recipe.state = {'download': True, 'runtimeSha256': 'a'*64} self.values = self.recipe.backend_values('serve') - self.release = {'name': self.recipe.glm, 'namespace': config['namespace'], + self.release = {'name': self.recipe.backend, 'namespace': config['namespace'], 'version': 6, 'info': {'status': 'deployed'}} self.resources = [] for suffix, role, container, component in [('', 'leader', 'llama', 'model-server'), ('-rpc-worker', 'worker', 'rpc', 'rpc-worker'), ('-artifacts', 'leader', 'artifacts', 'artifacts')]: - name = self.recipe.glm+suffix - labels = {'app.kubernetes.io/instance': self.recipe.glm, 'app.kubernetes.io/component': component} + name = self.recipe.backend+suffix + labels = {'app.kubernetes.io/instance': self.recipe.backend, 'app.kubernetes.io/component': component} self.resources.append({'kind': 'Deployment', 'metadata': self.metadata(name, generation=2), 'spec': {'replicas': 1, 'selector': {'matchLabels': labels}, 'template': { 'metadata': {'labels': labels, 'annotations': {'checksum/runtime': self.recipe.state['runtimeSha256']}}, @@ -42,14 +42,14 @@ def setUp(self): for resource in self.resources: pod = resource['spec']['template']['spec'] volume = 'rpc-cache' if resource['metadata']['name'].endswith('-rpc-worker') else 'artifacts' - pod['volumes'] = [{'name': volume, 'persistentVolumeClaim': {'claimName': self.recipe.glm+'-'+volume}}] + pod['volumes'] = [{'name': volume, 'persistentVolumeClaim': {'claimName': self.recipe.backend+'-'+volume}}] pod['containers'][0]['volumeMounts'] = [{'name': volume, 'mountPath': '/'+volume}] env = {'FIRST_SHARD': self.values['model']['firstShard'], 'SERVED_MODEL': self.values['model']['servedName'], - 'RPC_ENDPOINT': self.recipe.glm+'-rpc-worker:50052', 'SERVER_ARGS': json.dumps(self.values['model']['args'])} + 'RPC_ENDPOINT': self.recipe.backend+'-rpc-worker:50052', 'SERVER_ARGS': json.dumps(self.values['model']['args'])} self.resources[0]['spec']['template']['spec']['containers'][0]['env'] = [ {'name': key, 'value': value} for key, value in env.items()] for suffix in ('-artifacts', '-rpc-cache'): - name = self.recipe.glm+suffix + name = self.recipe.backend+suffix self.resources.append({'kind': 'PersistentVolumeClaim', 'metadata': self.metadata(name), 'spec': {'volumeName': name+'-volume', 'storageClassName': config['storageClass']}, 'status': {'phase': 'Bound'}}) @@ -62,7 +62,7 @@ def setUp(self): def metadata(self, name, **extra): return {'name': name, 'namespace': self.recipe.c['namespace'], 'uid': name+'-uid', - 'annotations': {'meta.helm.sh/release-name': self.recipe.glm, + 'annotations': {'meta.helm.sh/release-name': self.recipe.backend, 'meta.helm.sh/release-namespace': self.recipe.c['namespace']}, 'labels': {'app.kubernetes.io/managed-by': 'Helm'}, **extra} @@ -70,12 +70,12 @@ def command_output(self, command, **kwargs): self.commands.append(command) if 'exec' not in command: self.assertEqual(kwargs.get('timeout'), 45) - if command == self.recipe.hm+['status', self.recipe.glm, '-o', 'json']: + if command == self.recipe.hm+['status', self.recipe.backend, '-o', 'json']: self.status_calls += 1 return json.dumps(self.next_release if self.status_calls > 1 and self.next_release else self.release) - if command == self.recipe.hm+['get', 'values', self.recipe.glm, '--revision', '6', '-o', 'json']: + if command == self.recipe.hm+['get', 'values', self.recipe.backend, '--revision', '6', '-o', 'json']: return json.dumps(self.values) - if command == self.recipe.kc+['get', 'deployment', self.recipe.glm, '--ignore-not-found', '-o', 'json']: + if command == self.recipe.kc+['get', 'deployment', self.recipe.backend, '--ignore-not-found', '-o', 'json']: return self.existing if command[:len(self.recipe.kc)+1] == self.recipe.kc+['get']: self.resource_calls += 1 @@ -112,11 +112,11 @@ def test_completed_load_recovers_checkpoint_without_helm_or_preload_memory_check def test_first_load_keeps_memory_preflight_and_helm_apply(self): self.values = self.recipe.backend_values('download') - self.attempt().assert_called_once_with(self.recipe.glm, HERE/'charts/gguf-backend', + self.attempt().assert_called_once_with(self.recipe.backend, HERE/'charts/gguf-backend', self.recipe.backend_values('serve'), '70m', jobs=False) memory = [command for command in self.commands if 'exec' in command] self.assertEqual(len(memory), 2) - self.assertTrue(any('deploy/'+self.recipe.glm+'-rpc-leader' in command for command in memory)) + self.assertTrue(any('deploy/'+self.recipe.backend+'-rpc-leader' in command for command in memory)) self.assertTrue(self.recipe.state['serve']) def test_pending_and_failed_release_never_adopts_ready_resources(self): @@ -163,7 +163,7 @@ def test_missing_foreign_terminating_or_unready_resources_are_rejected(self): with self.subTest(index=index): self.resources = copy.deepcopy(original) mutate(self.resources) - self.attempt('GLM|Missing').assert_not_called() + self.attempt('[Mm]odel|Missing').assert_not_called() def test_revision_or_resource_replacement_during_check_is_rejected(self): self.next_release = copy.deepcopy(self.release) diff --git a/deploy/helm/llm-routing/recipes/tests/test_recipe.py b/deploy/helm/llm-routing/recipes/tests/test_recipe.py index 7620621d8..d4b69b1b6 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_recipe.py +++ b/deploy/helm/llm-routing/recipes/tests/test_recipe.py @@ -8,6 +8,7 @@ import json import os import pathlib +import shutil import stat import subprocess import tarfile @@ -63,6 +64,7 @@ def test_default_source_is_the_recipe_checkout_and_ignores_stale_work_source(sel stale.mkdir() (stale/'unrelated.txt').write_text('leave this old copy alone\n') recipe_path = seed/'deploy/helm/llm-routing/recipes' + shutil.copytree(HERE/'glm-5.3', recipe_path/'glm-5.3') with patch.object(tool, 'HERE', recipe_path), \ patch.object(tool, 'run') as run: recipe = tool.Recipe(self.config, self.recipe.work) @@ -425,7 +427,7 @@ def test_failed_verification_invalidates_previous_success(self): def test_attached_installation_requires_glm_verification_before_update(self): self.recipe.state = {'attachedExisting': True} with patch.object(self.recipe, 'bound_cluster'), patch.object(self.recipe, 'source_check'), patch.object(tool, 'run') as run, patch.object(tool, 'output') as output: - with self.assertRaisesRegex(RuntimeError, 'Verify GLM'): + with self.assertRaisesRegex(RuntimeError, 'Verify the model'): self.recipe.update('gateway', 'next-tag') run.assert_not_called() output.assert_not_called() @@ -439,8 +441,8 @@ def test_model_config_keeps_two_gpus_and_scoped_canary(self): self.assertEqual(len(values['model']['lock']['files']), 6) def qualification_job(self, suffix, condition='Failed', release=None): - release = release or self.recipe.glm - return {'metadata': {'name': self.recipe.glm+suffix, 'uid': suffix, + release = release or self.recipe.backend + return {'metadata': {'name': self.recipe.backend+suffix, 'uid': suffix, 'annotations': {'meta.helm.sh/release-name': release, 'meta.helm.sh/release-namespace': self.config['namespace']}}, 'status': {'conditions': [{'type': condition, 'status': 'True'}]}} @@ -452,7 +454,7 @@ def qualification_pod(self, name, job, phase='Succeeded'): def test_qualification_retry_archives_before_helm_and_persists_new_attempts(self): self.recipe.state = {'runtimeSha256': 'a'*64, 'download': True} jobs = [self.qualification_job('-qualify-5', 'Complete'), - self.qualification_job('-chain-3', 'Failed', self.recipe.glm+'-chain')] + self.qualification_job('-chain-3', 'Failed', self.recipe.backend+'-chain')] pods = [self.qualification_pod('old-qualification', jobs[0]), self.qualification_pod('failed-chain', jobs[1], 'Failed')] pods.append({'metadata': {'name': 'unrelated'}, 'status': {'phase': 'Running'}}) @@ -472,8 +474,8 @@ def check_archive(*args, **kwargs): self.recipe.backend_phase('qualify', retry=True) self.assertEqual(helm.call_args_list[0].args[2]['qualification']['attempt'], 6) self.assertEqual(helm.call_args_list[1].args[2]['chain']['attempt'], 4) - self.assertEqual(logs.call_args_list[0].kwargs['job'], self.recipe.glm+'-qualify-6') - self.assertEqual(logs.call_args_list[1].kwargs['job'], self.recipe.glm+'-chain-4') + self.assertEqual(logs.call_args_list[0].kwargs['job'], self.recipe.backend+'-qualify-6') + self.assertEqual(logs.call_args_list[1].kwargs['job'], self.recipe.backend+'-chain-4') resumed = tool.Recipe(self.config, self.tmp.name) self.assertTrue(resumed.state['qualify']) self.assertEqual(resumed.backend_values('qualify')['qualification']['attempt'], 6) @@ -570,7 +572,7 @@ def test_image_update_only_changes_selected_tag_and_preserves_other_pods(self): values['recipeSource'] = {'revision': 'an-older-build'} pods = [{'metadata': {'name': 'llm-api-gateway-old', 'uid': 'g1'}, 'status': {'phase': 'Running'}}, {'metadata': {'name': 'unrelated-workload', 'uid': 'u1'}, 'status': {'phase': 'Running'}}, - {'metadata': {'name': self.recipe.glm+'-leader', 'uid': 'm1'}, 'status': {'phase': 'Running'}}] + {'metadata': {'name': self.recipe.backend+'-leader', 'uid': 'm1'}, 'status': {'phase': 'Running'}}] after = copy.deepcopy(pods) after[0]['metadata']['uid'] = 'g2' with patch.object(self.recipe, 'source_check') as source, patch.object(self.recipe, 'bound_cluster'), patch.object(tool, 'output', side_effect=[json.dumps(values), json.dumps({'items': pods}), json.dumps({'items': after})]), patch.object(tool, 'run') as run: @@ -750,15 +752,15 @@ def test_existing_attachment_blocks_fresh_stack_and_recovery(self): helm.assert_not_called() def test_explicit_release_and_repository_mapping(self): - self.config['releases'] = {'stack': 'custom-front', 'operator': 'custom-operator', 'glm': 'custom-model'} + self.config['releases'] = {'stack': 'custom-front', 'operator': 'custom-operator', 'model': 'custom-model'} self.config['images']['repositories'] = {'gateway': 'registry.example.com/another/gateway'} recipe = tool.Recipe(self.config, self.tmp.name) self.assertEqual(recipe.stack, 'custom-front') - self.assertEqual(recipe.glm, 'custom-model') + self.assertEqual(recipe.backend, 'custom-model') self.assertEqual(recipe.image('gateway', 'new'), 'registry.example.com/another/gateway:new') def test_existing_attachment_ownership_failure_makes_no_mutation(self): - self.config['releases'] = {'stack': 'custom-front', 'operator': 'custom-operator', 'glm': 'custom-model'} + self.config['releases'] = {'stack': 'custom-front', 'operator': 'custom-operator', 'model': 'custom-model'} key = pathlib.Path(self.tmp.name)/'key' key.write_text('test-only-key') self.config['apiKeyFile'] = str(key) diff --git a/deploy/helm/llm-routing/recipes/tests/test_recipe_definitions.py b/deploy/helm/llm-routing/recipes/tests/test_recipe_definitions.py new file mode 100644 index 000000000..3eaf7bd3c --- /dev/null +++ b/deploy/helm/llm-routing/recipes/tests/test_recipe_definitions.py @@ -0,0 +1,106 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +import copy +import importlib.util +import json +import pathlib +import shutil +import tempfile +import unittest +from unittest.mock import patch + +HERE = pathlib.Path(__file__).resolve().parents[1] +spec = importlib.util.spec_from_file_location('recipe_definitions', HERE/'recipe.py') +tool = importlib.util.module_from_spec(spec) +spec.loader.exec_module(tool) + + +class RecipeDefinitionTests(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory(prefix='recipe-definitions-') + self.addCleanup(self.tmp.cleanup) + self.config = json.loads((HERE/'config.example.json').read_text()) + + def copy_recipe(self, name, **changes): + root = pathlib.Path(self.tmp.name)/'recipes' + shutil.copytree(HERE/'glm-5.3', root/name) + definition = json.loads((root/name/'recipe.json').read_text()) + definition.update(name=name, **changes) + (root/name/'recipe.json').write_text(json.dumps(definition)) + return root + + def test_glm_recipe_reads_its_folder_and_model_lock(self): + recipe = tool.load_recipe('glm-5.3') + self.assertEqual(recipe['servedName'], 'GLM-5.3-UD-IQ2_M') + self.assertEqual(recipe['endpointName'], 'glm53-iq2') + self.assertEqual(recipe['releaseName'], 'glm') + self.assertEqual(len(recipe['lock']['files']), recipe['lock']['weightFiles']) + self.assertEqual(recipe['firstShard'], recipe['lock']['files'][0]['rfilename']) + + def test_available_recipes_are_folders_with_a_definition(self): + self.assertEqual(tool.available_recipes(), ['glm-5.3']) + + def test_recipe_arguments_must_not_choose_placement(self): + base = json.loads((HERE/'glm-5.3/recipe.json').read_text())['serverArgs'] + for flag in ('--device', '--tensor-split', '--rpc'): + root = pathlib.Path(self.tmp.name)/flag.strip('-') + shutil.copytree(HERE/'glm-5.3', root/'bad') + definition = json.loads((root/'bad/recipe.json').read_text()) + definition.update(name='bad', serverArgs=base + [flag, 'x']) + (root/'bad/recipe.json').write_text(json.dumps(definition)) + with self.subTest(flag=flag), patch.object(tool, 'HERE', root), \ + self.assertRaisesRegex(RuntimeError, 'placement'): + tool.load_recipe('bad') + + def test_recipe_name_must_match_its_folder(self): + root = self.copy_recipe('renamed') + definition = json.loads((root/'renamed/recipe.json').read_text()) + definition['name'] = 'something-else' + (root/'renamed/recipe.json').write_text(json.dumps(definition)) + with patch.object(tool, 'HERE', root), self.assertRaisesRegex(RuntimeError, 'folder'): + tool.load_recipe('renamed') + + def test_unknown_recipe_is_rejected(self): + self.config['recipe'] = 'missing-model' + with self.assertRaisesRegex(RuntimeError, 'Unknown recipe'): + tool.validate(self.config) + + def test_recipe_defaults_to_the_only_available_recipe(self): + self.config.pop('recipe', None) + self.assertEqual(tool.recipe_name(self.config), 'glm-5.3') + + def test_recipe_must_be_explicit_when_several_exist(self): + root = self.copy_recipe('second-model') + shutil.copytree(HERE/'glm-5.3', root/'glm-5.3') + self.config.pop('recipe', None) + with patch.object(tool, 'HERE', root), self.assertRaisesRegex(RuntimeError, 'Set recipe'): + tool.recipe_name(self.config) + + def test_backend_values_take_model_settings_from_the_recipe(self): + recipe = tool.Recipe(copy.deepcopy(self.config), self.tmp.name) + definition = tool.load_recipe('glm-5.3') + values = recipe.backend_values('serve') + model = values['model'] + self.assertEqual(model['servedName'], definition['servedName']) + self.assertEqual(model['endpointName'], definition['endpointName']) + self.assertEqual(model['firstShard'], definition['firstShard']) + self.assertEqual(model['canary'], definition['canary']) + self.assertEqual(model['lock'], definition['lock']) + self.assertEqual(model['args'][:len(definition['serverArgs'])], definition['serverArgs']) + self.assertEqual(values['build']['revision'], definition['llamaCppRevision']) + self.assertEqual(values['runtime']['env'], definition['runtimeEnv']) + self.assertEqual(values['artifacts']['size'], definition['artifactsSize']) + self.assertEqual(values['rpc']['cache']['size'], definition['rpcCacheSize']) + + def test_model_release_name_comes_from_the_recipe(self): + recipe = tool.Recipe(copy.deepcopy(self.config), self.tmp.name) + self.assertEqual(recipe.backend, self.config['releasePrefix'] + '-glm') + + def test_explicit_model_release_overrides_the_recipe_default(self): + self.config['releases'] = {'model': 'custom-model'} + recipe = tool.Recipe(copy.deepcopy(self.config), self.tmp.name) + self.assertEqual(recipe.backend, 'custom-model') + + +if __name__ == '__main__': + unittest.main() diff --git a/deploy/helm/llm-routing/recipes/tests/test_reinitialization.py b/deploy/helm/llm-routing/recipes/tests/test_reinitialization.py index 763caed36..4c6c85bdb 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_reinitialization.py +++ b/deploy/helm/llm-routing/recipes/tests/test_reinitialization.py @@ -41,7 +41,7 @@ def validate(self, releases=None): before = copy.deepcopy(self.config) with patch.object(setup, 'items', side_effect=lambda ctx, resource, **kw: copy.deepcopy(self.resources[resource])), \ patch.object(setup.subprocess, 'check_output', return_value=json.dumps(releases or [])) as helm: - setup.validate_reinitialization(self.config) + setup.validate_reinitialization(self.config, self.config['releasePrefix'] + '-glm') command = helm.call_args.args[0] self.assertEqual(command[:5], ['helm', '--kube-context', self.config['context'], '-n', self.namespace]) self.assertIn('--pending', command) From be7538a4d4e1bb199076ba7435feb866db690e11 Mon Sep 17 00:00:00 2001 From: Kristina Pathak Date: Tue, 6 Oct 2026 16:10:42 -0700 Subject: [PATCH 3/6] feat(llm-routing)!: place models on one or more GPU nodes The recipe assumed exactly two GB10 model nodes plus a separate routing node. A GB300 holds the 222 GiB GLM model in one GPU, and a two-node cluster has no spare routing node. Placement is now a list, nodes.model, with nodes.control alongside. The backend chart renders one RPC worker and cache per node after the leader, and none for a single node. llama.cpp --device and --tensor-split are derived from the node count. The new gpu section records the GPU name, compute capability, memory and whether memory is shared with the CPU. The CUDA architectures, the InferenceEndpoint GPU product and pod memory are derived from it. The derived memory reproduces the previous 113 GiB check and 110Gi/114Gi limits for a GB10 split. init runs a short GPU probe pod per candidate node in a temporary namespace, picks the smallest node count that fits, and prefers a spare node for routing, sharing the leader when none exists. preflight checks the detected GPU and memory against the configuration before any download. qualify and recover adapt to single-node placements. BREAKING CHANGE: nodes.leader and nodes.worker are replaced by nodes.model, a gpu section is required, the rpc-leader and rpc-worker Deployments are now rpc-n0, rpc-n1, and the RPC cache claim is -rpc-cache-n1. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: Kristina Pathak --- .../charts/gguf-backend/files/qualify.py | 12 +- .../gguf-backend/files/rpc-gpu-check.cpp | 2 +- .../charts/gguf-backend/files/serve.py | 29 ++- .../charts/gguf-backend/templates/chain.yaml | 2 +- .../gguf-backend/templates/qualify.yaml | 2 +- .../gguf-backend/templates/runtime.yaml | 23 +- .../charts/gguf-backend/templates/serve.yaml | 6 +- .../recipes/charts/gguf-backend/values.yaml | 4 + .../helm/llm-routing/recipes/cluster_setup.py | 143 +++++++++--- .../llm-routing/recipes/config.example.json | 13 +- .../llm-routing/recipes/glm-5.3/recipe.json | 13 ++ deploy/helm/llm-routing/recipes/recipe.py | 177 ++++++++++----- deploy/helm/llm-routing/recipes/sizing.py | 72 ++++++ .../recipes/tests/test_attach_reuse.py | 7 +- .../llm-routing/recipes/tests/test_cli.py | 8 +- .../recipes/tests/test_cluster_setup.py | 111 +++++++++- .../recipes/tests/test_discovery.py | 22 +- .../recipes/tests/test_load_resume.py | 49 ++++- .../llm-routing/recipes/tests/test_recipe.py | 20 +- .../recipes/tests/test_reinitialization.py | 9 +- .../llm-routing/recipes/tests/test_sizing.py | 85 +++++++ .../recipes/tests/test_topology.py | 208 ++++++++++++++++++ 22 files changed, 861 insertions(+), 156 deletions(-) create mode 100644 deploy/helm/llm-routing/recipes/sizing.py create mode 100644 deploy/helm/llm-routing/recipes/tests/test_sizing.py create mode 100644 deploy/helm/llm-routing/recipes/tests/test_topology.py diff --git a/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/qualify.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/qualify.py index b8b220949..af2282f59 100644 --- a/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/qualify.py +++ b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/qualify.py @@ -6,8 +6,8 @@ import subprocess import time -endpoints = os.environ['RPC_ENDPOINTS'].split(',') -assert len(endpoints) == 2 +endpoints = [item for item in os.environ['RPC_ENDPOINTS'].split(',') if item] +assert endpoints, 'At least one RPC endpoint is required' for endpoint in endpoints: host, port = endpoint.rsplit(':', 1) for attempt in range(120): @@ -19,6 +19,10 @@ if attempt == 119: raise time.sleep(2) -subprocess.run(['/artifacts/runtime/test-rpc-multi-server', *endpoints], check=True, timeout=60) +tests = [] +if len(endpoints) > 1: + subprocess.run(['/artifacts/runtime/test-rpc-multi-server', *endpoints], check=True, timeout=60) + tests.append('upstream-rpc-buffer-isolation') subprocess.run(['/artifacts/runtime/rpc-gpu-check', *endpoints], check=True, timeout=120) -print(json.dumps({'result': 'PASS', 'transport': 'TCP', 'tests': ['upstream-rpc-buffer-isolation', 'two-gpu-f32-matmul-three-repeats']}), flush=True) +tests.append(str(len(endpoints)) + '-gpu-f32-matmul-three-repeats') +print(json.dumps({'result': 'PASS', 'transport': 'TCP', 'gpus': len(endpoints), 'tests': tests}), flush=True) diff --git a/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/rpc-gpu-check.cpp b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/rpc-gpu-check.cpp index 6bec54a6a..fc6d75710 100644 --- a/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/rpc-gpu-check.cpp +++ b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/rpc-gpu-check.cpp @@ -12,7 +12,7 @@ #include int main(int argc, char ** argv) { - if (argc != 3) throw std::runtime_error("Two RPC endpoints are required"); + if (argc < 2) throw std::runtime_error("At least one RPC endpoint is required"); constexpr int k = 256, m = 128, n = 64; std::vector a(k*m), b(k*n), expected(m*n), actual(m*n); for (int i = 0; i < k*m; ++i) a[i] = float(i % 17 - 8) / 8; diff --git a/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/serve.py b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/serve.py index e1aa1b2c2..d4a1c1441 100644 --- a/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/serve.py +++ b/deploy/helm/llm-routing/recipes/charts/gguf-backend/files/serve.py @@ -19,18 +19,23 @@ manifest = json.loads((root / 'runtime/build-manifest.json').read_text()) binary = root / 'runtime/llama-server' assert hashlib.file_digest(binary.open('rb'), 'sha256').hexdigest() == manifest['binaries']['llama-server'] -host, port = os.environ['RPC_ENDPOINT'].rsplit(':', 1) -for attempt in range(120): - try: - with socket.create_connection((host, int(port)), timeout=2): - pass - break - except OSError: - if attempt == 119: - raise - time.sleep(2) +# Empty for a single model node; otherwise one RPC worker endpoint per additional node. +endpoints = [item for item in os.environ.get('RPC_ENDPOINTS', '').split(',') if item] +for endpoint in endpoints: + host, port = endpoint.rsplit(':', 1) + for attempt in range(120): + try: + with socket.create_connection((host, int(port)), timeout=2): + pass + break + except OSError: + if attempt == 119: + raise + time.sleep(2) args = [str(binary), '--model', str(root / 'model' / os.environ['FIRST_SHARD']), - '--alias', os.environ['SERVED_MODEL'], '--host', '0.0.0.0', '--port', '8000', - '--rpc', os.environ['RPC_ENDPOINT'], *json.loads(os.environ['SERVER_ARGS'])] + '--alias', os.environ['SERVED_MODEL'], '--host', '0.0.0.0', '--port', '8000'] +if endpoints: + args += ['--rpc', ','.join(endpoints)] +args += json.loads(os.environ['SERVER_ARGS']) print(json.dumps({'verifiedModel': verified, 'runtimeRevision': manifest['revision'], 'command': args}), flush=True) raise SystemExit(supervise(args)) diff --git a/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/chain.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/chain.yaml index 3f6f07960..72b9d409d 100644 --- a/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/chain.yaml +++ b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/chain.yaml @@ -47,7 +47,7 @@ spec: - {name: GGML_RPC_NO_RDMA, value: '1'} - {name: GGML_SCHED_DEBUG, value: '1'} - {name: LLAMA_REVISION, value: {{ .Values.build.revision | quote }}} - - {name: RPC_ENDPOINTS, value: "{{ .Values.chain.runtimeRelease }}-rpc-leader:50052,{{ .Values.chain.runtimeRelease }}-rpc-worker:50052"} + - {name: RPC_ENDPOINTS, value: "{{ range $i, $target := .Values.targets }}{{ if $i }},{{ end }}{{ $.Values.chain.runtimeRelease }}-rpc-{{ $target.id }}:50052{{ end }}"} resources: requests: {cpu: '1', memory: 512Mi} limits: {cpu: '2', memory: 2Gi} diff --git a/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/qualify.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/qualify.yaml index 58a7492b9..35a08ed0c 100644 --- a/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/qualify.yaml +++ b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/qualify.yaml @@ -41,7 +41,7 @@ spec: - {name: NVIDIA_VISIBLE_DEVICES, value: none} - {name: NVIDIA_DRIVER_CAPABILITIES, value: 'compute,utility'} - {name: GGML_RPC_NO_RDMA, value: '1'} - - {name: RPC_ENDPOINTS, value: "{{ .Release.Name }}-rpc-leader:50052,{{ .Release.Name }}-rpc-worker:50052"} + - {name: RPC_ENDPOINTS, value: "{{ range $i, $target := .Values.targets }}{{ if $i }},{{ end }}{{ $.Release.Name }}-rpc-{{ $target.id }}:50052{{ end }}"} resources: requests: {cpu: '1', memory: 512Mi} limits: {cpu: '2', memory: 2Gi} diff --git a/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/runtime.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/runtime.yaml index e979f19a3..5a3209fbf 100644 --- a/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/runtime.yaml +++ b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/runtime.yaml @@ -1,21 +1,24 @@ {{/* SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. */}} {{/* SPDX-License-Identifier: Apache-2.0 */}} {{- if has .Values.phase (list "qualify" "download" "serve") }} +{{- $leader := (index (required "targets are required" .Values.targets) 0).id }} {{- if .Values.rpc.cache.enabled }} +{{- range $target := rest .Values.targets }} apiVersion: v1 kind: PersistentVolumeClaim metadata: - name: {{ .Release.Name }}-rpc-cache + name: {{ $.Release.Name }}-rpc-cache-{{ $target.id }} annotations: helm.sh/resource-policy: keep spec: accessModes: [ReadWriteOnce] - storageClassName: {{ .Values.rpc.cache.storageClassName | quote }} + storageClassName: {{ $.Values.rpc.cache.storageClassName | quote }} resources: requests: - storage: {{ .Values.rpc.cache.size }} + storage: {{ $.Values.rpc.cache.size }} --- {{- end }} +{{- end }} apiVersion: apps/v1 kind: Deployment metadata: @@ -79,7 +82,7 @@ spec: podSelector: matchLabels: {app.kubernetes.io/instance: {{ .Release.Name }}} matchExpressions: - - {key: app.kubernetes.io/component, operator: In, values: [artifacts, rpc-leader, rpc-worker]} + - {key: app.kubernetes.io/component, operator: In, values: [artifacts{{ range .Values.targets }}, rpc-{{ .id }}{{ end }}]} policyTypes: [Ingress] ingress: - from: @@ -101,7 +104,7 @@ data: fetch-runtime.py: | {{ .Files.Get "files/fetch-runtime.py" | indent 4 }} {{- range $target := .Values.targets }} -{{- if or (ne $.Values.phase "serve") (eq $target.id "worker") }} +{{- if or (ne $.Values.phase "serve") (ne $target.id $leader) }} --- apiVersion: apps/v1 kind: Deployment @@ -163,12 +166,12 @@ spec: - '50052' - --device - CUDA0 -{{- if and $.Values.rpc.cache.enabled (eq $target.id "worker") }} +{{- if and $.Values.rpc.cache.enabled (ne $target.id $leader) }} - --cache {{- end }} env: - {name: HOME, value: /tmp} -{{- if and $.Values.rpc.cache.enabled (eq $target.id "worker") }} +{{- if and $.Values.rpc.cache.enabled (ne $target.id $leader) }} - {name: LLAMA_CACHE, value: /rpc-cache} {{- end }} {{- range $name, $value := $.Values.runtime.env }} @@ -191,7 +194,7 @@ spec: - {name: runtime, mountPath: /work, readOnly: true} - {name: tmp, mountPath: /tmp} - {name: checks, mountPath: /checks, readOnly: true} -{{- if and $.Values.rpc.cache.enabled (eq $target.id "worker") }} +{{- if and $.Values.rpc.cache.enabled (ne $target.id $leader) }} - {name: rpc-cache, mountPath: /rpc-cache} {{- end }} volumes: @@ -201,9 +204,9 @@ spec: configMap: {name: {{ $.Release.Name }}-runtime} - name: tmp emptyDir: {sizeLimit: 2Gi} -{{- if and $.Values.rpc.cache.enabled (eq $target.id "worker") }} +{{- if and $.Values.rpc.cache.enabled (ne $target.id $leader) }} - name: rpc-cache - persistentVolumeClaim: {claimName: {{ $.Release.Name }}-rpc-cache} + persistentVolumeClaim: {claimName: {{ $.Release.Name }}-rpc-cache-{{ $target.id }}} {{- end }} --- apiVersion: v1 diff --git a/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/serve.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/serve.yaml index e761236d7..24cc5cfef 100644 --- a/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/serve.yaml +++ b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/serve.yaml @@ -18,7 +18,7 @@ kind: Deployment metadata: name: {{ .Release.Name }} spec: - replicas: 1 + replicas: {{ .Values.model.replicas }} progressDeadlineSeconds: 3660 strategy: {type: Recreate} selector: @@ -51,7 +51,7 @@ spec: - {name: PYTHONDONTWRITEBYTECODE, value: '1'} - {name: FIRST_SHARD, value: {{ required "model.firstShard is required" .Values.model.firstShard | quote }}} - {name: SERVED_MODEL, value: {{ required "model.servedName is required" .Values.model.servedName | quote }}} - - {name: RPC_ENDPOINT, value: "{{ .Release.Name }}-rpc-worker:50052"} + - {name: RPC_ENDPOINTS, value: "{{ range $i, $target := rest .Values.targets }}{{ if $i }},{{ end }}{{ $.Release.Name }}-rpc-{{ $target.id }}:50052{{ end }}"} - {name: SERVER_ARGS, value: {{ toJson .Values.model.args | quote }}} {{- range $name, $value := .Values.runtime.env }} - {name: {{ $name }}, value: {{ $value | quote }}} @@ -103,6 +103,6 @@ spec: canary: {{ toJson . }} {{- end }} maxEngineConcurrency: 1 - gpu: {product: NVIDIA-GB10} + gpu: {product: {{ required "gpu.product is required" .Values.gpu.product | quote }}} {{- end }} {{- end }} diff --git a/deploy/helm/llm-routing/recipes/charts/gguf-backend/values.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/values.yaml index 2bad47f65..aac63645e 100644 --- a/deploy/helm/llm-routing/recipes/charts/gguf-backend/values.yaml +++ b/deploy/helm/llm-routing/recipes/charts/gguf-backend/values.yaml @@ -3,7 +3,10 @@ phase: preflight image: "" runtimeClassName: nvidia +# targets[0] is the leader that runs the model server; later targets run RPC workers. targets: [] +gpu: + product: "" artifacts: storageClassName: local-path size: 400Gi @@ -28,6 +31,7 @@ rpc: qualification: attempt: 1 model: + replicas: 1 lock: {} servedName: "" firstShard: "" diff --git a/deploy/helm/llm-routing/recipes/cluster_setup.py b/deploy/helm/llm-routing/recipes/cluster_setup.py index 077250bb1..91602529c 100644 --- a/deploy/helm/llm-routing/recipes/cluster_setup.py +++ b/deploy/helm/llm-routing/recipes/cluster_setup.py @@ -7,9 +7,16 @@ import re import secrets import subprocess +import time + +import sizing HERE = pathlib.Path(__file__).resolve().parent CONTROL_LABELS = ('node-role.kubernetes.io/control-plane', 'node-role.kubernetes.io/master') +PROBE_COMMAND = ("nvidia-smi --query-gpu=name,compute_cap,memory.total --format=csv,noheader,nounits" + " && grep '^MemTotal:' /proc/meminfo") +# nvidia-smi reports no dedicated GPU memory when the GPU shares system memory, as on GB10. +UNREPORTED_MEMORY = {'[N/A]', 'N/A', '[Not Supported]', 'Not Supported'} class ClusterSetupError(RuntimeError): @@ -66,8 +73,80 @@ def requests_gpu(pod): return bool(spec.get('resourceClaims')) -def discover_config(context, namespace=None): - """Return fresh settings using eligible nodes and the cluster defaults.""" +def parse_probe(text): + """Turn GPU probe output into the configuration's gpu section.""" + lines = [line.strip() for line in text.splitlines() if line.strip()] + meminfo = [line for line in lines if line.startswith('MemTotal:')] + gpus = [line for line in lines if not line.startswith('MemTotal:')] + require(len(meminfo) == 1, 'The GPU probe did not report host memory.') + require(len(gpus) == 1, 'Expected one GPU per model node. The probe reported ' + str(len(gpus)) + '.') + fields = [field.strip() for field in gpus[0].split(',')] + require(len(fields) == 3, 'Unexpected nvidia-smi output: ' + gpus[0]) + name, capability, memory = fields + unified = memory in UNREPORTED_MEMORY + if unified: + memory_gib = int(meminfo[0].split()[1]) / 1024**2 + else: + require(memory.isdigit(), 'Unexpected GPU memory value from nvidia-smi: ' + memory) + memory_gib = int(memory) / 1024 + gpu = {'name': name, 'computeCapability': capability, 'memoryGiB': round(memory_gib, 1), + 'unifiedMemory': unified, 'cudaArchitectures': None} + try: + sizing.check_gpu(gpu) + except ValueError as error: + raise ClusterSetupError('GPU probe: ' + str(error)) from None + return gpu + + +def kubectl(context, *args, stdin=None, timeout=60): + try: + return subprocess.run(['kubectl', '--context', context, *args], input=stdin, text=True, check=True, + capture_output=True, timeout=timeout).stdout + except (OSError, subprocess.SubprocessError) as error: + detail = getattr(error, 'stderr', '') or str(error) + raise ClusterSetupError('kubectl ' + ' '.join(args[:2]) + ' failed: ' + detail.strip()[-500:]) from None + + +def probe_gpus(context, names, image, runtime_class): + """Run one short GPU pod per node in a temporary namespace, then delete the namespace.""" + namespace = 'llm-routing-probe-' + secrets.token_hex(3) + kubectl(context, 'create', 'namespace', namespace) + try: + pods = {} + for index, name in enumerate(names): + pod = 'gpu-probe-' + str(index) + manifest = {'apiVersion': 'v1', 'kind': 'Pod', 'metadata': {'name': pod, 'namespace': namespace}, + 'spec': {'restartPolicy': 'Never', 'automountServiceAccountToken': False, + 'runtimeClassName': runtime_class, 'nodeSelector': {'kubernetes.io/hostname': name}, + 'securityContext': {'runAsUser': 1000, 'runAsGroup': 1000, 'runAsNonRoot': True, + 'seccompProfile': {'type': 'RuntimeDefault'}}, + 'containers': [{'name': 'probe', 'image': image, 'command': ['sh', '-c', PROBE_COMMAND], + 'env': [{'name': 'NVIDIA_DRIVER_CAPABILITIES', 'value': 'compute,utility'}], + 'resources': {'requests': {'cpu': '100m', 'memory': '128Mi', 'nvidia.com/gpu': 1}, + 'limits': {'cpu': '1', 'memory': '512Mi', 'nvidia.com/gpu': 1}}, + 'securityContext': {'allowPrivilegeEscalation': False, 'readOnlyRootFilesystem': True, + 'capabilities': {'drop': ['ALL']}}}]}} + kubectl(context, 'create', '-f', '-', stdin=json.dumps(manifest)) + pods[name] = pod + results = {} + for name, pod in pods.items(): + # The first pull of the runtime image can take several minutes. + kubectl(context, '-n', namespace, 'wait', 'pod/' + pod, '--for=jsonpath={.status.phase}=Succeeded', + '--timeout=20m', timeout=1260) + results[name] = parse_probe(kubectl(context, '-n', namespace, 'logs', pod)) + return results + finally: + try: + kubectl(context, 'delete', 'namespace', namespace, '--wait=true', '--timeout=3m', timeout=200) + except ClusterSetupError: + print('Warning: delete the temporary GPU probe namespace manually:', namespace) + + +def discover_config(context, namespace=None, recipe=None, probe=None): + """Return fresh settings using eligible nodes, a GPU probe and the cluster defaults. + + probe(names) returns {node: gpu}; tests replace it to avoid touching a cluster. + """ require(isinstance(context, str) and bool(context.strip()), 'Select a Kubernetes context first.') config = json.loads((HERE/'config.example.json').read_text()) namespace = namespace or config['namespace'] @@ -86,18 +165,7 @@ def discover_config(context, namespace=None): idle = [node for node in candidates if int(node.get('status', {}).get('allocatable', {}).get('nvidia.com/gpu', 0)) >= 1 and node['metadata']['name'] not in busy] - controls = [node for node in candidates if control_plane(node)] - workers = [node for node in idle if not control_plane(node)] - if len(controls) != 1 or len(workers) < 2: - workers = idle - selected = {node['metadata']['name'] for node in sorted(workers, key=lambda item: item['metadata']['name'])[:2]} - controls = [node for node in candidates if node['metadata']['name'] not in selected] - workers = sorted(workers, key=lambda item: item['metadata']['name'])[:2] - controls = sorted(controls, key=lambda item: item['metadata']['name']) - require(len(workers) == 2 and controls, - 'Need two idle GPU nodes and one separate Ready ARM64 routing node. Free the required GPUs or set placement nodes explicitly in the configuration.') - control = controls[0]['metadata']['name'] - worker_names = sorted(node['metadata']['name'] for node in workers) + require(idle, 'No idle Ready ARM64 GPU node found. Free a GPU or set placement explicitly in the configuration.') imports = [node for node in nodes if node.get('metadata', {}).get('labels', {}).get('kubernetes.io/arch') == 'arm64'] require(all(node['metadata'].get('labels', {}).get('kubernetes.io/hostname') == node['metadata']['name'] for node in imports), 'Node hostname labels differ from node names. Configure placement and image import targets explicitly.') @@ -117,9 +185,30 @@ def discover_config(context, namespace=None): selected_runtime = named if named else nvidia require(len(selected_runtime) == 1, 'Expected an NVIDIA RuntimeClass with handler nvidia. Set runtimeClass explicitly for this cluster.') - config.update(context=context, namespace=namespace, clusterId=namespace, - nodes={'control': control, 'leader': worker_names[0], 'worker': worker_names[1]}, - storageClass=selected_storage[0]['metadata']['name'], runtimeClass=selected_runtime[0]['metadata']['name']) + runtime_class = selected_runtime[0]['metadata']['name'] + # Prefer GPU nodes outside the control plane so the control plane can host routing. + ordered = [node['metadata']['name'] for node in sorted(idle, key=lambda n: (control_plane(n), n['metadata']['name']))] + probe = probe or (lambda names: probe_gpus(context, names, config['runtimeImage'], runtime_class)) + detected = probe(ordered[:max(sizing.MODEL_NODE_COUNTS)]) + gpu = detected[ordered[0]] + try: + count = sizing.model_node_count(recipe, gpu) + except ValueError as error: + raise ClusterSetupError(str(error)) from None + require(len(ordered) >= count, recipe['name'] + ' needs ' + str(count) + ' idle ' + gpu['name'] + + ' nodes. Found ' + str(len(ordered)) + '. Free GPUs or set placement explicitly in the configuration.') + model = ordered[:count] + require(all((detected[name]['name'], detected[name]['computeCapability']) == (gpu['name'], gpu['computeCapability']) + for name in model), + 'Mixed GPU types are not supported: ' + ', '.join(n + '=' + detected[n]['name'] for n in model) + + '. Set nodes.model explicitly to matching nodes.') + others = sorted((node for node in candidates if node['metadata']['name'] not in model), + key=lambda n: (not control_plane(n), n['metadata']['name'])) + # With no spare node, routing shares the leader; it needs no GPU. + control = others[0]['metadata']['name'] if others else model[0] + config.update(context=context, namespace=namespace, clusterId=namespace, recipe=recipe['name'], + nodes={'control': control, 'model': model}, gpu=gpu, + storageClass=selected_storage[0]['metadata']['name'], runtimeClass=runtime_class) config['images'].update(prefix='localhost/' + namespace, tag='dev-' + datetime.datetime.now(datetime.timezone.utc).strftime('%Y%m%d%H%M%S') + '-' + secrets.token_hex(3), pullPolicy='Never', pullSecrets=[]) @@ -174,16 +263,16 @@ def owned(resource, release): require(not any(item['metadata'].get('deletionTimestamp') for item in namespaces), 'The saved namespace is terminating.') nodes = {item['metadata']['name']: item for item in items(context, 'nodes')} - for role, name in config['nodes'].items(): + model = config['nodes']['model'] + for name in sorted({config['nodes']['control'], *model}): require(name in nodes and eligible(nodes[name]), 'Saved node is unavailable: ' + name) require(nodes[name]['metadata']['labels'].get('kubernetes.io/hostname') == name, 'Saved node hostname no longer matches placement: ' + name) - if role in ('leader', 'worker'): + if name in model: require(int(nodes[name]['status'].get('allocatable', {}).get('nvidia.com/gpu', 0)) >= 1, 'Saved model node has no advertised GPU: ' + name) pods = items(context, 'pods', all_namespaces=True) - require(not any(requests_gpu(pod) and pod.get('spec', {}).get('nodeName') in - (config['nodes']['leader'], config['nodes']['worker']) for pod in pods), + require(not any(requests_gpu(pod) and pod.get('spec', {}).get('nodeName') in model for pod in pods), 'A saved model GPU is occupied.') if not namespaces: require(not crds, 'Retained CRD without the saved namespace requires ownership review.') @@ -192,12 +281,14 @@ def owned(resource, release): 'Workloads still exist in the saved namespace. Finish uninstalling before init.') claims = items(context, 'persistentvolumeclaims', namespace=namespace) volumes = {item['metadata']['name']: item for item in items(context, 'persistentvolumes')} if claims else {} - expected = {backend + '-artifacts': (backend, 'leader'), backend + '-rpc-cache': (backend, 'worker'), - prefix + '-monitoring-metrics': (prefix + '-monitoring', 'control')} + # claim name -> (owning release, node it must stay on) + expected = {backend + '-artifacts': (backend, model[0]), + prefix + '-monitoring-metrics': (prefix + '-monitoring', config['nodes']['control'])} + expected.update({backend + '-rpc-cache-n' + str(index): (backend, name) for index, name in enumerate(model) if index}) for claim in claims: name = claim['metadata']['name'] require(name in expected, 'Unexpected retained PVC: ' + name) - release, role = expected[name] + release, node = expected[name] owned(claim, release) spec = claim['spec'] require(claim.get('status', {}).get('phase') == 'Bound' @@ -212,13 +303,13 @@ def owned(resource, release): and ref.get('name') == name and ref.get('namespace') == namespace, 'Retained PVC binding changed: ' + name) selected = claim['metadata'].get('annotations', {}).get('volume.kubernetes.io/selected-node') - require(not selected or selected == config['nodes'][role], 'Retained PVC placement changed: ' + name) + require(not selected or selected == node, 'Retained PVC placement changed: ' + name) terms = volume.get('spec', {}).get('nodeAffinity', {}).get('required', {}).get('nodeSelectorTerms') if terms is not None: # Fail closed for affinity shapes that this K3s recipe cannot validate. require(any(not term.get('matchFields') and term.get('matchExpressions') and all(expr.get('key') == 'kubernetes.io/hostname' and expr.get('operator') == 'In' - and config['nodes'][role] in expr.get('values', []) + and node in expr.get('values', []) for expr in term['matchExpressions']) for term in terms), 'Retained PV affinity does not match saved placement: ' + name) secrets_by_name = {operator + '-cluster-credential': operator, config['caConfigMap']: stack} diff --git a/deploy/helm/llm-routing/recipes/config.example.json b/deploy/helm/llm-routing/recipes/config.example.json index 1814306ee..cba30d18c 100644 --- a/deploy/helm/llm-routing/recipes/config.example.json +++ b/deploy/helm/llm-routing/recipes/config.example.json @@ -6,8 +6,17 @@ "clusterId": "llm-routing-poc", "nodes": { "control": "control", - "leader": "model-0", - "worker": "model-1" + "model": [ + "model-0", + "model-1" + ] + }, + "gpu": { + "name": "NVIDIA GB10", + "computeCapability": "12.1", + "memoryGiB": 121.6, + "unifiedMemory": true, + "cudaArchitectures": null }, "storageClass": "local-path", "runtimeClass": "nvidia", diff --git a/deploy/helm/llm-routing/recipes/glm-5.3/recipe.json b/deploy/helm/llm-routing/recipes/glm-5.3/recipe.json index e7925a1fe..b1b514950 100644 --- a/deploy/helm/llm-routing/recipes/glm-5.3/recipe.json +++ b/deploy/helm/llm-routing/recipes/glm-5.3/recipe.json @@ -53,5 +53,18 @@ "runtimeEnv": { "GGML_CUDA_DISABLE_GRAPHS": "1", "GGML_RPC_NO_RDMA": "1" + }, + "memory": { + "unified": { + "reservedGiB": 4, + "checkOverShareGiB": 1, + "requestUnderShareGiB": 1, + "limitOverShareGiB": 2 + }, + "discrete": { + "gpuHeadroomGiB": 16, + "hostRequestGiB": 32, + "hostLimitGiB": 64 + } } } diff --git a/deploy/helm/llm-routing/recipes/recipe.py b/deploy/helm/llm-routing/recipes/recipe.py index b092bf9ac..9d818fb5c 100644 --- a/deploy/helm/llm-routing/recipes/recipe.py +++ b/deploy/helm/llm-routing/recipes/recipe.py @@ -25,6 +25,7 @@ import gateway_access import cluster_setup import console_output +import sizing COMPONENTS = {'gateway': 'src/invocation-plane-services/llm-api-gateway', 'router': 'src/libraries/rust/stargate', 'pylon': 'src/libraries/rust/stargate', 'operator': 'src/compute-plane-services/pylon-operator'} @@ -52,7 +53,7 @@ def output(command, **kwargs): RECIPE_KEYS = ('name', 'releaseName', 'llamaCppRevision', 'servedName', 'endpointName', 'firstShard', - 'artifactsSize', 'rpcCacheSize', 'canary', 'serverArgs', 'runtimeEnv') + 'artifactsSize', 'rpcCacheSize', 'canary', 'serverArgs', 'runtimeEnv', 'memory') # The tool derives these from nodes.model; a recipe that sets them would conflict. PLACEMENT_FLAGS = ('--device', '-dev', '--tensor-split', '-ts', '--rpc') QUALIFICATION_ATTEMPT = 2 @@ -98,8 +99,17 @@ def validate(c): for key in ('namespace', 'releasePrefix', 'clusterId'): require(re.fullmatch(r'[a-z0-9]([-a-z0-9]*[a-z0-9])?', c[key]) is not None, 'Invalid ' + key) require(len(c['releasePrefix']) <= 30, 'releasePrefix must be at most 30 characters.') - require(all(c['nodes'].get(role) for role in ('leader', 'worker', 'control')), 'All three placement roles are required.') - require(c['nodes']['leader'] != c['nodes']['worker'], 'The model needs two distinct GPU nodes.') + nodes = c.get('nodes', {}) + require(set(nodes) == {'control', 'model'}, 'Set nodes.control and nodes.model.') + require(isinstance(nodes['control'], str) and bool(nodes['control'].strip()), 'nodes.control must name a node.') + model = nodes['model'] + require(isinstance(model, list) and all(isinstance(n, str) and n.strip() for n in model), 'nodes.model must list node names.') + require(len(model) in sizing.MODEL_NODE_COUNTS, 'nodes.model must list 1 or 2 nodes.') + require(len(set(model)) == len(model), 'nodes.model must not repeat a node.') + try: + sizing.check_gpu(c.get('gpu') or {}) + except ValueError as error: + raise RuntimeError(str(error)) from None require(c['images']['pullPolicy'] in ('Never', 'IfNotPresent', 'Always'), 'Invalid pull policy.') require(re.fullmatch(r'[A-Za-z0-9_][A-Za-z0-9_.-]{0,127}', c['images']['tag']) is not None, 'Invalid image tag.') require('/' in c['images']['prefix'] and not c['images']['prefix'].endswith('/'), 'Use registry/path as image prefix.') @@ -110,6 +120,10 @@ def validate(c): +def all_nodes(c): + return sorted({c['nodes']['control'], *c['nodes']['model']}) + + def default_work_dir(context): require(isinstance(context, str) and bool(context.strip()), 'Set LLM_ROUTING_CONTEXT or --context. The current kubectl context is not selected automatically.') @@ -224,14 +238,18 @@ def placement(obj): require(len(matches) == 1, 'Existing endpoint does not match a recipe in this checkout: '+endpoint['metadata']['name']) model = owner(endpoint) require(endpoint['spec']['service']['name'] == model, 'Existing endpoint does not serve its own Helm release.') + backend = values(model) + targets = backend['targets'] + require(targets and [t['id'] for t in targets] == ['n'+str(i) for i in range(len(targets))], + 'Model release targets are not in the recipe format.') leader = get('deployment', model) - worker = get('deployment', model+'-rpc-worker') + workers = [get('deployment', model+'-rpc-'+t['id']) for t in targets[1:]] service = get('service', model) - require(all(owner(d) == model for d in (leader, worker, service)), 'Unexpected model resource ownership.') - backend = values(model) - nodes = {'control': control, 'leader': placement(leader), 'worker': placement(worker)} - targets = {t['id']: t['node'] for t in backend['targets']} - require(all(targets.get(role) == nodes[role] for role in ('leader', 'worker')), 'Model placement differs from Helm values.') + require(all(owner(d) == model for d in [leader, service, *workers]), 'Unexpected model resource ownership.') + require([placement(d) for d in [leader, *workers]] == [t['node'] for t in targets], + 'Model placement differs from Helm values.') + nodes = {'control': control, 'model': [t['node'] for t in targets]} + gpu = {key: backend.get('gpu', {}).get(key) for key in ('name', 'computeCapability', 'memoryGiB', 'unifiedMemory', 'cudaArchitectures')} releases = json.loads(output(hm+['list', '-o', 'json'])) operators = [] for release in releases: @@ -283,7 +301,7 @@ def placement(obj): image_prefix += '/attached' # TLS private material and existing caller/cluster key hashes are intentionally omitted. config = {'context': context, 'namespace': namespace, 'recipe': matches[0], 'releasePrefix': prefix, 'clusterId': v['clusterId'], - 'releases': {'stack': stack, 'operator': operator, 'model': model}, 'nodes': nodes, + 'releases': {'stack': stack, 'operator': operator, 'model': model}, 'nodes': nodes, 'gpu': gpu, 'storageClass': backend['artifacts']['storageClassName'], 'runtimeClass': backend['runtimeClassName'], 'images': {'prefix': image_prefix, 'tag': image['tag'], 'pullPolicy': image['pullPolicy'], 'pullSecrets': [], 'repositories': images}, @@ -311,6 +329,10 @@ def __init__(self, config, work, source=None): self.kc = ['kubectl', '--context', config['context'], '-n', config['namespace']] self.hm = ['helm', '--kube-context', config['context'], '-n', config['namespace']] self.definition = load_recipe(recipe_name(config)) + # targets[0] (n0) is the leader that runs the model server; later targets run RPC workers. + self.targets = [{'id': 'n'+str(index), 'node': node} for index, node in enumerate(config['nodes']['model'])] + self.workers = self.targets[1:] + self.plan = sizing.memory_plan(self.definition, config['gpu'], len(self.targets)) releases = config.get('releases', {}) self.backend = releases.get('model', config['releasePrefix'] + '-' + self.definition['releaseName']) self.operator = releases.get('operator', config['releasePrefix'] + '-operator') @@ -383,25 +405,30 @@ def prepare(self): print('Using source checkout:', self.source) def backend_values(self, phase='serve', register=False, render=False): - d = self.definition + d, plan, gpu = self.definition, self.plan, self.c['gpu'] + memory = {'requests': str(plan['requestGiB'])+'Gi', 'limits': str(plan['limitGiB'])+'Gi'} if phase == 'serve': - rpc_resources = {'requests': {'cpu': '2', 'memory': '110Gi', 'nvidia.com/gpu': 1}, - 'limits': {'cpu': '8', 'memory': '114Gi', 'nvidia.com/gpu': 1}} + rpc_resources = {'requests': {'cpu': '2', 'memory': memory['requests'], 'nvidia.com/gpu': 1}, + 'limits': {'cpu': '8', 'memory': memory['limits'], 'nvidia.com/gpu': 1}} else: rpc_resources = {'requests': {'cpu': '2', 'memory': '2Gi', 'nvidia.com/gpu': 1}, 'limits': {'cpu': '8', 'memory': '8Gi', 'nvidia.com/gpu': 1}} return { 'phase': phase, 'image': self.c['runtimeImage'], 'runtimeClassName': self.c['runtimeClass'], - 'targets': [{'id': r, 'node': self.c['nodes'][r]} for r in ('leader', 'worker')], + # The chart reads only gpu.product; attach-existing reads the rest back from the release. + 'targets': copy.deepcopy(self.targets), 'gpu': dict(gpu, product=sizing.gpu_product(gpu)), 'artifacts': {'storageClassName': self.c['storageClass'], 'size': d['artifactsSize']}, - 'build': {'revision': d['llamaCppRevision'], 'cudaArchitectures': '121a-real', 'parallel': 8}, + 'build': {'revision': d['llamaCppRevision'], 'cudaArchitectures': sizing.cuda_architectures(gpu), 'parallel': 8}, 'runtime': {'sha256': self.state.get('runtimeSha256', 'a'*64 if render else ''), 'env': dict(d['runtimeEnv'])}, 'qualification': {'attempt': self.state.get('qualificationAttempt', QUALIFICATION_ATTEMPT)}, 'model': {'lock': d['lock'], 'servedName': d['servedName'], 'firstShard': d['firstShard'], 'endpointName': d['endpointName'], 'register': register, 'canary': dict(d['canary']), - 'args': d['serverArgs'] + ['--device', 'CUDA0,RPC0', '--tensor-split', '1,1']}, + 'args': d['serverArgs'] + sizing.placement_args(len(self.targets)), + 'resources': {'requests': {'cpu': '4', 'memory': memory['requests'], 'nvidia.com/gpu': 1}, + 'limits': {'cpu': '12', 'memory': memory['limits'], 'nvidia.com/gpu': 1}}}, 'rpc': {'resources': rpc_resources, - 'cache': {'enabled': phase == 'serve', 'size': d['rpcCacheSize'], 'storageClassName': self.c['storageClass']}}, + 'cache': {'enabled': phase == 'serve' and bool(self.workers), 'size': d['rpcCacheSize'], + 'storageClassName': self.c['storageClass']}}, 'chain': {'runtimeRelease': self.backend, 'artifactClaim': self.backend+'-artifacts', 'attempt': self.state.get('chainAttempt', 1)}, } @@ -444,20 +471,20 @@ def inventory(self): nodes = json.loads(output(self.kc+['get', 'nodes', '-o', 'json']))['items'] pods = json.loads(output(self.kc+['get', 'pods', '-A', '-o', 'json']))['items'] crds = json.loads(output(self.kc+['get', 'crds', '-o', 'json']))['items'] - selected = self.c['nodes'] if self.c['images']['pullPolicy'] == 'Never': eligible = {n['metadata']['name'] for n in nodes if n['metadata']['labels'].get('kubernetes.io/arch') == 'arm64'} - imports = set(self.c['containerd'].get('nodeNames', selected.values())) + imports = set(self.c['containerd'].get('nodeNames', all_nodes(self.c))) require(eligible <= imports, 'Pre-import Pylon on every ARM64 node where it can schedule. Set containerd.nodeNames explicitly.') - for role in ('control', 'leader', 'worker'): - found = [n for n in nodes if n['metadata']['name'] == selected[role]] - require(len(found) == 1, 'Missing node for '+role) + roles = [('control', self.c['nodes']['control'])] + [('model', target['node']) for target in self.targets] + for role, name in roles: + found = [n for n in nodes if n['metadata']['name'] == name] + require(len(found) == 1, 'Missing '+role+' node: '+name) node = found[0] require(node['metadata']['labels'].get('kubernetes.io/arch') == 'arm64', 'Selected nodes must be ARM64.') require(any(c['type'] == 'Ready' and c['status'] == 'True' for c in node['status']['conditions']), 'Node is not Ready.') - if role != 'control': - require(int(node['status']['allocatable'].get('nvidia.com/gpu', 0)) >= 1, 'GPU device plugin has not advertised a GPU.') - busy = [p['metadata']['name'] for p in pods if p['spec'].get('nodeName') == selected[role] + if role == 'model': + require(int(node['status']['allocatable'].get('nvidia.com/gpu', 0)) >= 1, 'GPU device plugin has not advertised a GPU on '+name+'.') + busy = [p['metadata']['name'] for p in pods if p['spec'].get('nodeName') == name and cluster_setup.requests_gpu(p)] require(not busy, 'Selected model GPU is occupied: '+', '.join(busy)) for crd in crds: @@ -478,7 +505,7 @@ def attach_existing(self): require(all(self.state.get(phase) for phase in ('stack', 'serve', 'registered')), 'Saved installation is incomplete. Finish installing and registering model before attaching.') live = discover_config(self.c['context'], self.c['namespace']) - for field in ('context', 'namespace', 'clusterId', 'nodes', 'runtimeClass', + for field in ('context', 'namespace', 'clusterId', 'nodes', 'gpu', 'runtimeClass', 'runtimeImage', 'storageClass', 'caConfigMap'): require(live[field] == self.c[field], 'Existing installation differs from saved configuration: '+field) require(live['releases'] == {'stack': self.stack, 'operator': self.operator, 'model': self.backend}, @@ -496,9 +523,10 @@ def attach_existing(self): key = pathlib.Path(self.c['apiKeyFile']).expanduser().resolve(strict=True) if self.c.get('apiKeyFile') else None nodes = json.loads(output(self.kc+['get', 'nodes', '-o', 'json']))['items'] names = {n['metadata']['name'] for n in nodes} - require(set(self.c['nodes'].values()) <= names, 'Configured placement nodes do not exist.') + require(set(all_nodes(self.c)) <= names, 'Configured placement nodes do not exist.') owners = {'llm-api-gateway': self.stack, 'llm-request-router': self.stack, - self.operator: self.operator, self.backend: self.backend, self.backend+'-rpc-worker': self.backend} + self.operator: self.operator, self.backend: self.backend} + owners.update({self.backend+'-rpc-'+worker['id']: self.backend for worker in self.workers}) for deployment, release in owners.items(): obj = json.loads(output(self.kc+['get', 'deployment', deployment, '-o', 'json'])) annotations = obj['metadata'].get('annotations', {}) @@ -525,7 +553,7 @@ def bound_cluster(self): live = json.loads(output(self.kc+['get', 'nodes', '-o', 'json']))['items'] old = self.state['inventory']['nodes'] current = {n['metadata']['name']: n['metadata']['uid'] for n in live} - require(all(current.get(name) == old.get(name) for name in self.c['nodes'].values()), 'The selected node identities changed or the context points to another cluster.') + require(all(current.get(name) == old.get(name) for name in all_nodes(self.c)), 'The selected node identities changed or the context points to another cluster.') def logs(self, component, release=None, job=None): selector = 'app.kubernetes.io/instance='+(release or self.backend)+',app.kubernetes.io/component='+component @@ -604,13 +632,17 @@ def release(): require(release() == revision, 'Model Helm revision changed while checking load. Retry after the operation completes.') return False require(values == self.backend_values('serve'), 'Deployed model values differ from this load configuration.') + leader = self.targets[0]['node'] + # (node, container, component, volume name, claim) for each Deployment; PVCs map to None. expected = { - ('Deployment', self.backend): ('leader', 'llama', 'model-server'), - ('Deployment', self.backend+'-rpc-worker'): ('worker', 'rpc', 'rpc-worker'), - ('Deployment', self.backend+'-artifacts'): ('leader', 'artifacts', 'artifacts'), + ('Deployment', self.backend): (leader, 'llama', 'model-server', 'artifacts', self.backend+'-artifacts'), + ('Deployment', self.backend+'-artifacts'): (leader, 'artifacts', 'artifacts', 'artifacts', self.backend+'-artifacts'), ('PersistentVolumeClaim', self.backend+'-artifacts'): None, - ('PersistentVolumeClaim', self.backend+'-rpc-cache'): None, } + for worker in self.workers: + expected[('Deployment', self.backend+'-rpc-'+worker['id'])] = ( + worker['node'], 'rpc', 'rpc-'+worker['id'], 'rpc-cache', self.backend+'-rpc-cache-'+worker['id']) + expected[('PersistentVolumeClaim', self.backend+'-rpc-cache-'+worker['id'])] = None names = [('deployment/' if kind == 'Deployment' else 'pvc/')+name for kind, name in expected] def ready_resources(): @@ -641,32 +673,31 @@ def ready_resources(): ('replicas', 'updatedReplicas', 'readyReplicas', 'availableReplicas')) and not status.get('unavailableReplicas', 0), 'Model Deployment is not ready at its current generation: '+name+'. Wait, then rerun load.') - role, container, component = expected[(kind, name)] + node, container, component, volume, claim = expected[(kind, name)] template = spec['template'] labels = {'app.kubernetes.io/instance': self.backend, 'app.kubernetes.io/component': component} require(all(template['metadata'].get('labels', {}).get(key) == value for key, value in labels.items()) and spec.get('selector', {}).get('matchLabels') == labels, 'Unexpected model Deployment selector: '+name) pod = template['spec'] - require(pod.get('nodeSelector', {}).get('kubernetes.io/hostname') == self.c['nodes'][role], + require(pod.get('nodeSelector', {}).get('kubernetes.io/hostname') == node, 'Model Deployment targets another node: '+name) containers = pod.get('containers', []) require(len(containers) == 1 and containers[0].get('name') == container and containers[0].get('image') == self.c['runtimeImage'], 'Model Deployment image differs: '+name) - volume = 'rpc-cache' if component == 'rpc-worker' else 'artifacts' volumes = {item['name']: item for item in pod.get('volumes', [])} mounts = {item['name']: item for item in containers[0].get('volumeMounts', [])} - require(volumes.get(volume, {}).get('persistentVolumeClaim', {}).get('claimName') == self.backend+'-'+volume + require(volumes.get(volume, {}).get('persistentVolumeClaim', {}).get('claimName') == claim and mounts.get(volume, {}).get('mountPath') == '/'+volume, 'Model Deployment storage differs: '+name) if component == 'model-server': env = {item['name']: item.get('value') for item in containers[0].get('env', [])} require(env.get('FIRST_SHARD') == values['model']['firstShard'] and env.get('SERVED_MODEL') == values['model']['servedName'] - and env.get('RPC_ENDPOINT') == self.backend+'-rpc-worker:50052' + and env.get('RPC_ENDPOINTS') == ','.join(self.backend+'-rpc-'+w['id']+':50052' for w in self.workers) and json.loads(env.get('SERVER_ARGS') or 'null') == values['model']['args'], - 'Model model or RPC connection differs: '+name) + 'Model server settings or RPC connection differ: '+name) if component != 'artifacts': require(pod.get('runtimeClassName') == self.c['runtimeClass'] and template['metadata'].get('annotations', {}).get('checksum/runtime') == self.state['runtimeSha256'] @@ -693,10 +724,15 @@ def backend_phase(self, phase, retry=False): if phase == 'serve': if self.resume_load(): return - for role in ('leader', 'worker'): - raw = output(self.kc+['exec', 'deploy/'+self.backend+'-rpc-'+role, '-c', 'rpc', '--', 'cat', '/proc/meminfo']) + for target in self.targets: + raw = output(self.kc+['exec', 'deploy/'+self.backend+'-rpc-'+target['id'], '-c', 'rpc', '--', 'cat', '/proc/meminfo']) available = next(int(line.split()[1])*1024 for line in raw.splitlines() if line.startswith('MemAvailable:')) - require(available > 113*1024**3, 'Insufficient actual host memory on '+role) + require(available > self.plan['hostAvailableGiB']*1024**3, + 'Insufficient host memory on '+target['node']+': '+str(available//1024**3)+' GiB available, more than '+ + str(self.plan['hostAvailableGiB'])+' GiB required.') + if phase == 'preflight': + require(self.plan['fits'], self.definition['name']+' does not fit on '+str(len(self.targets))+' node(s) of '+ + self.c['gpu']['name']+' per gpu.memoryGiB. Add a model node to nodes.model or fix the gpu settings.') if phase == 'qualify': require(re.fullmatch(r'[a-f0-9]{64}', self.state['runtimeSha256']) is not None, 'Invalid built runtime checksum.') if retry: @@ -710,8 +746,7 @@ def backend_phase(self, phase, retry=False): records = self.logs(phase) require(bool(records), phase+' did not record PASS.') if phase == 'preflight': - require(len(records) == 2, 'Both GPU nodes must pass actual CUDA preflight.') - require(all(r['memoryBefore']['MemAvailable'] > 113*1024**3 for r in records), 'Insufficient actual host memory before CUDA allocation.') + self.check_preflight(records) if phase == 'build': self.stamp('runtimeSha256', records[-1]['runtimeSHA256']) if phase == 'download': @@ -720,12 +755,34 @@ def backend_phase(self, phase, retry=False): elif phase == 'qualify': records = self.logs('qualification', job=self.backend+'-qualify-'+str(values['qualification']['attempt'])) require(bool(records), 'RPC qualification did not record PASS.') + if not self.workers: + self.stamp(phase) + return chain = self.backend_values('chain') self.helm_apply(self.backend+'-chain', HERE/'charts/gguf-backend', chain, '15m', jobs=True) records = self.logs('chain-check', job=self.backend+'-chain-'+str(chain['chain']['attempt'])) require(bool(records), 'RPC chain check did not record PASS.') self.stamp(phase) + def check_preflight(self, records): + """Compare each model node's measured GPU and memory with the configuration.""" + gpu = self.c['gpu'] + require(len(records) == len(self.targets), + 'Every model node must pass CUDA preflight: '+str(len(records))+' of '+str(len(self.targets))+' passed.') + for record in records: + capability = '.'.join(str(part) for part in record['capability']) + require(sizing.gpu_product({'name': record['gpu']}) == sizing.gpu_product(gpu) and capability == gpu['computeCapability'], + 'Detected GPU '+record['gpu']+' ('+capability+') differs from gpu settings '+gpu['name']+' ('+ + gpu['computeCapability']+'). Run init again or correct the gpu section.') + available = record['memoryBefore']['MemAvailable'] + require(available > self.plan['hostAvailableGiB']*1024**3, + 'Insufficient host memory before CUDA allocation: '+str(available//1024**3)+' GiB available, more than '+ + str(self.plan['hostAvailableGiB'])+' GiB required.') + if self.plan['gpuFreeGiB'] is not None: + free = record['cudaFreeBytes'] + require(free >= self.plan['gpuFreeGiB']*1024**3, + 'Insufficient GPU memory: '+str(free//1024**3)+' GiB free, '+str(self.plan['gpuFreeGiB'])+' GiB required.') + def deploy_stack(self): require(not self.state.get('attachedExisting'), 'Existing installations support verification and image iteration, not fresh stack deployment.') self.bound_cluster() @@ -873,7 +930,7 @@ def import_images(self, archive, allow, component=None, tag=None): require(bool(tags & aliases), 'Archive is missing the configured image: '+image) cfg = self.c.get('containerd') require(cfg, 'No image importer was discovered. Use registry distribution or configure containerd import settings.') - nodes = [self.c['nodes']['control']] if component in ('gateway', 'router') else cfg.get('nodeNames', sorted(set(self.c['nodes'].values()))) + nodes = [self.c['nodes']['control']] if component in ('gateway', 'router') else cfg.get('nodeNames', all_nodes(self.c)) with archive.open('rb') as stream: archive_hash = hashlib.file_digest(stream, 'sha256').hexdigest() release = self.c['releasePrefix']+'-images' @@ -1019,7 +1076,13 @@ def recovery(self, confirm, port): require(confirm and self.state.get('registered'), 'Recovery requires a registered model and --confirm-model-interruption.') values = self.backend_values(register=True) down = copy.deepcopy(values) - down['rpc']['replicas'] = 0 + # Split models lose their RPC workers; a single-node model loses its model server. + if self.workers: + down['rpc']['replicas'] = 0 + stopped = {'rpc-'+worker['id'] for worker in self.workers} + else: + down['model']['replicas'] = 0 + stopped = {'model-server'} # Validate the same direct path before making an intentional interruption. self.verify(False, port) started = time.monotonic() @@ -1029,7 +1092,10 @@ def recovery(self, confirm, port): time.sleep(15) down_pods = json.loads(output(self.kc+['get', 'pods', '-o', 'json'])) save(self.work/'evidence/recovery-down.json', down_pods) - require(not any(p['metadata']['name'].startswith(self.backend+'-rpc-worker-') and p['status']['phase'] == 'Running' for p in down_pods['items']), 'Worker is still running. Restore and inspect the rollout.') + require(not any(p['metadata'].get('labels', {}).get('app.kubernetes.io/instance') == self.backend + and p['metadata'].get('labels', {}).get('app.kubernetes.io/component') in stopped + and p['status']['phase'] == 'Running' for p in down_pods['items']), + 'Interrupted model pods are still running. Restore and inspect the rollout.') try: with self.forward(False, port): connection = http.client.HTTPConnection('127.0.0.1', port, timeout=10) @@ -1037,7 +1103,7 @@ def recovery(self, confirm, port): connection.request('POST', '/v1/chat/completions', json.dumps({'model': values['model']['servedName'], 'messages': [{'role': 'user', 'content': 'What is 31 plus 17?'}], 'max_tokens': 32}), {'Content-Type': 'application/json'}) response = connection.getresponse() response.read() - observation['statusWhileWorkerDown'] = response.status + observation['statusWhileInterrupted'] = response.status observation['interruptionObserved'] = response.status != 200 finally: connection.close() @@ -1048,7 +1114,7 @@ def recovery(self, confirm, port): self.helm_apply(self.backend, HERE/'charts/gguf-backend', values, '70m') observation['restoreSeconds'] = time.monotonic() - started save(self.work/'evidence/recovery.json', observation) - require(observation['interruptionObserved'], 'Worker interruption was not demonstrated. Recovery restored the model but this test did not prove the failure path.') + require(observation['interruptionObserved'], 'Model interruption was not demonstrated. Recovery restored the model but this test did not prove the failure path.') self.verify(False, port) self.verify(True, port+1) @@ -1056,7 +1122,8 @@ def render(self): self.prepare() renders = [('stack', self.source/'deploy/helm/llm-gateway-stack/llm-gateway-stack', self.stack_values('a'*64, 'b'*64, 'offline-demo-ui-placeholder')), ('operator', self.source/'deploy/helm/pylon-operator/pylon-operator', self.operator_values())] - for phase in ('preflight', 'build', 'qualify', 'chain', 'download', 'serve'): + phases = ('preflight', 'build', 'qualify', 'chain', 'download', 'serve') if self.workers else ('preflight', 'build', 'qualify', 'download', 'serve') + for phase in phases: renders.append((self.definition['releaseName']+'-'+phase, HERE/'charts/gguf-backend', self.backend_values(phase, register=phase=='serve', render=True))) for name, chart, values in renders: path = self.work/'render'/(name+'-values.json') @@ -1074,6 +1141,7 @@ def main(argv=None, console=None): parser.add_argument('--namespace', help='Select the namespace when the cluster has multiple installations.') parser.add_argument('--work-dir', type=pathlib.Path, help='Private local state directory; defaults to a per-context directory.') parser.add_argument('--source-dir', type=pathlib.Path, help='Existing source checkout; defaults to the checkout containing this script.') + parser.add_argument('--recipe', help='Recipe folder for init; defaults to the only recipe present. Available: '+', '.join(available_recipes())+'.') parser.add_argument('phase', choices=['init', 'paths', 'context', 'prepare', 'render', 'inventory', 'attach-existing', 'build-images', 'push-images', 'export-images', 'import-images', 'stack', 'preflight', 'build-runtime', 'qualify', 'download', 'load', 'verify-direct', 'register', 'verify-gateway', 'chat', 'cleanup-key', 'update', 'rollback', 'recover']) parser.add_argument('prompt', nargs='?', help='Prompt for the chat command.') parser.add_argument('--stream', action='store_true', help='Stream the chat response.') @@ -1090,6 +1158,7 @@ def main(argv=None, console=None): console.phase = args.phase require(args.phase == 'chat' or (args.prompt is None and not args.stream), 'Prompt and --stream are supported only for chat.') require(not args.retry or args.phase == 'qualify', '--retry is supported only for qualify.') + require(not args.recipe or args.phase == 'init', '--recipe is supported only for init. Later commands use the saved configuration.') try: context, work, config_path, config = cli_settings(args) except ContextSelectionError as error: @@ -1115,7 +1184,8 @@ def execute(args, parser, context, work, config_path, config): 'Saved state is missing its configuration. Init does not overwrite an installation.') require(not config_path.is_relative_to(HERE.parents[3]), 'Keep generated configuration outside the checkout.') try: - config = cluster_setup.discover_config(context, args.namespace) + definition = load_recipe(args.recipe or recipe_name({})) + config = cluster_setup.discover_config(context, args.namespace, definition) except cluster_setup.ClusterSetupError as error: parser.exit(2, 'error: ' + str(error) + '\n') validate(config) @@ -1124,7 +1194,10 @@ def execute(args, parser, context, work, config_path, config): with os.fdopen(fd, 'w') as file: file.write(json.dumps(config, indent=2) + '\n') print('Configuration created:', config_path) - print('Model GPUs:', config['nodes']['leader'], 'and', config['nodes']['worker']) + print('Recipe:', config['recipe']) + print('GPU:', config['gpu']['name'], '('+config['gpu']['computeCapability']+',', + str(config['gpu']['memoryGiB'])+' GiB', 'shared with the CPU)' if config['gpu']['unifiedMemory'] else 'GPU memory)') + print('Model nodes:', ', '.join(config['nodes']['model'])) print('Routing node:', config['nodes']['control']) print('Run render, inventory, then preflight.') return diff --git a/deploy/helm/llm-routing/recipes/sizing.py b/deploy/helm/llm-routing/recipes/sizing.py new file mode 100644 index 000000000..9114ffe62 --- /dev/null +++ b/deploy/helm/llm-routing/recipes/sizing.py @@ -0,0 +1,72 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +"""GPU-derived settings and memory sizing for recipe placement. No cluster access.""" +import math +import re + +# The data model and charts handle any count; only these are exercised on hardware. +MODEL_NODE_COUNTS = (1, 2) + + +def check_gpu(gpu): + capability = gpu.get('computeCapability') + # NVIDIA minor versions are a single digit, so "10.3" is unambiguous as "103". + if not isinstance(capability, str) or re.fullmatch(r'\d+\.\d', capability) is None: + raise ValueError('gpu.computeCapability must look like "10.3".') + if not isinstance(gpu.get('name'), str) or not gpu['name'].strip(): + raise ValueError('gpu.name must be the GPU name reported by nvidia-smi.') + memory = gpu.get('memoryGiB') + if isinstance(memory, bool) or not isinstance(memory, (int, float)) or memory <= 0: + raise ValueError('gpu.memoryGiB must be a positive number.') + if not isinstance(gpu.get('unifiedMemory'), bool): + raise ValueError('gpu.unifiedMemory must be true or false.') + override = gpu.get('cudaArchitectures') + if override is not None and (not isinstance(override, str) or not override.strip()): + raise ValueError('gpu.cudaArchitectures must be null or a CMake architecture list.') + + +def cuda_architectures(gpu): + if gpu.get('cudaArchitectures'): + return gpu['cudaArchitectures'] + major, minor = gpu['computeCapability'].split('.') + return str(int(major)) + str(int(minor)) + 'a-real' + + +def gpu_product(gpu): + return re.sub(r'\s+', '-', gpu['name'].strip()) + + +def memory_plan(recipe, gpu, count): + """Return per-node fit, pod memory in GiB and the free-memory thresholds checked before loading.""" + if count not in MODEL_NODE_COUNTS: + raise ValueError('The recipe supports 1 or 2 model nodes, not ' + str(count) + '.') + share = recipe['lock']['weightGiB'] / count + if gpu['unifiedMemory']: + rules = recipe['memory']['unified'] + limit = math.ceil(share) + rules['limitOverShareGiB'] + return {'fits': limit <= gpu['memoryGiB'] - rules['reservedGiB'], + 'requestGiB': math.floor(share) - rules['requestUnderShareGiB'], 'limitGiB': limit, + 'hostAvailableGiB': math.ceil(share) + rules['checkOverShareGiB'], 'gpuFreeGiB': None} + rules = recipe['memory']['discrete'] + gpu_need = math.ceil(share) + rules['gpuHeadroomGiB'] + return {'fits': gpu_need <= gpu['memoryGiB'], + 'requestGiB': rules['hostRequestGiB'], 'limitGiB': rules['hostLimitGiB'], + 'hostAvailableGiB': rules['hostLimitGiB'], 'gpuFreeGiB': gpu_need} + + +def model_node_count(recipe, gpu): + for count in MODEL_NODE_COUNTS: + if memory_plan(recipe, gpu, count)['fits']: + return count + plan = memory_plan(recipe, gpu, MODEL_NODE_COUNTS[-1]) + need = plan['gpuFreeGiB'] or plan['limitGiB'] + raise ValueError(recipe['name'] + ' does not fit on ' + str(MODEL_NODE_COUNTS[-1]) + ' nodes of ' + gpu['name'] + + ': each needs ' + str(need) + ' GiB, and ' + str(gpu['memoryGiB']) + ' GiB is available.') + + +def placement_args(count): + devices = ['CUDA0'] + ['RPC' + str(index) for index in range(count - 1)] + args = ['--device', ','.join(devices)] + if count > 1: + args += ['--tensor-split', ','.join(['1'] * count)] + return args diff --git a/deploy/helm/llm-routing/recipes/tests/test_attach_reuse.py b/deploy/helm/llm-routing/recipes/tests/test_attach_reuse.py index a2c2d42d6..316dfd626 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_attach_reuse.py +++ b/deploy/helm/llm-routing/recipes/tests/test_attach_reuse.py @@ -34,7 +34,7 @@ def installation(self, explicit=False): key.write_text('test-only-key\n') recipe.state = { 'identity': recipe.identity, - 'inventory': {'nodes': {name: name+'-uid' for name in config['nodes'].values()}}, + 'inventory': {'nodes': {name: name+'-uid' for name in tool.all_nodes(config)}}, 'runtimeSha256': 'a'*64, 'download': True, 'qualify': True, 'stack': {'source': recipe.source_identity(), 'apiKeyFile': str(key)}, 'serve': True, 'registered': True, 'direct': True, 'gateway': True, @@ -115,7 +115,8 @@ def test_saved_source_revision_does_not_block_attachment(self): def test_live_identity_and_runtime_mismatches_preserve_local_state(self): changes = { 'context': 'another-context', 'namespace': 'another-namespace', - 'clusterId': 'another-cluster', 'nodes': {'control': 'c', 'leader': 'a', 'worker': 'b'}, + 'clusterId': 'another-cluster', 'nodes': {'control': 'c', 'model': ['a', 'b']}, + 'gpu': {'name': 'NVIDIA GB300', 'computeCapability': '10.3', 'memoryGiB': 268.0, 'unifiedMemory': False, 'cudaArchitectures': None}, 'runtimeClass': 'another-runtime', 'runtimeImage': 'example.com/another:runtime', 'storageClass': 'another-storage', 'caConfigMap': 'another-ca', } @@ -149,7 +150,7 @@ def test_changed_node_uid_is_rejected_before_any_rediscovery(self): recipe, _ = self.installation() before = recipe.state_path.read_bytes() nodes = {'items': [{'metadata': {'name': name, 'uid': name+'-replacement'}} - for name in recipe.c['nodes'].values()]} + for name in tool.all_nodes(recipe.c)]} with patch.object(tool, 'output', return_value=json.dumps(nodes)) as output, \ patch.object(tool, 'discover_config') as discover, patch.object(tool, 'save') as save: with self.assertRaisesRegex(RuntimeError, 'identities changed|another cluster'): diff --git a/deploy/helm/llm-routing/recipes/tests/test_cli.py b/deploy/helm/llm-routing/recipes/tests/test_cli.py index 76d753db2..3d949b5bb 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_cli.py +++ b/deploy/helm/llm-routing/recipes/tests/test_cli.py @@ -319,7 +319,7 @@ def test_init_creates_private_discovered_configuration(self): patch.object(tool.cluster_setup, 'discover_config', return_value=self.config) as discover: tool.main(['init']) recipe.assert_not_called() - discover.assert_called_once_with('team-context', None) + discover.assert_called_once_with('team-context', None, tool.load_recipe('glm-5.3')) expected = self.config path = self.work/'config.json' self.assertEqual(json.loads(path.read_text()), expected) @@ -381,7 +381,7 @@ def test_init_supports_new_explicit_config_path_and_namespace(self): with redirect_stdout(io.StringIO()), patch.object(tool.cluster_setup, 'discover_config', \ return_value=dict(self.config, namespace='my-stack')) as discover: tool.main(['--config', str(path), '--namespace', 'my-stack', 'init']) - discover.assert_called_once_with('team-context', 'my-stack') + discover.assert_called_once_with('team-context', 'my-stack', tool.load_recipe('glm-5.3')) config = json.loads(path.read_text()) self.assertEqual(config['context'], 'team-context') self.assertEqual(config['namespace'], 'my-stack') @@ -390,9 +390,9 @@ def test_init_supports_new_explicit_config_path_and_namespace(self): def test_repeated_attach_checks_existing_node_bindings_before_refresh(self): recipe = tool.Recipe(self.config, self.work) recipe.state = {'identity': recipe.identity, 'attachedExisting': True, - 'inventory': {'nodes': {name: name+'-old' for name in self.config['nodes'].values()}}} + 'inventory': {'nodes': {name: name+'-old' for name in tool.all_nodes(self.config)}}} original = copy.deepcopy(recipe.state) - live = {'items': [{'metadata': {'name': name, 'uid': name+'-new'}} for name in self.config['nodes'].values()]} + live = {'items': [{'metadata': {'name': name, 'uid': name+'-new'}} for name in tool.all_nodes(self.config)]} with patch.object(tool, 'output', return_value=json.dumps(live)) as output, \ self.assertRaisesRegex(RuntimeError, 'node identities changed'): recipe.attach_existing() diff --git a/deploy/helm/llm-routing/recipes/tests/test_cluster_setup.py b/deploy/helm/llm-routing/recipes/tests/test_cluster_setup.py index a4e78e188..2bb813d56 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_cluster_setup.py +++ b/deploy/helm/llm-routing/recipes/tests/test_cluster_setup.py @@ -13,6 +13,11 @@ setup = importlib.util.module_from_spec(spec) spec.loader.exec_module(setup) +GLM = json.loads((HERE/'glm-5.3/recipe.json').read_text()) +GLM['lock'] = json.loads((HERE/'glm-5.3/model.lock.json').read_text()) +GB10 = {'name': 'NVIDIA GB10', 'computeCapability': '12.1', 'memoryGiB': 121.6, 'unifiedMemory': True, 'cudaArchitectures': None} +GB300 = {'name': 'NVIDIA GB300', 'computeCapability': '10.3', 'memoryGiB': 268.0, 'unifiedMemory': False, 'cudaArchitectures': None} + def node(name, control=False): labels = {'kubernetes.io/hostname': name, 'kubernetes.io/arch': 'arm64', 'kubernetes.io/os': 'linux'} @@ -41,8 +46,14 @@ def setUp(self): {'metadata': {'name': 'nvidia-experimental'}, 'handler': 'nvidia-experimental'}], } self.calls = [] + self.probed = [] + self.gpus = {} self.template = json.loads((HERE/'config.example.json').read_text()) + def probe(self, names): + self.probed.append(list(names)) + return {name: copy.deepcopy(self.gpus.get(name, GB10)) for name in names} + def query(self, command, **kwargs): self.calls.append(command) self.assertEqual(command[:4], ['kubectl', '--context', 'test-context', 'get']) @@ -56,11 +67,14 @@ def query(self, command, **kwargs): def discover(self, namespace=None): with patch.object(setup.subprocess, 'check_output', side_effect=self.query): - return setup.discover_config('test-context', namespace) + return setup.discover_config('test-context', namespace, GLM, self.probe) def test_unambiguous_setup_uses_live_metadata_and_public_pins(self): config = self.discover() - self.assertEqual(config['nodes'], {'control': 'routing', 'leader': 'gpu-a', 'worker': 'gpu-b'}) + self.assertEqual(config['nodes'], {'control': 'routing', 'model': ['gpu-a', 'gpu-b']}) + self.assertEqual(config['gpu'], GB10) + self.assertEqual(config['recipe'], 'glm-5.3') + self.assertEqual(self.probed, [['gpu-a', 'gpu-b']]) self.assertEqual(config['context'], 'test-context') self.assertEqual(config['storageClass'], 'local-storage') self.assertEqual(config['runtimeClass'], 'nvidia') @@ -81,20 +95,20 @@ def test_shared_control_plane_labels_use_only_two_idle_gpus(self): for item in self.resources['nodes']: item['metadata']['labels']['node-role.kubernetes.io/control-plane'] = 'true' self.resources['pods'] = [gpu_pod('routing')] - self.assertEqual(self.discover()['nodes'], {'control': 'routing', 'leader': 'gpu-a', 'worker': 'gpu-b'}) + self.assertEqual(self.discover()['nodes'], {'control': 'routing', 'model': ['gpu-a', 'gpu-b']}) def test_three_idle_gpus_with_shared_labels_choose_stable_placement(self): for item in self.resources['nodes']: item['metadata']['labels']['node-role.kubernetes.io/control-plane'] = 'true' config = self.discover() - self.assertEqual(config['nodes'], {'control': 'routing', 'leader': 'gpu-a', 'worker': 'gpu-b'}) + self.assertEqual(config['nodes'], {'control': 'routing', 'model': ['gpu-a', 'gpu-b']}) self.resources['nodes'].reverse() self.assertEqual(self.discover()['nodes'], config['nodes']) def test_more_gpu_workers_do_not_require_an_extra_manual_choice(self): self.resources['nodes'].append(node('gpu-c')) config = self.discover() - self.assertEqual(config['nodes'], {'control': 'routing', 'leader': 'gpu-a', 'worker': 'gpu-b'}) + self.assertEqual(config['nodes'], {'control': 'routing', 'model': ['gpu-a', 'gpu-b']}) self.assertEqual(config['containerd']['nodeNames'], ['gpu-a', 'gpu-b', 'gpu-c', 'routing']) def test_missing_role_labels_can_use_two_idle_gpus_and_non_gpu_routing(self): @@ -107,12 +121,12 @@ def test_gpu_requests_in_normal_or_init_containers_block_selection(self): for init in (False, True): with self.subTest(init=init): self.resources['pods'] = [gpu_pod('gpu-a', init=init)] - with self.assertRaisesRegex(setup.ClusterSetupError, 'Free the required GPUs'): + with self.assertRaisesRegex(setup.ClusterSetupError, 'needs 2 idle NVIDIA GB10 nodes'): self.discover() def test_completed_gpu_pods_do_not_block(self): self.resources['pods'] = [gpu_pod('gpu-a', 'Succeeded'), gpu_pod('gpu-b', 'Failed')] - self.assertEqual(self.discover()['nodes']['leader'], 'gpu-a') + self.assertEqual(self.discover()['nodes']['model'][0], 'gpu-a') def test_pending_and_terminating_gpu_allocations_still_block(self): self.resources['nodes'][0]['status']['allocatable']['nvidia.com/gpu'] = '0' @@ -140,8 +154,32 @@ def test_unready_cordoned_tainted_pressure_nodes_are_not_selected(self): with self.subTest(change=index): self.resources['nodes'][1] = node('gpu-b') change(self.resources['nodes'][1]) - with self.assertRaises(setup.ClusterSetupError): - self.discover() + config = self.discover() + self.assertNotIn('gpu-b', [config['nodes']['control'], *config['nodes']['model']]) + + def test_gb300_pair_serves_from_one_node_and_routes_from_the_control_plane(self): + self.resources['nodes'] = [node('server', True), node('agent')] + self.gpus = {'server': GB300, 'agent': GB300} + config = self.discover() + self.assertEqual(config['nodes'], {'control': 'server', 'model': ['agent']}) + self.assertEqual(config['gpu'], GB300) + self.assertEqual(self.probed, [['agent', 'server']]) + + def test_two_gb10_nodes_split_the_model_and_share_routing_with_the_leader(self): + self.resources['nodes'] = [node('spark-a', True), node('spark-b')] + config = self.discover() + self.assertEqual(config['nodes'], {'control': 'spark-b', 'model': ['spark-b', 'spark-a']}) + + def test_mixed_gpu_types_are_not_combined(self): + self.gpus = {'gpu-b': GB300} + with self.assertRaisesRegex(setup.ClusterSetupError, 'Mixed GPU types'): + self.discover() + + def test_model_that_fits_nowhere_is_rejected_with_its_requirement(self): + small = dict(GB300, memoryGiB=80.0) + self.gpus = {name: small for name in ('gpu-a', 'gpu-b', 'routing')} + with self.assertRaisesRegex(setup.ClusterSetupError, 'does not fit on 2 nodes'): + self.discover() def test_custom_unused_namespace_needs_no_prior_private_settings(self): self.resources['namespaces'] = [{'metadata': {'name': self.template['namespace']}}] @@ -192,19 +230,68 @@ def test_valid_names_are_checked_before_kubernetes_queries(self): with patch.object(setup.subprocess, 'check_output') as query: for value in ('Bad Namespace', '../other', 'x'*64): with self.subTest(value=value), self.assertRaises(setup.ClusterSetupError): - setup.discover_config('test-context', value) + setup.discover_config('test-context', value, GLM, self.probe) with self.assertRaises(setup.ClusterSetupError): - setup.discover_config('') + setup.discover_config('', None, GLM, self.probe) query.assert_not_called() def test_failed_queries_hide_stderr_and_return_actionable_error(self): failure = subprocess.CalledProcessError(1, ['kubectl'], stderr='credential-fixture') with patch.object(setup.subprocess, 'check_output', side_effect=failure), \ self.assertRaises(setup.ClusterSetupError) as result: - setup.discover_config('test-context') + setup.discover_config('test-context', None, GLM, self.probe) self.assertIn('Check Kubernetes access', str(result.exception)) self.assertNotIn('credential-fixture', str(result.exception)) +class ProbeTests(unittest.TestCase): + def test_discrete_gpu_memory_comes_from_nvidia_smi(self): + gpu = setup.parse_probe('NVIDIA GB300, 10.3, 281250\nMemTotal: 503316480 kB\n') + self.assertEqual(gpu, dict(GB300, memoryGiB=274.7)) + + def test_unified_gpu_memory_comes_from_host_memory(self): + for reported in ('[N/A]', '[Not Supported]'): + with self.subTest(reported=reported): + gpu = setup.parse_probe('NVIDIA GB10, 12.1, ' + reported + '\nMemTotal: 127526000 kB\n') + self.assertEqual(gpu, GB10) + + def test_unexpected_probe_output_is_rejected(self): + for text in ('MemTotal: 1 kB\n', 'NVIDIA GB300, 10.3, 1\nNVIDIA GB300, 10.3, 1\nMemTotal: 1 kB\n', + 'NVIDIA GB300, 10.3\nMemTotal: 1 kB\n', 'NVIDIA GB300, 10.3, lots\nMemTotal: 1 kB\n', + 'NVIDIA GB300, 103, 1\nMemTotal: 1 kB\n', 'NVIDIA GB300, 10.3, 1\n'): + with self.subTest(text=text), self.assertRaises(setup.ClusterSetupError): + setup.parse_probe(text) + + def test_probe_runs_gpu_pods_in_a_temporary_namespace_and_always_deletes_it(self): + for fail in (False, True): + calls = [] + + def kubectl(context, *args, stdin=None, timeout=60): + calls.append((args, json.loads(stdin) if stdin else None)) + if args[:1] == ('create',) and fail and stdin: + raise setup.ClusterSetupError('quota exceeded') + if 'logs' in args: + return 'NVIDIA GB300, 10.3, 281250\nMemTotal: 503316480 kB\n' + return '' + + with self.subTest(fail=fail), patch.object(setup, 'kubectl', side_effect=kubectl): + if fail: + with self.assertRaisesRegex(setup.ClusterSetupError, 'quota'): + setup.probe_gpus('test-context', ['agent'], 'runtime:pinned', 'nvidia') + else: + self.assertEqual(setup.probe_gpus('test-context', ['agent'], 'runtime:pinned', 'nvidia'), + {'agent': dict(GB300, memoryGiB=274.7)}) + namespace = calls[0][0][2] + self.assertRegex(namespace, r'^llm-routing-probe-[a-f0-9]{6}$') + self.assertEqual(calls[-1][0][:3], ('delete', 'namespace', namespace)) + pods = [manifest for _, manifest in calls if manifest] + self.assertEqual(len(pods), 1) + spec = pods[0]['spec'] + self.assertEqual(spec['runtimeClassName'], 'nvidia') + self.assertEqual(spec['nodeSelector'], {'kubernetes.io/hostname': 'agent'}) + self.assertEqual(spec['containers'][0]['image'], 'runtime:pinned') + self.assertEqual(spec['containers'][0]['resources']['limits']['nvidia.com/gpu'], 1) + + if __name__ == '__main__': unittest.main() diff --git a/deploy/helm/llm-routing/recipes/tests/test_discovery.py b/deploy/helm/llm-routing/recipes/tests/test_discovery.py index 6e1a6dc11..c824c8a39 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_discovery.py +++ b/deploy/helm/llm-routing/recipes/tests/test_discovery.py @@ -30,7 +30,7 @@ def owned(name, release, node=None): self.resources = {('deployment', 'llm-api-gateway'): self.gateway, ('deployment', 'llm-request-router'): self.router, ('deployment', 'team-glm'): owned('team-glm', 'team-glm', 'leader'), - ('deployment', 'team-glm-rpc-worker'): owned('team-glm-rpc-worker', 'team-glm', 'worker'), + ('deployment', 'team-glm-rpc-n1'): owned('team-glm-rpc-n1', 'team-glm', 'worker'), ('deployment', 'operator'): owned('operator', 'operator', 'control'), ('service', 'team-glm'): owned('team-glm', 'team-glm'), ('inferenceendpoint', 'glm53-iq2'): endpoint} @@ -42,7 +42,9 @@ def owned(name, release, node=None): 'llm-api-gateway': {'llmApiGateway': {'image': {'registry': 'registry.example.com', 'repository': 'gateway', 'tag': 'current', 'pullPolicy': 'Never'}, 'auth': {'mode': 'staticKeys', 'staticKeys': {'existingSecret': 'caller-keys'}}}}, 'llm-request-router': {'llmRequestRouter': {'image': {'registry': 'other.example.com', 'repository': 'router', 'tag': 'previous'}}}}, - 'team-glm': {'targets': [{'id': 'leader', 'node': 'leader'}, {'id': 'worker', 'node': 'worker'}], + 'team-glm': {'targets': [{'id': 'n0', 'node': 'leader'}, {'id': 'n1', 'node': 'worker'}], + 'gpu': {'name': 'NVIDIA GB10', 'computeCapability': '12.1', 'memoryGiB': 121.6, + 'unifiedMemory': True, 'cudaArchitectures': None, 'product': 'NVIDIA-GB10'}, 'artifacts': {'storageClassName': 'storage'}, 'runtimeClassName': 'gpu', 'image': 'cuda/runtime:pinned'}, 'operator': {'clusterId': 'demo', 'watchNamespaces': ['demo'], 'router': {'grpcAddress': 'http://llm-request-router.demo.svc.cluster.local:50071'}, 'fullnameOverride': 'operator', 'trustBundle': {'configMap': 'public-ca'}, @@ -87,6 +89,22 @@ def test_discovers_distinct_repositories_import_settings_and_omits_credentials(s self.assertEqual(config['releasePrefix'], 'shared') self.assertIsNone(config['apiKeyFile']) self.assertNotIn('DO-NOT-COPY', json.dumps(config)) + self.assertEqual(config['recipe'], 'glm-5.3') + self.assertEqual(config['nodes'], {'control': 'control', 'model': ['leader', 'worker']}) + self.assertEqual(config['gpu'], {'name': 'NVIDIA GB10', 'computeCapability': '12.1', 'memoryGiB': 121.6, + 'unifiedMemory': True, 'cudaArchitectures': None}) + + def test_single_node_installation_has_no_rpc_workers(self): + self.values['team-glm']['targets'] = [{'id': 'n0', 'node': 'leader'}] + del self.resources[('deployment', 'team-glm-rpc-n1')] + config = self.discover() + self.assertEqual(config['nodes'], {'control': 'control', 'model': ['leader']}) + self.assertFalse(any('-rpc-' in part for command in self.commands for part in command)) + + def test_unknown_endpoint_is_not_adopted(self): + self.resources[('inferenceendpoint', 'glm53-iq2')]['spec']['modelName'] = 'another-model' + with self.assertRaisesRegex(RuntimeError, 'does not match a recipe'): + self.discover() def test_legacy_archive_directory_is_not_carried_into_new_configuration(self): self.values['shared-images']['archiveDirectory'] = '/unused/old-user-directory' diff --git a/deploy/helm/llm-routing/recipes/tests/test_load_resume.py b/deploy/helm/llm-routing/recipes/tests/test_load_resume.py index 26366c74b..d2c9cacf5 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_load_resume.py +++ b/deploy/helm/llm-routing/recipes/tests/test_load_resume.py @@ -19,36 +19,44 @@ class LoadResumeTests(unittest.TestCase): def setUp(self): self.tmp = tempfile.TemporaryDirectory(prefix='recipe-resume-') self.addCleanup(self.tmp.cleanup) - config = json.loads((HERE/'config.example.json').read_text()) + self.build(json.loads((HERE/'config.example.json').read_text())) + + def build(self, config): self.recipe = tool.Recipe(config, self.tmp.name) self.recipe.state = {'download': True, 'runtimeSha256': 'a'*64} self.values = self.recipe.backend_values('serve') self.release = {'name': self.recipe.backend, 'namespace': config['namespace'], 'version': 6, 'info': {'status': 'deployed'}} self.resources = [] - for suffix, role, container, component in [('', 'leader', 'llama', 'model-server'), - ('-rpc-worker', 'worker', 'rpc', 'rpc-worker'), ('-artifacts', 'leader', 'artifacts', 'artifacts')]: + leader = config['nodes']['model'][0] + deployments = [('', leader, 'llama', 'model-server')] + deployments += [('-rpc-'+w['id'], w['node'], 'rpc', 'rpc-'+w['id']) for w in self.recipe.workers] + deployments += [('-artifacts', leader, 'artifacts', 'artifacts')] + for suffix, node, container, component in deployments: name = self.recipe.backend+suffix labels = {'app.kubernetes.io/instance': self.recipe.backend, 'app.kubernetes.io/component': component} self.resources.append({'kind': 'Deployment', 'metadata': self.metadata(name, generation=2), 'spec': {'replicas': 1, 'selector': {'matchLabels': labels}, 'template': { 'metadata': {'labels': labels, 'annotations': {'checksum/runtime': self.recipe.state['runtimeSha256']}}, 'spec': {'runtimeClassName': config['runtimeClass'], - 'nodeSelector': {'kubernetes.io/hostname': config['nodes'][role]}, + 'nodeSelector': {'kubernetes.io/hostname': node}, 'containers': [{'name': container, 'image': config['runtimeImage'], 'resources': {'requests': {'nvidia.com/gpu': '1'}, 'limits': {'nvidia.com/gpu': '1'}}}]}}}, 'status': {'observedGeneration': 2, 'replicas': 1, 'updatedReplicas': 1, 'readyReplicas': 1, 'availableReplicas': 1}}) for resource in self.resources: pod = resource['spec']['template']['spec'] - volume = 'rpc-cache' if resource['metadata']['name'].endswith('-rpc-worker') else 'artifacts' - pod['volumes'] = [{'name': volume, 'persistentVolumeClaim': {'claimName': self.recipe.backend+'-'+volume}}] + name = resource['metadata']['name'] + worker = name[len(self.recipe.backend+'-rpc-'):] if name.startswith(self.recipe.backend+'-rpc-') else None + volume, claim = ('rpc-cache', self.recipe.backend+'-rpc-cache-'+worker) if worker else ('artifacts', self.recipe.backend+'-artifacts') + pod['volumes'] = [{'name': volume, 'persistentVolumeClaim': {'claimName': claim}}] pod['containers'][0]['volumeMounts'] = [{'name': volume, 'mountPath': '/'+volume}] env = {'FIRST_SHARD': self.values['model']['firstShard'], 'SERVED_MODEL': self.values['model']['servedName'], - 'RPC_ENDPOINT': self.recipe.backend+'-rpc-worker:50052', 'SERVER_ARGS': json.dumps(self.values['model']['args'])} + 'RPC_ENDPOINTS': ','.join(self.recipe.backend+'-rpc-'+w['id']+':50052' for w in self.recipe.workers), + 'SERVER_ARGS': json.dumps(self.values['model']['args'])} self.resources[0]['spec']['template']['spec']['containers'][0]['env'] = [ {'name': key, 'value': value} for key, value in env.items()] - for suffix in ('-artifacts', '-rpc-cache'): + for suffix in ['-artifacts'] + ['-rpc-cache-'+w['id'] for w in self.recipe.workers]: name = self.recipe.backend+suffix self.resources.append({'kind': 'PersistentVolumeClaim', 'metadata': self.metadata(name), 'spec': {'volumeName': name+'-volume', 'storageClassName': config['storageClass']}, @@ -116,9 +124,32 @@ def test_first_load_keeps_memory_preflight_and_helm_apply(self): self.recipe.backend_values('serve'), '70m', jobs=False) memory = [command for command in self.commands if 'exec' in command] self.assertEqual(len(memory), 2) - self.assertTrue(any('deploy/'+self.recipe.backend+'-rpc-leader' in command for command in memory)) + self.assertEqual([command[command.index('exec')+1] for command in memory], + ['deploy/'+self.recipe.backend+'-rpc-n0', 'deploy/'+self.recipe.backend+'-rpc-n1']) self.assertTrue(self.recipe.state['serve']) + def test_single_node_load_resumes_without_rpc_workers(self): + config = json.loads((HERE/'config.example.json').read_text()) + config['nodes']['model'] = ['model-0'] + config['gpu'] = {'name': 'NVIDIA GB300', 'computeCapability': '10.3', 'memoryGiB': 268.0, + 'unifiedMemory': False, 'cudaArchitectures': None} + self.build(config) + self.attempt().assert_not_called() + evidence = json.loads((self.recipe.work/'evidence/load-resume.json').read_text()) + self.assertEqual(sorted(evidence['resources']), ['Deployment/'+self.recipe.backend, 'Deployment/'+self.recipe.backend+'-artifacts', + 'PersistentVolumeClaim/'+self.recipe.backend+'-artifacts']) + + def test_first_single_node_load_checks_only_the_leader_memory(self): + config = json.loads((HERE/'config.example.json').read_text()) + config['nodes']['model'] = ['model-0'] + config['gpu'] = {'name': 'NVIDIA GB300', 'computeCapability': '10.3', 'memoryGiB': 268.0, + 'unifiedMemory': False, 'cudaArchitectures': None} + self.build(config) + self.values = self.recipe.backend_values('download') + self.attempt() + memory = [command[command.index('exec')+1] for command in self.commands if 'exec' in command] + self.assertEqual(memory, ['deploy/'+self.recipe.backend+'-rpc-n0']) + def test_pending_and_failed_release_never_adopts_ready_resources(self): for status in ('pending-upgrade', 'pending-install', 'pending-rollback', 'failed', 'uninstalling'): with self.subTest(status=status): diff --git a/deploy/helm/llm-routing/recipes/tests/test_recipe.py b/deploy/helm/llm-routing/recipes/tests/test_recipe.py index d4b69b1b6..9a6f6e700 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_recipe.py +++ b/deploy/helm/llm-routing/recipes/tests/test_recipe.py @@ -244,10 +244,10 @@ def test_operator_build_identifies_actual_checkout_and_local_edits(self): self.assertEqual(run.call_args.args[0][-1], str(self.recipe.source/tool.COMPONENTS['operator'])) def test_duplicate_model_nodes_and_missing_context_are_rejected(self): - self.config['nodes']['worker'] = self.config['nodes']['leader'] - with self.assertRaisesRegex(RuntimeError, 'distinct GPU'): + self.config['nodes']['model'] = ['model-0', 'model-0'] + with self.assertRaisesRegex(RuntimeError, 'must not repeat'): tool.validate(self.config) - self.config['nodes']['worker'] = 'another-node' + self.config['nodes']['model'] = ['model-0', 'another-node'] self.config['context'] = '' with self.assertRaisesRegex(RuntimeError, 'context'): tool.validate(self.config) @@ -310,7 +310,7 @@ def test_stack_generates_private_key_hash_and_only_glm_stack_components(self): def test_inventory_rejects_all_gpu_allocations_on_model_nodes(self): nodes = [{'metadata': {'name': name, 'uid': name, 'labels': {'kubernetes.io/arch': 'arm64'}}, 'status': {'conditions': [{'type': 'Ready', 'status': 'True'}], - 'allocatable': {'nvidia.com/gpu': '1'}}} for name in self.config['nodes'].values()] + 'allocatable': {'nvidia.com/gpu': '1'}}} for name in tool.all_nodes(self.config)] allocations = [ {'containers': [{'resources': {'requests': {'nvidia.com/gpu': '1'}}}]}, {'containers': [{'resources': {'limits': {'nvidia.com/gpu': '1'}}}]}, @@ -318,10 +318,10 @@ def test_inventory_rejects_all_gpu_allocations_on_model_nodes(self): {'resourceClaims': [{'name': 'gpu', 'resourceClaimName': 'gpu-claim'}]}, ] for allocation in allocations: - for phase, node, busy in [('Pending', self.config['nodes']['leader'], True), - ('Running', self.config['nodes']['worker'], True), - ('Succeeded', self.config['nodes']['leader'], False), - ('Failed', self.config['nodes']['leader'], False), + for phase, node, busy in [('Pending', self.config['nodes']['model'][0], True), + ('Running', self.config['nodes']['model'][1], True), + ('Succeeded', self.config['nodes']['model'][0], False), + ('Failed', self.config['nodes']['model'][0], False), ('Running', 'another-node', False)]: pod = {'metadata': {'name': 'gpu-consumer', 'namespace': 'another-namespace'}, 'spec': {'nodeName': node, **allocation}, 'status': {'phase': phase}} @@ -434,7 +434,7 @@ def test_attached_installation_requires_glm_verification_before_update(self): def test_model_config_keeps_two_gpus_and_scoped_canary(self): values = self.recipe.backend_values(register=True, render=True) - self.assertEqual([t['id'] for t in values['targets']], ['leader', 'worker']) + self.assertEqual(values['targets'], [{'id': 'n0', 'node': 'model-0'}, {'id': 'n1', 'node': 'model-1'}]) self.assertEqual(values['model']['canary'], {'timeoutSeconds': 180, 'intervalSeconds': 60}) self.assertEqual(values['model']['args'][values['model']['args'].index('--parallel')+1], '1') self.assertEqual(values['model']['args'][values['model']['args'].index('--ctx-size')+1], '2048') @@ -765,7 +765,7 @@ def test_existing_attachment_ownership_failure_makes_no_mutation(self): key.write_text('test-only-key') self.config['apiKeyFile'] = str(key) recipe = tool.Recipe(self.config, self.tmp.name) - nodes = {'items': [{'metadata': {'name': name, 'uid': name}} for name in self.config['nodes'].values()]} + nodes = {'items': [{'metadata': {'name': name, 'uid': name}} for name in tool.all_nodes(self.config)]} foreign = {'metadata': {'annotations': {'meta.helm.sh/release-name': 'another-owner'}}} with patch.object(tool, 'output', side_effect=[json.dumps(nodes), json.dumps(foreign)]), patch.object(recipe, 'helm_apply') as helm: with self.assertRaisesRegex(RuntimeError, 'ownership'): diff --git a/deploy/helm/llm-routing/recipes/tests/test_reinitialization.py b/deploy/helm/llm-routing/recipes/tests/test_reinitialization.py index 4c6c85bdb..a85d89d68 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_reinitialization.py +++ b/deploy/helm/llm-routing/recipes/tests/test_reinitialization.py @@ -14,7 +14,7 @@ def setUp(self): self.prefix = self.config['releasePrefix'] self.resources = { 'namespaces': [{'metadata': {'name': self.namespace}}], - 'nodes': [node(name) for name in self.config['nodes'].values()], + 'nodes': [node(name) for name in sorted({self.config['nodes']['control'], *self.config['nodes']['model']})], 'pods': [], 'inferenceendpoints.pylon.nvidia.com': [], 'deployments,statefulsets,daemonsets,jobs,cronjobs,pods,services,ingresses': [], 'persistentvolumeclaims': [], 'persistentvolumes': [], 'secrets': [], @@ -22,7 +22,8 @@ def setUp(self): 'spec': {'group': 'pylon.nvidia.com', 'scope': 'Namespaced', 'names': {'kind': 'InferenceEndpoint'}, 'versions': [{'name': 'v1alpha1', 'served': True, 'storage': True}]}}], } - for suffix, role in [('artifacts', 'leader'), ('rpc-cache', 'worker')]: + model = self.config['nodes']['model'] + for suffix, placed in [('artifacts', model[0]), ('rpc-cache-n1', model[1])]: name = self.prefix+'-glm-'+suffix self.resources['persistentvolumeclaims'].append({ 'metadata': self.metadata(name, self.prefix+'-glm'), 'status': {'phase': 'Bound'}, @@ -31,7 +32,7 @@ def setUp(self): self.resources['persistentvolumes'].append({'metadata': {'name': name}, 'spec': { 'claimRef': {'uid': name, 'name': name, 'namespace': self.namespace}, 'nodeAffinity': {'required': {'nodeSelectorTerms': [{'matchExpressions': [ - {'key': 'kubernetes.io/hostname', 'operator': 'In', 'values': [self.config['nodes'][role]]}]}]}}}}) + {'key': 'kubernetes.io/hostname', 'operator': 'In', 'values': [placed]}]}]}}}}) def metadata(self, name, release): return {'name': name, 'uid': name, 'annotations': {'meta.helm.sh/release-name': release, @@ -105,7 +106,7 @@ def test_foreign_secret_and_incompatible_crd_are_rejected(self): self.validate() def test_busy_gpu_and_terminating_namespace_are_rejected(self): - self.resources['pods'] = [gpu_pod(self.config['nodes']['leader'])] + self.resources['pods'] = [gpu_pod(self.config['nodes']['model'][0])] with self.assertRaisesRegex(setup.ClusterSetupError, 'occupied'): self.validate() self.resources['pods'] = [] diff --git a/deploy/helm/llm-routing/recipes/tests/test_sizing.py b/deploy/helm/llm-routing/recipes/tests/test_sizing.py new file mode 100644 index 000000000..d5ef39895 --- /dev/null +++ b/deploy/helm/llm-routing/recipes/tests/test_sizing.py @@ -0,0 +1,85 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +import importlib.util +import json +import pathlib +import unittest + +HERE = pathlib.Path(__file__).resolve().parents[1] +spec = importlib.util.spec_from_file_location('recipe_sizing', HERE/'sizing.py') +sizing = importlib.util.module_from_spec(spec) +spec.loader.exec_module(sizing) + +GLM = json.loads((HERE/'glm-5.3/recipe.json').read_text()) +GLM['lock'] = json.loads((HERE/'glm-5.3/model.lock.json').read_text()) +GB10 = {'name': 'NVIDIA GB10', 'computeCapability': '12.1', 'memoryGiB': 121.6, 'unifiedMemory': True, 'cudaArchitectures': None} +GB300 = {'name': 'NVIDIA GB300', 'computeCapability': '10.3', 'memoryGiB': 268.0, 'unifiedMemory': False, 'cudaArchitectures': None} + + +class ArchitectureTests(unittest.TestCase): + def test_compute_capability_becomes_an_arch_specific_cmake_target(self): + for capability, expected in (('10.3', '103a-real'), ('12.1', '121a-real'), ('9.0', '90a-real'), ('10.0', '100a-real')): + with self.subTest(capability=capability): + self.assertEqual(sizing.cuda_architectures(dict(GB300, computeCapability=capability)), expected) + + def test_explicit_architectures_pass_through_verbatim(self): + self.assertEqual(sizing.cuda_architectures(dict(GB300, cudaArchitectures='103f;120a-real')), '103f;120a-real') + + def test_malformed_compute_capability_is_rejected(self): + for capability in ('103', '10.10', 'ten.three', '', '10.3.1', 10.3): + with self.subTest(capability=capability), self.assertRaisesRegex(ValueError, 'computeCapability'): + sizing.check_gpu(dict(GB300, computeCapability=capability)) + + def test_gpu_product_label_replaces_spaces(self): + self.assertEqual(sizing.gpu_product(GB300), 'NVIDIA-GB300') + self.assertEqual(sizing.gpu_product(dict(GB300, name=' NVIDIA GB10 ')), 'NVIDIA-GB10') + + def test_gpu_settings_are_validated(self): + sizing.check_gpu(GB300) + for change in ({'name': ''}, {'memoryGiB': 0}, {'memoryGiB': 'big'}, {'unifiedMemory': 'no'}, {'cudaArchitectures': 103}): + with self.subTest(change=change), self.assertRaises(ValueError): + sizing.check_gpu(dict(GB300, **change)) + + +class MemoryTests(unittest.TestCase): + def test_spark_split_reproduces_the_original_hard_coded_numbers(self): + plan = sizing.memory_plan(GLM, GB10, 2) + self.assertTrue(plan['fits']) + self.assertEqual((plan['requestGiB'], plan['limitGiB'], plan['hostAvailableGiB']), (110, 114, 113)) + self.assertIsNone(plan['gpuFreeGiB']) + + def test_glm_does_not_fit_on_one_gb10(self): + self.assertFalse(sizing.memory_plan(GLM, GB10, 1)['fits']) + + def test_glm_fits_on_one_gb300(self): + plan = sizing.memory_plan(GLM, GB300, 1) + self.assertTrue(plan['fits']) + self.assertEqual(plan['gpuFreeGiB'], 223 + GLM['memory']['discrete']['gpuHeadroomGiB']) + self.assertEqual((plan['requestGiB'], plan['limitGiB'], plan['hostAvailableGiB']), (32, 64, 64)) + + def test_model_node_count_is_the_smallest_that_fits(self): + self.assertEqual(sizing.model_node_count(GLM, GB300), 1) + self.assertEqual(sizing.model_node_count(GLM, GB10), 2) + + def test_model_that_fits_nowhere_reports_the_requirement(self): + small = dict(GB300, memoryGiB=80.0) + with self.assertRaisesRegex(ValueError, 'does not fit on 2'): + sizing.model_node_count(GLM, small) + + def test_node_count_outside_the_supported_range_is_rejected(self): + for count in (0, 3): + with self.subTest(count=count), self.assertRaisesRegex(ValueError, '1 or 2'): + sizing.memory_plan(GLM, GB300, count) + + +class PlacementArgumentTests(unittest.TestCase): + def test_single_node_uses_only_the_local_gpu(self): + self.assertEqual(sizing.placement_args(1), ['--device', 'CUDA0']) + + def test_split_adds_one_rpc_device_per_worker_and_an_equal_split(self): + self.assertEqual(sizing.placement_args(2), ['--device', 'CUDA0,RPC0', '--tensor-split', '1,1']) + self.assertEqual(sizing.placement_args(3), ['--device', 'CUDA0,RPC0,RPC1', '--tensor-split', '1,1,1']) + + +if __name__ == '__main__': + unittest.main() diff --git a/deploy/helm/llm-routing/recipes/tests/test_topology.py b/deploy/helm/llm-routing/recipes/tests/test_topology.py new file mode 100644 index 000000000..2f45b5e6c --- /dev/null +++ b/deploy/helm/llm-routing/recipes/tests/test_topology.py @@ -0,0 +1,208 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +"""One-node and split placements: values, rendered charts, preflight and recovery.""" +import copy +import importlib.util +import json +import pathlib +import shutil +import subprocess +import tempfile +import unittest +from unittest.mock import patch + +HERE = pathlib.Path(__file__).resolve().parents[1] +spec = importlib.util.spec_from_file_location('recipe_topology', HERE/'recipe.py') +tool = importlib.util.module_from_spec(spec) +spec.loader.exec_module(tool) + +GB10 = {'name': 'NVIDIA GB10', 'computeCapability': '12.1', 'memoryGiB': 121.6, 'unifiedMemory': True, 'cudaArchitectures': None} +GB300 = {'name': 'NVIDIA GB300', 'computeCapability': '10.3', 'memoryGiB': 268.0, 'unifiedMemory': False, 'cudaArchitectures': None} + + +def configuration(model, gpu, control='control'): + config = json.loads((HERE/'config.example.json').read_text()) + config['nodes'] = {'control': control, 'model': list(model)} + config['gpu'] = copy.deepcopy(gpu) + config['containerd']['nodeNames'] = sorted({control, *model}) + return config + + +SHAPES = { + 'spark-split': configuration(['model-0', 'model-1'], GB10), + 'gb300-single': configuration(['agent'], GB300, control='server'), + 'gb300-split': configuration(['server', 'agent'], GB300, control='server'), +} + + +class TopologyTests(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory(prefix='recipe-topology-') + self.addCleanup(self.tmp.cleanup) + + def recipe(self, shape): + return tool.Recipe(copy.deepcopy(SHAPES[shape]), pathlib.Path(self.tmp.name)/shape) + + def test_spark_split_values_match_the_original_two_gb10_deployment(self): + values = self.recipe('spark-split').backend_values('serve') + self.assertEqual(values['targets'], [{'id': 'n0', 'node': 'model-0'}, {'id': 'n1', 'node': 'model-1'}]) + self.assertEqual(values['model']['args'][-4:], ['--device', 'CUDA0,RPC0', '--tensor-split', '1,1']) + self.assertEqual(values['build']['cudaArchitectures'], '121a-real') + self.assertEqual(values['gpu']['product'], 'NVIDIA-GB10') + for section in (values['model']['resources'], values['rpc']['resources']): + self.assertEqual((section['requests']['memory'], section['limits']['memory']), ('110Gi', '114Gi')) + self.assertTrue(values['rpc']['cache']['enabled']) + + def test_gb300_single_node_values_use_only_the_local_gpu(self): + values = self.recipe('gb300-single').backend_values('serve') + self.assertEqual(values['targets'], [{'id': 'n0', 'node': 'agent'}]) + self.assertEqual(values['model']['args'][-2:], ['--device', 'CUDA0']) + self.assertNotIn('--tensor-split', values['model']['args']) + self.assertEqual(values['build']['cudaArchitectures'], '103a-real') + self.assertEqual(values['gpu']['product'], 'NVIDIA-GB300') + self.assertEqual((values['model']['resources']['requests']['memory'], values['model']['resources']['limits']['memory']), + ('32Gi', '64Gi')) + self.assertFalse(values['rpc']['cache']['enabled']) + + def test_gb300_split_shares_the_leader_with_routing(self): + recipe = self.recipe('gb300-split') + values = recipe.backend_values('serve') + self.assertEqual([t['node'] for t in values['targets']], ['server', 'agent']) + self.assertEqual(recipe.c['nodes']['control'], 'server') + self.assertEqual(values['model']['args'][-4:], ['--device', 'CUDA0,RPC0', '--tensor-split', '1,1']) + + def test_architecture_override_reaches_the_build(self): + config = copy.deepcopy(SHAPES['gb300-single']) + config['gpu']['cudaArchitectures'] = '103f' + recipe = tool.Recipe(config, self.tmp.name) + self.assertEqual(recipe.backend_values('build')['build']['cudaArchitectures'], '103f') + + def test_model_node_count_and_gpu_settings_are_validated(self): + for model in ([], ['a', 'b', 'c']): + with self.subTest(model=model), self.assertRaisesRegex(RuntimeError, '1 or 2'): + tool.validate(configuration(model, GB300)) + with self.assertRaisesRegex(RuntimeError, 'computeCapability'): + tool.validate(configuration(['a'], dict(GB300, computeCapability='103'))) + with self.assertRaisesRegex(RuntimeError, 'nodes.control and nodes.model'): + config = configuration(['a'], GB300) + config['nodes']['leader'] = 'a' + tool.validate(config) + tool.validate(configuration(['a', 'b'], GB300, control='a')) + + +class PreflightTests(TopologyTests): + def record(self, gpu=GB300, available=100*1024**3, free=260*1024**3): + return {'gpu': gpu['name'], 'capability': [int(x) for x in gpu['computeCapability'].split('.')], + 'memoryBefore': {'MemAvailable': available}, 'cudaFreeBytes': free, 'cudaTotalBytes': free} + + def test_matching_hardware_passes(self): + self.recipe('gb300-single').check_preflight([self.record()]) + spark = self.recipe('spark-split') + spark.check_preflight([self.record(GB10, available=114*1024**3, free=1)] * 2) + + def test_hardware_that_differs_from_the_configuration_is_rejected(self): + recipe = self.recipe('gb300-single') + cases = { + 'Every model node': [], + 'differs from gpu settings': [self.record(GB10)], + 'Insufficient host memory': [self.record(available=60*1024**3)], + 'Insufficient GPU memory': [self.record(free=200*1024**3)], + } + for message, records in cases.items(): + with self.subTest(message=message), self.assertRaisesRegex(RuntimeError, message): + recipe.check_preflight(records) + + def test_unified_memory_skips_the_gpu_memory_check_but_not_host_memory(self): + recipe = self.recipe('spark-split') + with self.assertRaisesRegex(RuntimeError, 'more than 113 GiB'): + recipe.check_preflight([self.record(GB10, available=112*1024**3, free=1)] * 2) + + def test_preflight_refuses_a_placement_that_cannot_fit(self): + config = configuration(['model-0'], GB10) + recipe = tool.Recipe(config, self.tmp.name) + recipe.state = {'inventory': {'nodes': {}}} + with patch.object(recipe, 'bound_cluster'), patch.object(recipe, 'helm_apply') as helm, \ + self.assertRaisesRegex(RuntimeError, 'does not fit on 1 node'): + recipe.backend_phase('preflight') + helm.assert_not_called() + + +class RecoveryTests(TopologyTests): + def interrupt(self, shape): + recipe = self.recipe(shape) + recipe.state = {'registered': True} + applied = [] + with patch.object(recipe, 'bound_cluster'), patch.object(recipe, 'verify'), \ + patch.object(recipe, 'helm_apply', side_effect=lambda release, chart, values, *a, **k: applied.append(copy.deepcopy(values))), \ + patch.object(recipe, 'forward', side_effect=RuntimeError('no endpoints')), \ + patch.object(tool, 'output', return_value=json.dumps({'items': []})), patch.object(tool.time, 'sleep'): + recipe.recovery(True, 18443) + return applied + + def test_split_recovery_stops_the_rpc_workers(self): + down, restored = self.interrupt('spark-split') + self.assertEqual(down['rpc']['replicas'], 0) + self.assertNotIn('replicas', down['model']) + self.assertNotIn('replicas', restored['rpc']) + + def test_single_node_recovery_stops_the_model_server(self): + down, restored = self.interrupt('gb300-single') + self.assertEqual(down['model']['replicas'], 0) + self.assertNotIn('replicas', restored['model']) + + +@unittest.skipUnless(shutil.which('helm'), 'helm is required to render the backend chart') +class RenderedChartTests(TopologyTests): + def render(self, shape, phase): + values = self.recipe(shape).backend_values(phase, register=phase == 'serve', render=True) + path = pathlib.Path(self.tmp.name)/(shape+'-'+phase+'.json') + path.write_text(json.dumps(values)) + text = subprocess.run(['helm', 'template', 'llm-poc-glm', str(HERE/'charts/gguf-backend'), '-f', str(path)], + check=True, capture_output=True, text=True).stdout + return [doc for doc in self.documents(text) if doc] + + @staticmethod + def documents(text): + import re + return [chunk for chunk in re.split(r'^---\s*$', text, flags=re.M) if chunk.strip()] + + def names(self, documents, kind): + import re + found = [] + for doc in documents: + if re.search(r'^kind: '+kind+r'$', doc, flags=re.M): + found.append(re.search(r'^ name: (\S+)$', doc, flags=re.M).group(1)) + return found + + def test_single_node_serve_has_no_rpc_workers_or_rpc_caches(self): + docs = self.render('gb300-single', 'serve') + self.assertEqual(self.names(docs, 'Deployment'), ['llm-poc-glm-artifacts', 'llm-poc-glm']) + self.assertEqual(self.names(docs, 'PersistentVolumeClaim'), ['llm-poc-glm-artifacts']) + serve = next(doc for doc in docs if 'name: RPC_ENDPOINTS' in doc) + self.assertIn('{name: RPC_ENDPOINTS, value: ""}', serve) + endpoint = next(doc for doc in docs if 'kind: InferenceEndpoint' in doc) + self.assertIn('gpu: {product: "NVIDIA-GB300"}', endpoint) + + def test_split_serve_runs_one_cached_rpc_worker_per_extra_node(self): + docs = self.render('spark-split', 'serve') + self.assertIn('llm-poc-glm-rpc-n1', self.names(docs, 'Deployment')) + self.assertNotIn('llm-poc-glm-rpc-n0', self.names(docs, 'Deployment')) + self.assertEqual(self.names(docs, 'PersistentVolumeClaim'), ['llm-poc-glm-artifacts', 'llm-poc-glm-rpc-cache-n1']) + serve = next(doc for doc in docs if 'name: RPC_ENDPOINTS' in doc) + self.assertIn('{name: RPC_ENDPOINTS, value: "llm-poc-glm-rpc-n1:50052"}', serve) + + def test_qualification_targets_every_model_gpu(self): + for shape, endpoints in (('gb300-single', 'llm-poc-glm-rpc-n0:50052'), + ('spark-split', 'llm-poc-glm-rpc-n0:50052,llm-poc-glm-rpc-n1:50052')): + with self.subTest(shape=shape): + docs = self.render(shape, 'qualify') + qualify = next(doc for doc in docs if 'name: RPC_ENDPOINTS' in doc) + self.assertIn('{name: RPC_ENDPOINTS, value: "'+endpoints+'"}', qualify) + + def test_build_compiles_for_the_configured_architecture(self): + docs = self.render('gb300-single', 'build') + self.assertTrue(any('{name: CUDA_ARCHITECTURES, value: "103a-real"}' in doc for doc in docs)) + + +if __name__ == '__main__': + unittest.main() From 87c84fcaf83f2484c55130b157a3803603a305c9 Mon Sep 17 00:00:00 2001 From: Kristina Pathak Date: Tue, 6 Oct 2026 16:12:43 -0700 Subject: [PATCH 4/6] docs(llm-routing): document recipes, placement and the GPU settings Retitle the runbook for ARM64 NVIDIA GPU clusters and explain how init chooses one model node or a split, with DGX Spark and GB300 as examples. Document the gpu section for external configurations, how to add a recipe folder, and how to move an installation from the Spark recipe. Update the subtree AGENTS.md guidance to match. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: Kristina Pathak --- deploy/helm/llm-routing/AGENTS.md | 2 +- deploy/helm/llm-routing/README.md | 91 +++++++++++++++++------ deploy/helm/llm-routing/recipes/AGENTS.md | 4 +- deploy/helm/llm-routing/recipes/README.md | 4 +- 4 files changed, 73 insertions(+), 28 deletions(-) diff --git a/deploy/helm/llm-routing/AGENTS.md b/deploy/helm/llm-routing/AGENTS.md index 2b0848e78..af5cd53fe 100644 --- a/deploy/helm/llm-routing/AGENTS.md +++ b/deploy/helm/llm-routing/AGENTS.md @@ -1,6 +1,6 @@ # LLM routing stack -This directory deploys the LLM routing stack and GLM on DGX Spark. Read `README.md` and `recipes/AGENTS.md`. Edit runtime code in its owning source directories in the current checkout. Builds include local edits. Commit IDs are informational. Image updates compare routing chart contents with the fingerprint recorded in the installed stack. +This directory deploys the LLM routing stack and a model recipe on ARM64 NVIDIA GPU clusters, such as DGX Spark and GB300. Read `README.md` and `recipes/AGENTS.md`. Edit runtime code in its owning source directories in the current checkout. Builds include local edits. Commit IDs are informational. Image updates compare routing chart contents with the fingerprint recorded in the installed stack. Run the Python tests and offline Helm render documented in the README. Always specify the Kubernetes context. Packaging validation must not change a live model deployment. diff --git a/deploy/helm/llm-routing/README.md b/deploy/helm/llm-routing/README.md index a9bbb8b90..941801dcd 100644 --- a/deploy/helm/llm-routing/README.md +++ b/deploy/helm/llm-routing/README.md @@ -1,24 +1,35 @@ -# LLM routing stack on DGX Spark +# LLM routing recipes -Deploy the LLM Gateway Stack (LLM API Gateway and request router) and Pylon Operator on an existing ARM64 DGX Spark Kubernetes cluster. The stack serves full GLM-5.3 `UD-IQ2_M` on exactly two GPUs and validates inference through authenticated gateway requests. +Deploy the LLM Gateway Stack (LLM API Gateway and request router), the Pylon Operator and a model recipe on an existing ARM64 Kubernetes cluster with NVIDIA GPUs. Each recipe serves one model and validates inference through authenticated gateway requests. ## Overview -The `recipe.py` installer coordinates the combined gateway/router chart, the Pylon Operator chart and the GLM backend chart. The backend chart builds llama.cpp, downloads and verifies the GGUF model files, runs GLM across the two GPUs, and creates an `InferenceEndpoint` for Pylon to register. +The `recipe.py` tool coordinates the combined gateway/router chart, the Pylon Operator chart and the model backend chart. The backend chart builds llama.cpp, downloads and verifies the model files, runs the model on one or more GPU nodes, and creates an `InferenceEndpoint` for Pylon to register. + +Recipes live in folders under `recipes/`. The only recipe today is `glm-5.3`: GLM-5.3 `UD-IQ2_M`, 222 GiB of GGUF weights. A recipe folder holds the model lock, the llama.cpp server arguments and the memory rules. The tool derives placement and GPU-specific settings from the cluster. Application images and charts use your checkout, including local edits. +## Placement + +The model runs on one node, or is split by layer across two nodes with llama.cpp RPC. `init` picks the smallest node count that fits the model in GPU memory: + +- DGX Spark (GB10): CPU and GPU share about 122 GiB, so GLM-5.3 needs two nodes. +- GB300: one GPU has about 268 GiB of its own memory, so GLM-5.3 fits on one node. + +The gateway, router and operator run on a separate node when the cluster has one, and share the first model node otherwise. A cluster needs at least one GPU node per model node; two nodes are enough for either example. + ## Prerequisites -- Hardware: two dedicated DGX Spark GB10 model nodes with one GPU each, plus a separate ARM64 control node for gateway/router/operator. Each model node needs more than 113 GiB host `MemAvailable` before loading. -- Kubernetes and storage: an existing ARM64 Kubernetes cluster containing these nodes, with pod networking, `cluster.local` DNS, NetworkPolicy enforcement and node-compatible `ReadWriteOnce` persistent storage. Reserve 400 GiB on the leader and 160 GiB on the worker. +- Hardware: idle ARM64 nodes with one NVIDIA GPU each, enough of them for the model (see [Placement](#placement)). Mixed GPU types are not supported. +- Kubernetes and storage: an existing ARM64 Kubernetes cluster containing these nodes, with pod networking, `cluster.local` DNS, NetworkPolicy enforcement and node-compatible `ReadWriteOnce` persistent storage. Reserve 400 GiB on the first model node and 160 GiB on each additional model node. - GPU enablement: model nodes with a working NVIDIA driver, NVIDIA Container Toolkit configured for the container runtime, and a device plugin advertising `nvidia.com/gpu`. Use an installed `RuntimeClass` matching `runtimeClass` in the config (`nvidia` in the example). - Access: kubeconfig for the target cluster. For a kubeconfig with multiple contexts, pass `--context ` to the commands below. Initial installation requires creating namespaces, custom resource definitions (CRDs), role-based access control (RBAC) resources and namespaced Helm resources. - Workstation: Python 3.11+, Git, Helm 3.14+ or Helm 4, and kubectl. Provide access to GitHub, model downloads and container images. The [image build guide](recipes/BUILDING.md) covers build tools and distribution credentials. ## Installation -Follow these steps for the first installation of the stack and GLM model. +Follow these steps for the first installation of the stack and model. ### Configure @@ -31,7 +42,9 @@ Use your existing kubeconfig. python3 recipe.py init ``` - Review the selected nodes and generated configuration. `init` discovers idle GPUs, storage and image preload settings for K3s. The configuration is saved at the printed path in a private work directory. + Review the selected nodes and generated configuration. `init` discovers idle GPUs, storage and image preload settings for K3s. It runs a short GPU probe pod on candidate nodes in a temporary namespace, then deletes the namespace. The probe uses the runtime image, so the first run can take several minutes to pull it. The configuration is saved at the printed path in a private work directory. + + To place the model yourself, edit `nodes.model` in the saved configuration. For example, list two GB300 nodes to split the model across them. `preflight` checks that the placement fits. 2. Render the manifests and inventory the cluster. @@ -62,21 +75,21 @@ python3 recipe.py register python3 recipe.py verify-gateway ``` -1. `preflight`: Check GPU calculations and available memory on both model nodes. The reference GPU environment uses NVIDIA driver `580.178.04` and CUDA 13. Rerun `preflight` and `qualify` after changing these versions. +1. `preflight`: Check GPU calculations, the GPU type and available memory on each model node against the configuration. The tested environments use NVIDIA driver 580 or newer and CUDA 13. Rerun `preflight` and `qualify` after changing these versions. 2. `stack`: Install the gateway, router and Pylon Operator. This generates the default caller API key and self-signed certificates. To supply your own, complete [Optional configuration](#optional-configuration) before running `stack`. 3. `build-runtime`: Build llama.cpp with CUDA and remote procedure call (RPC) support. -4. `qualify`: Test calculations and data transfer across both GPUs. (After fixing a failed qualification Job, run `python3 recipe.py qualify --retry`.) -5. `download`: Download the six GLM files and verify their sizes and SHA256 checksums. See the [model and runtime licenses](recipes/NOTICE). -6. `load`: Load GLM across both GPUs and wait for the model server. +4. `qualify`: Test calculations on each model GPU, and data transfer between them when the model is split. (After fixing a failed qualification Job, run `python3 recipe.py qualify --retry`.) +5. `download`: Download the model files and verify their sizes and SHA256 checksums. See the [runtime](recipes/NOTICE) and [model](recipes/glm-5.3/NOTICE) licenses. +6. `load`: Load the model on its GPUs and wait for the model server. 7. `verify-direct`: Test model answers and streaming directly. -8. `register`: Register GLM with Pylon and wait for readiness. -9. `verify-gateway`: Test GLM answers, streaming, authentication, discovery and registration through the gateway. For failures, see [Gateway check troubleshooting](#gateway-check-failures). +8. `register`: Register the model with Pylon and wait for readiness. +9. `verify-gateway`: Test model answers, streaming, authentication, discovery and registration through the gateway. For failures, see [Gateway check troubleshooting](#gateway-check-failures). -Pinned runtime: +Pinned `glm-5.3` runtime: - Model revision: `346b3591c7f28d1a23716f97a065ecf12ec14771`, with 238,577,585,701 bytes across six GGUF shards. - llama.cpp revision: `f872b591121761ac7b2af18283bd99bdc092a63a`. -- Capacity: two model nodes, equal layer split, context 2048 and one request slot. +- Capacity: equal layer split across the model nodes, context 2048 and one request slot. ## Verification @@ -116,7 +129,7 @@ If this workstation has not used the running installation before, [attach to it 4. Save the printed update record path for rollback. -GLM stays loaded. Gateway updates briefly interrupt requests. Router updates reconnect Pylon transports. Image-only updates require unchanged routing charts. For chart or API changes, follow [Update the stack](#update-the-stack). +The model stays loaded. Gateway updates briefly interrupt requests. Router updates reconnect Pylon transports. Image-only updates require unchanged routing charts. For chart or API changes, follow [Update the stack](#update-the-stack). To roll back the image update: @@ -155,7 +168,7 @@ Continue with [Update only gateway or router](#update-only-gateway-or-router). Run the entire block, including parentheses, from `deploy/helm/llm-routing/recipes` in the same configured terminal used for installation. The context lookup uses the recipe's normal selection. If you passed `--context`, `--config` or `--work-dir` during installation, pass the same options before `context` in the lookup below. -The namespace and release names below are the default K3s recipe values. If you changed `namespace`, `releasePrefix` or `releases` in your saved configuration, replace these names to match. The model chain release is the GLM release name plus `-chain`, and the image-import release is the release prefix plus `-images`. +The namespace, release and endpoint names below are the defaults for the `glm-5.3` recipe. If you changed `namespace`, `releasePrefix` or `releases` in your saved configuration, replace these names to match. The model release is the release prefix plus the recipe's `releaseName`, and the endpoint name is the recipe's `endpointName`. The chain release, `-chain`, exists only for split models. The image-import release is the release prefix plus `-images`. The block skips absent releases, including the optional image importer, and stops on other failures. It keeps the operator running until endpoint cleanup finishes. @@ -202,7 +215,7 @@ The stack records a SHA-256 fingerprint of its routing chart files. Image update ### Recovery and limits -This optional resilience check restarts the RPC worker, interrupts model service and verifies inference after recovery. +This optional resilience check interrupts model service and verifies inference after recovery. A split model loses its RPC workers; a single-node model loses its model server. 1. Schedule a time when a model interruption is acceptable. 2. Run the explicit recovery check. @@ -213,22 +226,30 @@ This optional resilience check restarts the RPC worker, interrupts model service 3. Save and review the results from the target cluster. -The tested setup took about 26 minutes for a cold load and 11 minutes for recovery with cached weights. +On two DGX Spark nodes, a cold load took about 26 minutes and recovery with cached weights about 11 minutes. Memory and runtime limits: -- CPU and GPU share memory. Monitor host `MemAvailable` when changing context size or adding workloads. +- On GPUs that share memory with the CPU, such as GB10, monitor host `MemAvailable` when changing context size or adding workloads. - The runtime stops below 1 GiB available memory or when the model process/container swaps. -- GLM uses two-bit quantization, context 2048, one request slot and TCP/RPC. -- GLM canary timing is 180 seconds for the timeout and 60 seconds for the interval. +- `glm-5.3` uses two-bit quantization, context 2048, one request slot and TCP/RPC between split nodes. +- `glm-5.3` canary timing is 180 seconds for the timeout and 60 seconds for the interval. -Both model persistent volume claims (PVCs) remain after uninstall. +The model persistent volume claims (PVCs) remain after uninstall. ## Optional configuration ### Alternative container runtimes and external configuration -For a non-K3s cluster or custom container runtime, prepare an external copy of [config.example.json](recipes/config.example.json) before installation. Set the context, node placement, storage, runtime and image settings for your cluster. Use that file instead of `init`, and pass `--config /path/to/config.json` to each recipe command, starting with `render` and `inventory`. +For a non-K3s cluster or custom container runtime, prepare an external copy of [config.example.json](recipes/config.example.json) before installation. Set the context, node placement, GPU, storage, runtime and image settings for your cluster. Use that file instead of `init`, and pass `--config /path/to/config.json` to each recipe command, starting with `render` and `inventory`. + +The `gpu` section describes the GPU on every model node: + +- `name`: the name `nvidia-smi` reports, such as `NVIDIA GB300`. The `InferenceEndpoint` GPU product is derived from it. +- `computeCapability`: such as `"10.3"`. The llama.cpp build targets it as `103a-real`. +- `memoryGiB`: GPU memory, or host memory when `unifiedMemory` is `true`. +- `unifiedMemory`: `true` when the GPU shares system memory, as on GB10. +- `cudaArchitectures`: `null`, or a CMake architecture list that replaces the derived build target. ### Runtime image mirror @@ -258,6 +279,28 @@ To use existing certificates, complete these steps before running `stack`: 2. Create Secrets `llm-gateway-stack-gateway-tls` and `llm-gateway-stack-router-tls` in the namespace with valid `tls.crt` and `tls.key` fields. 3. Create the configured CA ConfigMap with a `ca.crt` field. Certificates must cover the configured service names and client address. +## Add a recipe + +Create a folder under `recipes/` with these files: + +- `recipe.json`: the name (matching the folder), release and endpoint names, llama.cpp revision, server arguments, runtime environment, storage sizes and memory rules. Copy `glm-5.3/recipe.json` as a starting point. Do not set `--device`, `--tensor-split` or `--rpc`; the tool derives them from `nodes.model`. +- `model.lock.json`: the pinned model files with sizes and SHA256 checksums. +- `NOTICE`: the model license terms. + +Select the recipe with `python3 recipe.py init --recipe `. When only one recipe exists, `init` uses it. + +## Migrating from the Spark recipe + +This tool replaces `deploy/helm/llm-routing/spark/spark.py`. Configuration and stored values changed without compatibility fallbacks: + +- `SPARK_CONTEXT` is now `LLM_ROUTING_CONTEXT`. +- `nodes.leader` and `nodes.worker` are now the `nodes.model` list, and a `gpu` section is required. +- `releases.glm` is now `releases.model`. +- The stack records `recipeSource` and `recipeChartsSha256` instead of the `sparkRecipe` values. +- Model resources are renamed, for example `rpc-worker` to `rpc-n1`. + +To move an existing installation, [uninstall](#uninstall) it with the Spark recipe's names, then run `init` and deploy again. + ## Troubleshooting If a command fails, follow the next check and diagnostic log path printed by the CLI. Detailed tool output is saved in private `evidence/*.log` files inside the work directory. Use `python3 recipe.py paths` to locate that directory. diff --git a/deploy/helm/llm-routing/recipes/AGENTS.md b/deploy/helm/llm-routing/recipes/AGENTS.md index b9fe6459f..bcba82c08 100644 --- a/deploy/helm/llm-routing/recipes/AGENTS.md +++ b/deploy/helm/llm-routing/recipes/AGENTS.md @@ -1,9 +1,11 @@ -# Spark GLM recipe +# LLM routing recipe tool This entry point uses the checkout containing `recipe.py`. Builds include local edits and record the current commit ID for debugging. Image updates and rollback require the routing chart fingerprint recorded at installation. Helm dependencies are built automatically. `--source-dir` selects another existing checkout. Keep test mocks under `tests` and out of the normal model deployment. Keep environment-specific values, credentials, kubeconfigs, generated TLS material and runtime evidence outside this repository. Pass an explicit context on every Kubernetes and Helm command. Do not change a live cluster while testing packaging. +Keep model-specific settings in the recipe folder (`recipe.json`, `model.lock.json`, `NOTICE`), not in `recipe.py` or the charts. Keep `sizing.py` free of cluster access. Placement must work for one model node and for a split across nodes; cover both in `tests/test_topology.py`. + Run `python3 -m unittest discover -s tests -v` and the chart tests under `charts/gguf-backend/tests`. Run `python3 recipe.py --config config.example.json --work-dir /tmp/recipe-render render` for offline chart validation. A render does not establish a fresh-cluster deployment. Land runtime and chart fixes in their owning source directories with generated API files and tests. Never hand-edit generated CRDs or deepcopy code. diff --git a/deploy/helm/llm-routing/recipes/README.md b/deploy/helm/llm-routing/recipes/README.md index dfb0d06d7..89e103e83 100644 --- a/deploy/helm/llm-routing/recipes/README.md +++ b/deploy/helm/llm-routing/recipes/README.md @@ -1,3 +1,3 @@ -# Spark recipe implementation +# Recipe tool implementation -Use the [standalone Spark and GLM runbook](../README.md). This directory contains its pinned dependencies, runner, model charts, client and regression tests. +Use the [LLM routing recipes runbook](../README.md). This directory contains the `recipe.py` tool, its model charts, client and regression tests. Each subfolder with a `recipe.json`, such as `glm-5.3`, is one recipe. From 98335b122cc18999a19b4b43230cb763124634fa Mon Sep 17 00:00:00 2001 From: Kristina Pathak Date: Tue, 6 Oct 2026 16:13:47 -0700 Subject: [PATCH 5/6] fix(llm-routing): report a failed GPU probe without waiting for timeout Poll the probe pod phase instead of waiting for Succeeded, so a pod that fails to start or cannot use its GPU is reported at once with its log. init also rejects a missing recipe. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: Kristina Pathak --- .../helm/llm-routing/recipes/cluster_setup.py | 18 +++++++++--- .../recipes/tests/test_cluster_setup.py | 28 +++++++++++++++++++ 2 files changed, 42 insertions(+), 4 deletions(-) diff --git a/deploy/helm/llm-routing/recipes/cluster_setup.py b/deploy/helm/llm-routing/recipes/cluster_setup.py index 91602529c..040158870 100644 --- a/deploy/helm/llm-routing/recipes/cluster_setup.py +++ b/deploy/helm/llm-routing/recipes/cluster_setup.py @@ -17,6 +17,7 @@ " && grep '^MemTotal:' /proc/meminfo") # nvidia-smi reports no dedicated GPU memory when the GPU shares system memory, as on GB10. UNREPORTED_MEMORY = {'[N/A]', 'N/A', '[Not Supported]', 'Not Supported'} +PROBE_TIMEOUT_SECONDS = 20 * 60 class ClusterSetupError(RuntimeError): @@ -130,10 +131,18 @@ def probe_gpus(context, names, image, runtime_class): pods[name] = pod results = {} for name, pod in pods.items(): - # The first pull of the runtime image can take several minutes. - kubectl(context, '-n', namespace, 'wait', 'pod/' + pod, '--for=jsonpath={.status.phase}=Succeeded', - '--timeout=20m', timeout=1260) - results[name] = parse_probe(kubectl(context, '-n', namespace, 'logs', pod)) + # The first pull of the runtime image can take several minutes; a failed pod is reported at once. + deadline = time.monotonic() + PROBE_TIMEOUT_SECONDS + while True: + phase = kubectl(context, '-n', namespace, 'get', 'pod', pod, '-o', 'jsonpath={.status.phase}').strip() + if phase in ('Succeeded', 'Failed'): + break + require(time.monotonic() < deadline, 'GPU probe on ' + name + ' did not finish within ' + + str(PROBE_TIMEOUT_SECONDS // 60) + ' minutes. Last phase: ' + (phase or 'unknown') + '.') + time.sleep(5) + logs = kubectl(context, '-n', namespace, 'logs', pod) + require(phase == 'Succeeded', 'GPU probe failed on ' + name + ': ' + logs.strip()[-500:]) + results[name] = parse_probe(logs) return results finally: try: @@ -148,6 +157,7 @@ def discover_config(context, namespace=None, recipe=None, probe=None): probe(names) returns {node: gpu}; tests replace it to avoid touching a cluster. """ require(isinstance(context, str) and bool(context.strip()), 'Select a Kubernetes context first.') + require(isinstance(recipe, dict) and recipe.get('name'), 'Select a recipe first.') config = json.loads((HERE/'config.example.json').read_text()) namespace = namespace or config['namespace'] require(isinstance(namespace, str) and len(namespace) <= 63 diff --git a/deploy/helm/llm-routing/recipes/tests/test_cluster_setup.py b/deploy/helm/llm-routing/recipes/tests/test_cluster_setup.py index 2bb813d56..db307bf7e 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_cluster_setup.py +++ b/deploy/helm/llm-routing/recipes/tests/test_cluster_setup.py @@ -245,6 +245,32 @@ def test_failed_queries_hide_stderr_and_return_actionable_error(self): class ProbeTests(unittest.TestCase): + def test_a_failed_probe_pod_is_reported_without_waiting(self): + phases = iter(['Pending', 'Failed']) + + def kubectl(context, *args, stdin=None, timeout=60): + if 'jsonpath={.status.phase}' in args: + return next(phases) + if 'logs' in args: + return 'Failed to initialize NVML: Driver/library version mismatch' + return '' + + with patch.object(setup, 'kubectl', side_effect=kubectl), patch.object(setup.time, 'sleep') as sleep, \ + self.assertRaisesRegex(setup.ClusterSetupError, 'GPU probe failed on agent: Failed to initialize NVML'): + setup.probe_gpus('test-context', ['agent'], 'runtime:pinned', 'nvidia') + sleep.assert_called_once() + + def test_a_probe_that_never_finishes_times_out(self): + clock = iter([0, 0, setup.PROBE_TIMEOUT_SECONDS + 1]) + + def kubectl(context, *args, stdin=None, timeout=60): + return 'Pending' if 'jsonpath={.status.phase}' in args else '' + + with patch.object(setup, 'kubectl', side_effect=kubectl), patch.object(setup.time, 'sleep'), \ + patch.object(setup.time, 'monotonic', side_effect=lambda: next(clock)), \ + self.assertRaisesRegex(setup.ClusterSetupError, 'did not finish within 20 minutes. Last phase: Pending'): + setup.probe_gpus('test-context', ['agent'], 'runtime:pinned', 'nvidia') + def test_discrete_gpu_memory_comes_from_nvidia_smi(self): gpu = setup.parse_probe('NVIDIA GB300, 10.3, 281250\nMemTotal: 503316480 kB\n') self.assertEqual(gpu, dict(GB300, memoryGiB=274.7)) @@ -272,6 +298,8 @@ def kubectl(context, *args, stdin=None, timeout=60): raise setup.ClusterSetupError('quota exceeded') if 'logs' in args: return 'NVIDIA GB300, 10.3, 281250\nMemTotal: 503316480 kB\n' + if 'jsonpath={.status.phase}' in args: + return 'Succeeded' return '' with self.subTest(fail=fail), patch.object(setup, 'kubectl', side_effect=kubectl): From 6c7866f061ff9228097b44c5130eccd392b3fea3 Mon Sep 17 00:00:00 2001 From: Kristina Pathak Date: Tue, 6 Oct 2026 16:28:48 -0700 Subject: [PATCH 6/6] fix(llm-routing): show init placement, rename build Jobs on input change - The console only passes through known line prefixes, so init hid the new Recipe, GPU and Model nodes lines. Add those prefixes. - The build Job name hashed only build.py. With the CUDA architectures now derived from the gpu settings, correcting them and rerunning build-runtime would update an immutable Job. Hash the llama.cpp revision and architectures into the name too. - The documented offline render needed a prior init. Use the example configuration with an explicit context and a temporary work directory. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: Kristina Pathak --- deploy/helm/llm-routing/README.md | 4 ++-- .../recipes/charts/gguf-backend/templates/build.yaml | 3 ++- deploy/helm/llm-routing/recipes/console_output.py | 4 ++-- .../llm-routing/recipes/tests/test_console_output.py | 11 +++++++++++ .../helm/llm-routing/recipes/tests/test_topology.py | 11 +++++++++++ 5 files changed, 28 insertions(+), 5 deletions(-) diff --git a/deploy/helm/llm-routing/README.md b/deploy/helm/llm-routing/README.md index 941801dcd..02fee27aa 100644 --- a/deploy/helm/llm-routing/README.md +++ b/deploy/helm/llm-routing/README.md @@ -319,11 +319,11 @@ python3 recipe.py verify-gateway ## Local validation -From the recipe directory, run the runner/client tests, runtime chart tests and offline render checks. +From the recipe directory, run the runner/client tests, runtime chart tests and offline render checks. The render uses the example configuration, so it needs no cluster, context or `init`. ```bash python3 -m unittest discover -s tests -v python3 -m unittest discover -s charts/gguf-backend/tests -v -python3 recipe.py render +python3 recipe.py --context llm-routing-demo --config config.example.json --work-dir "$(mktemp -d)" render git diff --check ``` diff --git a/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/build.yaml b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/build.yaml index dc76a5147..becf48c0a 100644 --- a/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/build.yaml +++ b/deploy/helm/llm-routing/recipes/charts/gguf-backend/templates/build.yaml @@ -29,7 +29,8 @@ data: apiVersion: batch/v1 kind: Job metadata: - name: {{ .Release.Name }}-build-{{ .Files.Get "files/build.py" | sha256sum | trunc 8 }} + # Job templates are immutable, so any change to the build inputs needs a new Job name. + name: {{ .Release.Name }}-build-{{ list (.Files.Get "files/build.py") .Values.build.revision .Values.build.cudaArchitectures | join "\n" | sha256sum | trunc 8 }} spec: backoffLimit: 0 activeDeadlineSeconds: 7200 diff --git a/deploy/helm/llm-routing/recipes/console_output.py b/deploy/helm/llm-routing/recipes/console_output.py index fea90b6e0..4e45b3028 100644 --- a/deploy/helm/llm-routing/recipes/console_output.py +++ b/deploy/helm/llm-routing/recipes/console_output.py @@ -112,7 +112,7 @@ def run(self, phase, work, action): with self.log_path.open() as saved_log: lines = [line.rstrip() for line in saved_log if line.startswith(('Helm lint/render passed for', 'Configuration created:', 'Reusing configuration:', - 'Previous progress archived:', 'Model GPUs:', 'Routing node:', 'Run render,', + 'Previous progress archived:', 'Recipe:', 'GPU:', 'Model nodes:', 'Routing node:', 'Run render,', 'Image archive:', 'Update recorded:')) or re.match(r'^(?:WARNING|Warning|warning)[: ]|^W[0-9]{4} ', line)] for line in lines: if re.match(r'^(?:WARNING|Warning|warning)[: ]|^W[0-9]{4} ', line): @@ -124,7 +124,7 @@ def run(self, phase, work, action): print(phase + ' passed.', file=self.terminal) prefixes = { 'init': ('Configuration created:', 'Reusing configuration:', 'Previous progress archived:', - 'Model GPUs:', 'Routing node:', 'Run render,'), + 'Recipe:', 'GPU:', 'Model nodes:', 'Routing node:', 'Run render,'), 'export-images': ('Image archive:',), 'update': ('Update recorded:',), }.get(phase, ()) diff --git a/deploy/helm/llm-routing/recipes/tests/test_console_output.py b/deploy/helm/llm-routing/recipes/tests/test_console_output.py index 94c742454..64dd6316a 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_console_output.py +++ b/deploy/helm/llm-routing/recipes/tests/test_console_output.py @@ -123,6 +123,17 @@ def test_render_keeps_configuration_count(self): self.assertIn('Helm lint/render passed for 8 configurations.', self.stdout.getvalue()) self.assertNotIn('render passed.\n', self.stdout.getvalue()) + def test_init_shows_the_recipe_gpu_and_placement(self): + config = json.loads((HERE/'config.example.json').read_text()) + config['context'] = 'console-test' + work = self.work/'init' + with self.captured(), patch.object(tool.cluster_setup, 'discover_config', return_value=config): + self.console.run('init', work, lambda: tool.main(['--context', 'console-test', '--work-dir', str(work), 'init'])) + shown = self.stdout.getvalue() + for line in ('Recipe: glm-5.3', 'GPU: NVIDIA GB10 (12.1, 121.6 GiB shared with the CPU)', + 'Model nodes: model-0, model-1', 'Routing node: control', 'Configuration created:'): + self.assertIn(line, shown) + def cli_arguments(self): config = json.loads((HERE/'config.example.json').read_text()) config['context'] = 'console-test' diff --git a/deploy/helm/llm-routing/recipes/tests/test_topology.py b/deploy/helm/llm-routing/recipes/tests/test_topology.py index 2f45b5e6c..459345c66 100644 --- a/deploy/helm/llm-routing/recipes/tests/test_topology.py +++ b/deploy/helm/llm-routing/recipes/tests/test_topology.py @@ -199,6 +199,17 @@ def test_qualification_targets_every_model_gpu(self): qualify = next(doc for doc in docs if 'name: RPC_ENDPOINTS' in doc) self.assertIn('{name: RPC_ENDPOINTS, value: "'+endpoints+'"}', qualify) + def test_build_job_name_changes_with_the_build_inputs(self): + def job(shape, **gpu): + if gpu: + SHAPES[shape]['gpu'].update(gpu) + self.addCleanup(SHAPES[shape]['gpu'].update, {key: GB300[key] for key in gpu}) + docs = self.render(shape, 'build') + return [name for name in self.names(docs, 'Job') if '-build-' in name][0] + first = job('gb300-single') + self.assertEqual(job('gb300-single'), first) + self.assertNotEqual(job('gb300-single', cudaArchitectures='103f'), first) + def test_build_compiles_for_the_configured_architecture(self): docs = self.render('gb300-single', 'build') self.assertTrue(any('{name: CUDA_ARCHITECTURES, value: "103a-real"}' in doc for doc in docs))