From 17b96e43ebb31b4df10b47282a991dc6a4e7cbe7 Mon Sep 17 00:00:00 2001 From: Atif Mahmood Date: Mon, 24 Aug 2026 12:02:31 -0400 Subject: [PATCH 1/3] feat(recipes): OKE RDMA fabric wiring (L40S RoCE + GB200 IB) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Upstream the OKE network fabric, closing gb200-oke-training's 'NET/RDMA intentionally left out until OCI-specific pod RDMA exposure is verified on the testbed' carve-out — the exposure below is validated on a production BM.GPU.GB200.4 NVL72 rack and a BM.GPU.L40S.4 RoCE cluster. - network-operator on both OKE training chains, NicClusterPolicy supplied by manifest (chart deployCR off). L40S (RoCE): SR-IOV VF device plugin advertising nvidia.com/mlnxnics (ConnectX VF device IDs 101a/101e) plus nv-ipam and multus. GB200 (IB): rdmaSharedDevicePlugin over the NVL72 east-west rdma0-3 netdevs, same nvidia.com/mlnxnics resource name; no SR-IOV/nv-ipam. Neither deploys ofedDriver: OCI nodes carry host MOFED in every image. Present in every gpuStack value (fabric is orthogonal to driver/plugin ownership); incompatible with Oracle's opt-in NvidiaNetworkOperator add-on. - GB200 kernel-module-params wiring (NVreg_GrdmaPciTopoCheckOverride=1): dma-buf attach over the IB fabric — GPUDirect RDMA without nvidia-peermem, whose chroot modprobe fails against the -64k Grace kernel. - nccl-all-reduce-bw-net (>= 40, matching gb200-eks-training) added to the gb200-oke training chain; supportedNCCLCombinations[variantNET] gains oke/gb200 with the ported testdata/gb200/oke/runtime-net.yaml TrainingRuntime (IB via the shared HCAs; NVLS/MNNVL forced off). - NicClusterPolicy image digest exemptions (repository/image/version triplet CRD schema, same as the AKS entries). Stock-render golden and BOM regenerated. Signed-off-by: Atif Mahmood --- docs/user/validation.md | 16 +- pkg/recipe/performance_goals_oke_test.go | 9 +- .../nic-cluster-policy-oke-gb200.yaml | 62 +++++ .../nic-cluster-policy-oke-l40s.yaml | 87 +++++++ .../network-operator/values-oke-gb200.yaml | 39 ++++ .../network-operator/values-oke-l40s.yaml | 48 ++++ recipes/manifest_images_test.go | 34 ++- recipes/overlays/gb200-oke-training.yaml | 48 +++- recipes/overlays/l40s-oke-training.yaml | 20 ++ validators/deployment/expected_resources.go | 72 +++--- .../expected_resources_rdma_test.go | 14 +- .../deployment/expected_resources_test.go | 5 +- validators/deployment/rdma_fabric_resource.go | 213 ++++++++++++++++++ .../deployment/rdma_fabric_resource_test.go | 207 +++++++++++++++++ .../nccl_all_reduce_bw_constraint.go | 17 +- .../nccl_benchmark_profile_test.go | 2 +- .../performance/nccl_preflight_nvreg.go | 47 ++-- .../performance/nccl_preflight_nvreg_test.go | 8 +- .../testdata/gb200/oke/runtime-net.yaml | 208 +++++++++++++++++ 19 files changed, 1075 insertions(+), 81 deletions(-) create mode 100644 recipes/components/network-operator/manifests/nic-cluster-policy-oke-gb200.yaml create mode 100644 recipes/components/network-operator/manifests/nic-cluster-policy-oke-l40s.yaml create mode 100644 recipes/components/network-operator/values-oke-gb200.yaml create mode 100644 recipes/components/network-operator/values-oke-l40s.yaml create mode 100644 validators/deployment/rdma_fabric_resource.go create mode 100644 validators/deployment/rdma_fabric_resource_test.go create mode 100644 validators/performance/testdata/gb200/oke/runtime-net.yaml diff --git a/docs/user/validation.md b/docs/user/validation.md index 53587cda58..da9ae3639b 100644 --- a/docs/user/validation.md +++ b/docs/user/validation.md @@ -50,8 +50,8 @@ ones) that match the target fabric: | Check | Transport | Default applicability (from recipe criteria) | |---|---|---| | `nccl-all-reduce-bw` | Auto-detect (whatever NCCL picks) | H100/H200 on EKS, H100 on GKE, H100 on AKS (ND-series InfiniBand — NCCL's built-in IB/verbs transport over the `rdma/hca_shared_devices_a` shared device pool), and B200/GB200 on self-managed clusters (`service=any`). Preserves the pre-variant behavior. | -| `nccl-all-reduce-bw-net` | NET (EFA on EKS by default; ConnectX RoCE via `AICR_NCCL_FABRIC=roce`) | GB200 + EKS. Asserts EFA actually carried traffic — catches silent fallback to Socket when the NVIDIA driver is missing `NVreg_GrdmaPciTopoCheckOverride=1`. | -| `nccl-all-reduce-bw-nvls` | NVLS (MNNVL across an NVL72 IMEX domain) | GB200 + EKS, and GB200 + OKE. Asserts the NVLS communicator actually initialized — catches silent fallback to EFA (EKS) or Socket (OKE) when the IMEX domain is misconfigured. | +| `nccl-all-reduce-bw-net` | NET (EFA on EKS by default; ConnectX RoCE via `AICR_NCCL_FABRIC=roce`; built-in IB/verbs on OKE) | GB200 + EKS, and GB200 + OKE. Asserts the intended NET fabric actually carried traffic — EFA on EKS, the NVL72 InfiniBand east-west fabric (`nvidia.com/mlnxnics` shared HCAs) on OKE — catching silent fallback to Socket when the NVIDIA driver is missing `NVreg_GrdmaPciTopoCheckOverride=1`. | +| `nccl-all-reduce-bw-nvls` | NVLS (MNNVL across an NVL72 IMEX domain) | GB200 + EKS, and GB200 + OKE. Asserts the NVLS communicator actually initialized — catches silent fallback to the NET fabric (EFA on EKS, InfiniBand on OKE) when the IMEX domain is misconfigured. | The applicability column is the *default*, derived from the recipe's `criteria`. A recipe whose criteria fall outside it can still run these @@ -61,7 +61,7 @@ whose fabric matches its hardware, or, for a private service whose fabric matches no embedded template, by [supplying its own benchmark runtime](#supplying-a-benchmark-runtime-for-a-private-service). -The `-net` check defaults to the AWS EFA fabric. On a ConnectX **RoCE** cluster +On EKS, the `-net` check defaults to the AWS EFA fabric. On a ConnectX **RoCE** cluster (e.g. DGXC GB300 `p6e-gb300r`), set `AICR_NCCL_FABRIC=roce` in the `aicr validate` environment to run the NET test over NCCL's built-in IB/verbs transport across `roce.networking.k8s.aws` DRA devices instead. The value is @@ -78,9 +78,13 @@ GB200/EKS recipes (both `training` and `inference` intents) enable `-net` and expose two inter-node fabrics simultaneously and a single auto-detect test would only exercise one of them. -GB200/OKE recipes enable `-nvls` only: OKE NET/RDMA stays out of the support -matrix until the OCI testbed proves a non-Socket NCCL transport end to end, so -OKE validates the NVL72 IMEX fabric without an EFA/NET counterpart. +GB200/OKE training recipes follow the same pattern and enable `-net` and +`-nvls` together, with the same `>= 40` / `>= 500` GB/s floors as GB200/EKS. +On OKE, `-net` exercises the NVL72 rack's InfiniBand east-west fabric +(`rdma0-3`, advertised as `nvidia.com/mlnxnics` by the recipe's +`rdmaSharedDevicePlugin` NicClusterPolicy) over NCCL's built-in IB/verbs +transport — no EFA or RoCE plumbing is involved. Both variants were validated +end to end on a `BM.GPU.GB200.4` NVL72 rack. ```bash # Capture snapshot, generate training recipe, validate the performance phase. diff --git a/pkg/recipe/performance_goals_oke_test.go b/pkg/recipe/performance_goals_oke_test.go index b6e8e77d87..18a32d742a 100644 --- a/pkg/recipe/performance_goals_oke_test.go +++ b/pkg/recipe/performance_goals_oke_test.go @@ -34,22 +34,25 @@ func TestOKEPerformanceGoalsFollowTrainingInferencePattern(t *testing.T) { }{ { name: "gb200-oke-training", - wantChecks: []string{"nccl-all-reduce-bw-nvls"}, + wantChecks: []string{"nccl-all-reduce-bw-net", "nccl-all-reduce-bw-nvls"}, wantConstraints: map[string]string{ + "nccl-all-reduce-bw-net": ">= 40", "nccl-all-reduce-bw-nvls": ">= 500", }, }, { name: "gb200-oke-ubuntu-training", - wantChecks: []string{"nccl-all-reduce-bw-nvls"}, + wantChecks: []string{"nccl-all-reduce-bw-net", "nccl-all-reduce-bw-nvls"}, wantConstraints: map[string]string{ + "nccl-all-reduce-bw-net": ">= 40", "nccl-all-reduce-bw-nvls": ">= 500", }, }, { name: "gb200-oke-ubuntu-training-kubeflow", - wantChecks: []string{"nccl-all-reduce-bw-nvls"}, + wantChecks: []string{"nccl-all-reduce-bw-net", "nccl-all-reduce-bw-nvls"}, wantConstraints: map[string]string{ + "nccl-all-reduce-bw-net": ">= 40", "nccl-all-reduce-bw-nvls": ">= 500", }, }, diff --git a/recipes/components/network-operator/manifests/nic-cluster-policy-oke-gb200.yaml b/recipes/components/network-operator/manifests/nic-cluster-policy-oke-gb200.yaml new file mode 100644 index 0000000000..a7c344376f --- /dev/null +++ b/recipes/components/network-operator/manifests/nic-cluster-policy-oke-gb200.yaml @@ -0,0 +1,62 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# NicClusterPolicy for GB200 OKE (OCI) — rdmaSharedDevicePlugin over InfiniBand. +# +# Mirrors the AOR OCI GB200 config validated on gb200-ew. No ofedDriver (host +# MOFED), no SR-IOV: the NVL72 east-west fabric is IB on the rdma0-3 netdevs +# (oci_hpc.rdma_device_names_mode=2 kernel cmdline names them deterministically). +# +# The IB devices are advertised as nvidia.com/mlnxnics — the same resource +# name the L40S SR-IOV path uses, so workloads request RDMA uniformly +# across OKE fabrics. +apiVersion: mellanox.com/v1alpha1 +kind: NicClusterPolicy +metadata: + name: nic-cluster-policy + annotations: + helm.sh/hook: post-install,post-upgrade + helm.sh/hook-weight: "5" + helm.sh/hook-delete-policy: before-hook-creation + labels: + app.kubernetes.io/managed-by: {{ .Release.Service }} + helm.sh/chart: {{ printf "%s-%s" .Chart.Name .Chart.Version | replace "+" "_" | trunc 63 | trimSuffix "-" }} +spec: + rdmaSharedDevicePlugin: + image: k8s-rdma-shared-dev-plugin + repository: nvcr.io/nvidia/mellanox + version: network-operator-v26.4.1 + config: | + { + "configList": [ + { + "resourcePrefix": "nvidia.com", + "resourceName": "mlnxnics", + "rdmaHcaMax": 63, + "selectors": { + "linkTypes": ["infiniband"], + "ifNames": ["rdma0", "rdma1", "rdma2", "rdma3"] + } + } + ] + } + deploymentTolerations: + - key: CriticalAddonsOnly + operator: Exists + tolerations: + # RDMA DaemonSets must land on tainted GPU nodes. + - key: nvidia.com/gpu + operator: Exists + - key: CriticalAddonsOnly + operator: Exists diff --git a/recipes/components/network-operator/manifests/nic-cluster-policy-oke-l40s.yaml b/recipes/components/network-operator/manifests/nic-cluster-policy-oke-l40s.yaml new file mode 100644 index 0000000000..dd6d39c6a2 --- /dev/null +++ b/recipes/components/network-operator/manifests/nic-cluster-policy-oke-l40s.yaml @@ -0,0 +1,87 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# NicClusterPolicy for L40S OKE (OCI) SR-IOV RoCE. +# +# The network-operator Helm chart installs the operator + CRD but does not template +# a NicClusterPolicy CR (values-oke-l40s.yaml sets deployCR: false). This manifest +# creates it so the operator reconciles the RoCE fabric stack. Hand-rendered from +# AOR's network-operator/nicclusterpolicy.yaml.tmpl (provider: oci branch, with +# network.type == roce → nvIpam + secondaryNetwork included). +# +# OCI specifics (vs Forge IB): NO ofedDriver — OCI nodes carry host MOFED, consumed +# by the GPU Operator driver via driver.rdma.useHostMofed (l40s-oke-ubuntu leaf). One +# sriovDevicePlugin resource, nvidia.com/mlnxnics, selecting the OCI ConnectX VF +# device IDs (101a = ConnectX-5 Ex VF, 101e = mlx5Gen VF). RoCE also needs nv-ipam +# (VF IP allocation) + secondaryNetwork/multus (attach the VF into workload pods). +# vendor 15b3 = Mellanox. +apiVersion: mellanox.com/v1alpha1 +kind: NicClusterPolicy +metadata: + name: nic-cluster-policy + annotations: + helm.sh/hook: post-install,post-upgrade + helm.sh/hook-weight: "5" + helm.sh/hook-delete-policy: before-hook-creation + labels: + app.kubernetes.io/managed-by: {{ .Release.Service }} + helm.sh/chart: {{ printf "%s-%s" .Chart.Name .Chart.Version | replace "+" "_" | trunc 63 | trimSuffix "-" }} +spec: + # RoCE: allocate IPs for the RDMA VFs and wire them into pods via multus. + nvIpam: + image: nvidia-k8s-ipam + repository: ghcr.io/mellanox + version: v0.2.0 + enableWebhook: false + containerResources: + - name: nv-ipam-node + requests: + cpu: 500m + memory: 1Gi + limits: + cpu: "1" + memory: 2Gi + secondaryNetwork: + cniPlugins: + image: plugins + repository: ghcr.io/k8snetworkplumbingwg + version: v1.6.2-update.1 + multus: + image: multus-cni + repository: ghcr.io/k8snetworkplumbingwg + version: v4.2.1 + sriovDevicePlugin: + image: sriov-network-device-plugin + repository: ghcr.io/k8snetworkplumbingwg + version: v3.9.0 + config: | + { + "resourceList": [ + { + "resourcePrefix": "nvidia.com", + "resourceName": "mlnxnics", + "selectors": {"isRdma":true,"vendors":["15b3"],"devices":["101a","101e"]} + } + ] + } + # Operator DaemonSet placement: system/monitoring nodes only (matches AOR). + deploymentTolerations: + - key: CriticalAddonsOnly + operator: Exists + tolerations: + # RDMA DaemonSets must land on tainted GPU nodes. + - key: nvidia.com/gpu + operator: Exists + - key: CriticalAddonsOnly + operator: Exists diff --git a/recipes/components/network-operator/values-oke-gb200.yaml b/recipes/components/network-operator/values-oke-gb200.yaml new file mode 100644 index 0000000000..752d60dc1f --- /dev/null +++ b/recipes/components/network-operator/values-oke-gb200.yaml @@ -0,0 +1,39 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# network-operator Helm values for GB200 OKE (OCI) InfiniBand. +# +# OCI GB200 NVL72 model (vs L40S RoCE / Forge IB): NO ofedDriver — nodes carry host +# MOFED — and no SR-IOV/nv-ipam/multus either. East-west is InfiniBand (rdma0-3), +# served by rdmaSharedDevicePlugin from the post-install NicClusterPolicy manifest, +# NOT the chart. deployCR off so that manifest CR is authoritative. +# nfd.enabled: false — GPU Operator's NFD is used; no second NFD. +deployCR: false +nvIpam: + enabled: false +secondaryNetwork: + deploy: false +nfd: + enabled: false +# Operator placement comes from the bundler's system-node scheduling +# injection (registry nodeScheduling: operator.nodeSelector / +# operator.tolerations) — no hardcoded affinity here. +operator: + resources: + limits: + cpu: "1" + memory: 2Gi + requests: + cpu: 500m + memory: 2Gi diff --git a/recipes/components/network-operator/values-oke-l40s.yaml b/recipes/components/network-operator/values-oke-l40s.yaml new file mode 100644 index 0000000000..6f0b4b2a60 --- /dev/null +++ b/recipes/components/network-operator/values-oke-l40s.yaml @@ -0,0 +1,48 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# network-operator Helm values for L40S OKE (OCI) SR-IOV RoCE. +# Hand-rendered from AOR's network-operator/values.yaml.tmpl (provider: oci) + +# nicclusterpolicy.yaml.tmpl (oci branch, network.type == roce). +# +# OCI model (vs Forge IB / Mistral DOCA): NO ofedDriver — OCI bare-metal nodes carry +# host MOFED, so the GPU Operator uses it via driver.rdma.useHostMofed (set on the +# l40s-oke-ubuntu leaf). network-operator's job here is the SR-IOV VF device plugin +# (advertises nvidia.com/mlnxnics RDMA VFs) plus nv-ipam + secondaryNetwork (multus) +# for RoCE — all supplied by the post-install NicClusterPolicy manifest, NOT the chart. +# +# deployCR/nvIpam/secondaryNetwork: AICR's wrapper defaults are on (deployCR: true, +# nvIpam.enabled: true, secondaryNetwork.deploy: true) — they template the wrapper's +# own NicClusterPolicy. Turn deployCR off so our manifest CR is authoritative (it is +# the only place the OCI VF selectors 101a/101e can be expressed); the operator +# reconciles nv-ipam + secondaryNetwork + sriovDevicePlugin from that CR regardless. +# nfd.enabled: false — GPU Operator's NFD is used; no second NFD. +deployCR: false +nvIpam: + enabled: false +secondaryNetwork: + deploy: false +nfd: + enabled: false +# Operator placement comes from the bundler's system-node scheduling +# injection (registry nodeScheduling: operator.nodeSelector / +# operator.tolerations) — no hardcoded affinity here. +operator: + resources: + limits: + cpu: "1" + memory: 2Gi + requests: + cpu: 500m + memory: 2Gi diff --git a/recipes/manifest_images_test.go b/recipes/manifest_images_test.go index 23461a37ec..c5c44a11e0 100644 --- a/recipes/manifest_images_test.go +++ b/recipes/manifest_images_test.go @@ -87,25 +87,43 @@ func TestComponentManifestImagesAreFullyQualified(t *testing.T) { // these refs is delivered by admission-time digest or signature // verification at deploy time (#745) plus the upstream signing requests // filed under the supply-chain epic (#739). -var imageDigestExemptions = map[string]string{ +// imageDigestExemption scopes an exemption to the manifest that carries +// the reference: the exemption applies only when the walked path contains +// Manifest, so another resource reusing the same tag elsewhere cannot ride +// an existing exemption past the digest check. +type imageDigestExemption struct { + // Manifest is a path substring under components/ that must appear in + // the manifest path for the exemption to apply. + Manifest string + Reason string +} + +var imageDigestExemptions = map[string]imageDigestExemption{ // NicClusterPolicy (network-operator AKS): repository/image/version // triplet schema; no digest field. - "nvcr.io/nvidia/mellanox/doca-driver:doca3.2.0-25.10-1.2.8.0-2": "NicClusterPolicy CRD does not accept image digests; tracked via #745 and Mellanox/network-operator#2555", - "nvcr.io/nvidia/mellanox/k8s-rdma-shared-dev-plugin:network-operator-v26.4.1": "NicClusterPolicy CRD does not accept image digests; tracked via #745 and Mellanox/network-operator#2555", - "nvcr.io/nvidia/doca/doca_telemetry:1.22.5-doca3.1.0-host": "NicClusterPolicy CRD does not accept image digests; tracked via #745 and Mellanox/network-operator#2555", + "nvcr.io/nvidia/mellanox/doca-driver:doca3.2.0-25.10-1.2.8.0-2": {"network-operator/manifests/nic-cluster-policy-aks", "NicClusterPolicy CRD does not accept image digests; tracked via #745 and Mellanox/network-operator#2555"}, + "nvcr.io/nvidia/mellanox/k8s-rdma-shared-dev-plugin:network-operator-v26.4.1": {"network-operator/manifests/nic-cluster-policy-", "NicClusterPolicy CRD does not accept image digests; tracked via #745 and Mellanox/network-operator#2555"}, + "nvcr.io/nvidia/doca/doca_telemetry:1.22.5-doca3.1.0-host": {"network-operator/manifests/nic-cluster-policy-aks", "NicClusterPolicy CRD does not accept image digests; tracked via #745 and Mellanox/network-operator#2555"}, // Skyhook Package (nodewright-customizations no-op): shellscript package // is pinned by tag only — `containerSHA` is not surfaced for this // upstream image. nodewright-packages/* refs are digest-pinned via the // Skyhook Package `containerSHA` field (issue #1031), folded into the // extracted image ref as `@sha256:...` by pkg/bom.ExtractImagesFromYAML. - "ghcr.io/nvidia/skyhook-packages/shellscript:1.1.1": "Skyhook Package CRD does not accept image digests; tracked via #745 and NVIDIA/nodewright#224", + "ghcr.io/nvidia/skyhook-packages/shellscript:1.1.1": {"nodewright-customizations/manifests/", "Skyhook Package CRD does not accept image digests; tracked via #745 and NVIDIA/nodewright#224"}, // cos-gpu-installer (gcp-driver-installer): node-local image preloaded on // every COS node and referenced with imagePullPolicy: Never — there is no // registry to pin a digest against, the "digest" differs per COS build, // and the ref must never be mirrored or pulled. Issue #1716. - "cos-nvidia-installer:fixed": "COS-node-local preloaded image (imagePullPolicy: Never); no registry digest exists and it must not be mirrored; issue #1716", + "cos-nvidia-installer:fixed": {"gcp-driver-installer/manifests/", "COS-node-local preloaded image (imagePullPolicy: Never); no registry digest exists and it must not be mirrored; issue #1716"}, + + // NicClusterPolicy (network-operator OKE): same repository/image/version + // triplet schema as the AKS entries above — no digest field in the CRD. + "ghcr.io/mellanox/nvidia-k8s-ipam:v0.2.0": {"network-operator/manifests/nic-cluster-policy-oke-", "NicClusterPolicy CRD does not accept image digests; tracked via #745 and Mellanox/network-operator#2555"}, + "ghcr.io/k8snetworkplumbingwg/multus-cni:v4.2.1": {"network-operator/manifests/nic-cluster-policy-oke-", "NicClusterPolicy CRD does not accept image digests; tracked via #745 and Mellanox/network-operator#2555"}, + "ghcr.io/k8snetworkplumbingwg/plugins:v1.6.2-update.1": {"network-operator/manifests/nic-cluster-policy-oke-", "NicClusterPolicy CRD does not accept image digests; tracked via #745 and Mellanox/network-operator#2555"}, + "ghcr.io/k8snetworkplumbingwg/sriov-network-device-plugin:v3.9.0": {"network-operator/manifests/nic-cluster-policy-oke-", "NicClusterPolicy CRD does not accept image digests; tracked via #745 and Mellanox/network-operator#2555"}, } // TestComponentManifestImagesAreDigestPinned asserts that every image @@ -151,8 +169,8 @@ func TestComponentManifestImagesAreDigestPinned(t *testing.T) { t.Errorf("%s: image %q uses non-sha256 digest %q; ADR-006 requires @sha256:", p, img, ref.Digest) continue } - if reason, ok := imageDigestExemptions[img]; ok { - t.Logf("exempted: %s — %s", img, reason) + if exemption, ok := imageDigestExemptions[img]; ok && strings.Contains(p, exemption.Manifest) { + t.Logf("exempted: %s (%s) — %s", img, exemption.Manifest, exemption.Reason) continue } t.Errorf("%s: image %q is not digest-pinned and not in the documented exemption set (recipes/manifest_images_test.go::imageDigestExemptions); per ADR-006 layer 2, append an @sha256: or add an exemption with a reason", p, img) diff --git a/recipes/overlays/gb200-oke-training.yaml b/recipes/overlays/gb200-oke-training.yaml index d9543c6b7e..33e20f2d0c 100644 --- a/recipes/overlays/gb200-oke-training.yaml +++ b/recipes/overlays/gb200-oke-training.yaml @@ -39,9 +39,28 @@ spec: value: ">= 1.34" componentRefs: - # GB200-specific GPU Operator overrides (inherits valuesFile from oke-training) + # GB200-specific GPU Operator overrides (inherits valuesFile from oke-training). + # kernel-module-params sets NVreg_GrdmaPciTopoCheckOverride=1, required + # for dma-buf attach over the IB fabric (GPUDirect RDMA without + # nvidia-peermem, whose chroot modprobe fails to build against the -64k + # Grace kernel). + # + # driver.kernelModuleConfig is consumed ONLY when a GPU Operator driver + # DaemonSet runs, i.e. under gpuStack=operator-managed (#2355); under the + # default oci-managed profile (driver.enabled: false from values-oke.yaml, + # driver ships in the node image) this value is rendered but inert, and + # the image or node bootstrap must set the module parameter instead. It + # cannot live in the profile's operator-managed fragment: fragments are + # family-wide (recipes/overlays/oke-ol.yaml), and only this leaf ships the + # nvidia-kernel-module-params ConfigMap — a family-wide reference would + # dangle on every other OKE recipe. The NCCL -net preflight + # (validators/performance/nccl_preflight_nvreg.go) fails closed on either + # profile when the loaded driver is missing the flag, so an oci-managed + # image driver without it is caught before NCCL degrades to Socket. - name: gpu-operator type: Helm + preManifestFiles: + - components/gpu-operator/manifests/kernel-module-params.yaml dependencyRefs: - nfd - cert-manager @@ -49,6 +68,9 @@ spec: overrides: gdrcopy: enabled: true + driver: + kernelModuleConfig: + name: nvidia-kernel-module-params - name: nfd type: Helm @@ -56,13 +78,33 @@ spec: topologyUpdater: enable: true + # InfiniBand east-west fabric (NVL72 rdma0-3). rdmaSharedDevicePlugin + # advertises the shared HCAs as nvidia.com/mlnxnics; no SR-IOV/nv-ipam + # (that is the L40S RoCE path) and no ofedDriver (OCI nodes carry host + # MOFED). NicClusterPolicy is manifest-supplied (chart deployCR off). + # Present in every gpuStack value; incompatible with Oracle's opt-in + # NvidiaNetworkOperator add-on. + - name: network-operator + type: Helm + valuesFile: components/network-operator/values-oke-gb200.yaml + manifestFiles: + - components/network-operator/manifests/nic-cluster-policy-oke-gb200.yaml + dependencyRefs: + - nfd + - cert-manager + validation: performance: - # NVLS runtime support is OKE-specific. NET/RDMA is intentionally left - # out until OCI-specific pod RDMA exposure is verified on the testbed. + # Both transport variants: NVLS (MNNVL across the NVL72 IMEX domain) + # and NET (the IB east-west fabric this leaf's NicClusterPolicy + # exposes — validated on a BM.GPU.GB200.4 NVL72 rack). Constraints + # match gb200-eks-training. checks: + - nccl-all-reduce-bw-net - nccl-all-reduce-bw-nvls constraints: + - name: nccl-all-reduce-bw-net + value: ">= 40" - name: nccl-all-reduce-bw-nvls value: ">= 500" conformance: diff --git a/recipes/overlays/l40s-oke-training.yaml b/recipes/overlays/l40s-oke-training.yaml index 957e43b5a9..897e35193c 100644 --- a/recipes/overlays/l40s-oke-training.yaml +++ b/recipes/overlays/l40s-oke-training.yaml @@ -54,6 +54,26 @@ spec: topologyUpdater: enable: true + # RDMA fabric (RoCE over SR-IOV VFs). Every L40S OCI cluster runs RoCE; + # the SR-IOV VF device plugin advertises nvidia.com/mlnxnics RDMA VFs, + # with nv-ipam + multus attaching the VFs into workload pods. The + # NicClusterPolicy is supplied by the manifest (the chart's deployCR is + # off — the manifest is the only place the OCI VF selectors 101a/101e + # can be expressed). OCI nodes carry host MOFED, so there is no + # ofedDriver in any configuration. GPUDirect RDMA works via DMA-BUF; + # nvidia-peermem stays off (base default) — inert on this topology. + # Present in every gpuStack value: the fabric is orthogonal to GPU + # driver/plugin ownership. Incompatible with Oracle's opt-in + # NvidiaNetworkOperator add-on (two lifecycle managers, one release). + - name: network-operator + type: Helm + valuesFile: components/network-operator/values-oke-l40s.yaml + manifestFiles: + - components/network-operator/manifests/nic-cluster-policy-oke-l40s.yaml + dependencyRefs: + - nfd + - cert-manager + # Validation checks for L40S on OKE training workloads. # Defined at the intent layer (not OS-specific) so all OS variants inherit them. # diff --git a/validators/deployment/expected_resources.go b/validators/deployment/expected_resources.go index 7d34cf1ae0..5c3f9d386a 100644 --- a/validators/deployment/expected_resources.go +++ b/validators/deployment/expected_resources.go @@ -87,17 +87,19 @@ const ( runtimeRequiredTaintKey = "skyhook.nvidia.com" runtimeRequiredTaintValue = "runtime-required" - // nicClusterPolicyManifestMarker identifies the AKS NicClusterPolicy manifest - // (recipes/components/network-operator/manifests/nic-cluster-policy-aks.yaml) - // among a network-operator ComponentRef's ManifestFiles. Its presence means - // the recipe stands up an RDMA fabric (MOFED + rdma-shared-device-plugin), so - // a GPU node not yet advertising the shared resource is "still converging", - // not "no fabric". OCP wires a different component (network-operator-ocp) and - // manifest name and so does not match; kind/talos enable network-operator - // without this manifest and are likewise (correctly) not gated. The shared - // resource name and the RDMA node label are defined in validators/helper - // (AKSRdmaSharedResource, PCIMellanoxPresentLabel) so this gate and the NCCL - // consumer cannot drift. + // nicClusterPolicyManifestMarker identifies a NicClusterPolicy manifest + // (nic-cluster-policy-aks.yaml, nic-cluster-policy-oke-{gb200,l40s}.yaml + // under recipes/components/network-operator/manifests/) among a + // network-operator ComponentRef's ManifestFiles. Its presence means the + // recipe stands up an RDMA fabric (an RDMA-shared or SR-IOV device plugin), + // so a GPU node not yet advertising the fabric resource is "still + // converging", not "no fabric". OCP wires a different component + // (network-operator-ocp) and manifest name and so does not match; kind/talos + // enable network-operator without this manifest and are likewise (correctly) + // not gated. The polled resource name is derived per recipe from the matched + // manifest itself (rdmaFabricResource), so the gate always waits for exactly + // what this recipe's fabric advertises; the RDMA node label lives in + // validators/helper (PCIMellanoxPresentLabel). nicClusterPolicyManifestMarker = "nic-cluster-policy" ) @@ -433,7 +435,15 @@ func verifyGPUReadinessSignals(ctx *validators.Context, refs []recipe.ComponentR } if ref, ok := findEnabledComponent(refs, networkOperatorComponent); ok && recipeDeclaresRDMAFabric(ref) { - capture(verifyRDMAFabricReady(ctx)) + // The polled resource is derived from the recipe's own NicClusterPolicy + // manifest (rdma/hca_shared_devices_a on AKS, nvidia.com/mlnxnics on OKE) + // so the gate waits for exactly what this recipe's fabric advertises. A + // derivation failure fails the gate closed — never "skip the fabric". + if fabricResource, ferr := rdmaFabricResource(ctx.Ctx, ref); ferr != nil { + capture(ferr) + } else { + capture(verifyRDMAFabricReady(ctx, fabricResource)) + } } return failures, firstStructured @@ -1043,12 +1053,13 @@ func draKubeletPluginProbe(ctx *validators.Context, namespace string) (string, e } // recipeDeclaresRDMAFabric reports whether a network-operator ComponentRef -// stands up an RDMA fabric on this cluster — i.e. it declares the AKS -// NicClusterPolicy manifest that creates MOFED + the rdma-shared-device-plugin +// stands up an RDMA fabric on this cluster — i.e. it declares a +// NicClusterPolicy manifest that creates an RDMA device plugin // (nicClusterPolicyManifestMarker). When true, a GPU node that does not yet -// advertise aksRDMASharedResource is "still converging", so verifyRDMAFabricReady -// waits for it; when false (kind's single-node nvkind, talos' namespace-only -// ref) there is no shared fabric to gate on and the check is skipped. +// advertise the manifest-derived fabric resource is "still converging", so +// verifyRDMAFabricReady waits for it; when false (kind's single-node nvkind, +// talos' namespace-only ref) there is no shared fabric to gate on and the +// check is skipped. // // Sibling predicate: pkg/bundler/readiness.go's // recipeAttachesNicClusterPolicy encodes the same "does this recipe stand @@ -1070,7 +1081,8 @@ func recipeDeclaresRDMAFabric(ref recipe.ComponentRef) bool { } // verifyRDMAFabricReady blocks the deployment gate until the network operator's -// shared RDMA device (helper.AKSRdmaSharedResource) is allocatable in a uniform, +// shared RDMA device (fabricResource, derived from the recipe's own +// NicClusterPolicy manifest by rdmaFabricResource) is allocatable in a uniform, // positive count across every Mellanox RDMA-capable GPU node, held continuously for the // stability window. // @@ -1098,19 +1110,19 @@ func recipeDeclaresRDMAFabric(ref recipe.ComponentRef) bool { // so it survives the default redaction policy into the signed bundle (#1951/#1952) — // a cordoned node narrowing the fabric cohort can no longer hide behind a // stdout-only line the publisher strips. -func verifyRDMAFabricReady(ctx *validators.Context) error { +func verifyRDMAFabricReady(ctx *validators.Context, fabricResource string) error { // Production emit seam: publish the structured coverage as an EmitExtra // sentinel. verifyRDMAFabricReadyEmit injects it so tests can record the eager // floor and terminal disclosures without capturing the EmitExtra stdout // transport (which lives in the validators package). - return verifyRDMAFabricReadyEmit(ctx, func(validated, total int) { + return verifyRDMAFabricReadyEmit(ctx, fabricResource, func(validated, total int) { emitExtraOrWarn(rdmaFabricCoverageExtra(validated, total)) }) } // verifyRDMAFabricReadyEmit is verifyRDMAFabricReady with the structured // coverage emit injected. See verifyRDMAFabricReady for the gate contract. -func verifyRDMAFabricReadyEmit(ctx *validators.Context, emitCoverage func(validated, total int)) error { +func verifyRDMAFabricReadyEmit(ctx *validators.Context, fabricResource string, emitCoverage func(validated, total int)) error { var coverage rdmaFabricCoverage // emittedEarly gates the eager disclosure floor to exactly one emit. var emittedEarly bool @@ -1120,9 +1132,9 @@ func verifyRDMAFabricReadyEmit(ctx *validators.Context, emitCoverage func(valida // each tick — the settled disclosure must land exactly once, at the final // outcome). err := pollUntilStable(ctx, - fmt.Sprintf("RDMA shared-device fabric (%s) across RDMA GPU nodes", helper.AKSRdmaSharedResource), + fmt.Sprintf("RDMA shared-device fabric (%s) across RDMA GPU nodes", fabricResource), func() error { - cov, probeErr := rdmaFabricProbeCoverage(ctx) + cov, probeErr := rdmaFabricProbeCoverage(ctx, fabricResource) coverage = cov // Eager disclosure floor: emit the structured coverage once, on the // first observation that actually enumerated an RDMA-candidate node, @@ -1172,7 +1184,7 @@ func verifyRDMAFabricReadyEmit(ctx *validators.Context, emitCoverage func(valida if err == nil { fmt.Printf(" RDMA fabric (%s): allocatable (uniform) on all %d schedulable RDMA GPU node(s) (stable ≥%s)\n", - helper.AKSRdmaSharedResource, coverage.schedulable, gpuReadinessStabilityWindow) + fabricResource, coverage.schedulable, gpuReadinessStabilityWindow) } return err } @@ -1243,13 +1255,13 @@ func rdmaFabricCoverageExtra(validated, total int) map[string]string { // then validates only the schedulable cohort: nodes carrying the NicClusterPolicy // nodeAffinity label helper.PCIMellanoxPresentLabel. It returns nil — plus the // coverage partition — only when every schedulable such node advertises -// helper.AKSRdmaSharedResource in a uniform, positive count. It fails closed on a +// fabricResource in a uniform, positive count. It fails closed on a // List error and when no schedulable RDMA GPU node is observed yet: "could not // observe the fabric" must never read as "fabric ready". The returned error rides // the poll's dwell reset like any other unhealthy sample; the coverage is // returned alongside every error so the terminal disclosure can still name the // cordoned nodes it saw. -func rdmaFabricProbeCoverage(ctx *validators.Context) (rdmaFabricCoverage, error) { +func rdmaFabricProbeCoverage(ctx *validators.Context, fabricResource string) (rdmaFabricCoverage, error) { listCtx, cancel := ctx.Timeout(defaults.ResourceVerificationTimeout) defer cancel() @@ -1263,7 +1275,7 @@ func rdmaFabricProbeCoverage(ctx *validators.Context) (rdmaFabricCoverage, error "failed to list nodes for the RDMA fabric readiness gate") } - fabric := corev1.ResourceName(helper.AKSRdmaSharedResource) + fabric := corev1.ResourceName(fabricResource) type rdmaNode struct { name string count int64 @@ -1323,8 +1335,8 @@ func rdmaFabricProbeCoverage(ctx *validators.Context) (rdmaFabricCoverage, error if len(notReady) > 0 { return coverage, errors.New(errors.ErrCodeInternal, fmt.Sprintf("%s not yet allocatable on %d of %d RDMA GPU node(s): %s "+ - "(network operator MOFED / rdma-shared-device-plugin still rolling out)", - helper.AKSRdmaSharedResource, len(notReady), len(cohort), formatNames(notReady))) + "(network operator MOFED / RDMA device plugin still rolling out)", + fabricResource, len(notReady), len(cohort), formatNames(notReady))) } // All present and positive: require a uniform count, matching the NCCL @@ -1340,7 +1352,7 @@ func rdmaFabricProbeCoverage(ctx *validators.Context) (rdmaFabricCoverage, error if len(skew) > 0 { return coverage, errors.New(errors.ErrCodeInternal, fmt.Sprintf("%s allocatable count is non-uniform across %d RDMA GPU node(s) (want all == %d): %s", - helper.AKSRdmaSharedResource, len(cohort), want, formatNames(skew))) + fabricResource, len(cohort), want, formatNames(skew))) } return coverage, nil } diff --git a/validators/deployment/expected_resources_rdma_test.go b/validators/deployment/expected_resources_rdma_test.go index 7009c43b39..c95e502402 100644 --- a/validators/deployment/expected_resources_rdma_test.go +++ b/validators/deployment/expected_resources_rdma_test.go @@ -183,7 +183,7 @@ func TestVerifyRDMAFabricReady_Poll(t *testing.T) { }) ctx := &validators.Context{Ctx: context.Background(), Clientset: clientset} - err := verifyRDMAFabricReady(ctx) + err := verifyRDMAFabricReady(ctx, helper.AKSRdmaSharedResource) if tt.wantErrSub != "" { if err == nil { t.Fatalf("verifyRDMAFabricReady() error = nil, want error containing %q", tt.wantErrSub) @@ -255,7 +255,7 @@ func TestVerifyRDMAFabricReady_EagerDisclosureFloor(t *testing.T) { // pollUntilStable runs the probe synchronously in this goroutine, so // the injected emit is never called concurrently — no lock needed. var calls []emitCall - err := verifyRDMAFabricReadyEmit(ctx, func(validated, total int) { + err := verifyRDMAFabricReadyEmit(ctx, helper.AKSRdmaSharedResource, func(validated, total int) { calls = append(calls, emitCall{validated, total}) }) @@ -290,7 +290,7 @@ func TestRDMAFabricProbe_FailsClosedOnListError(t *testing.T) { }) ctx := &validators.Context{Ctx: context.Background(), Clientset: clientset} - cov, err := rdmaFabricProbeCoverage(ctx) + cov, err := rdmaFabricProbeCoverage(ctx, helper.AKSRdmaSharedResource) if err == nil { t.Fatal("expected an error when listing nodes fails, got nil (must fail closed)") } @@ -323,7 +323,7 @@ func TestRDMAFabricProbe_FailsClosedWithoutRDMANodes(t *testing.T) { ) ctx := &validators.Context{Ctx: context.Background(), Clientset: clientset} - cov, err := rdmaFabricProbeCoverage(ctx) + cov, err := rdmaFabricProbeCoverage(ctx, helper.AKSRdmaSharedResource) if err == nil { t.Fatal("expected an error when no RDMA GPU nodes are present, got nil (must fail closed)") } @@ -352,7 +352,7 @@ func TestRDMAFabricProbe_ExcludesCordonedNonRDMAAndCPU(t *testing.T) { ) ctx := &validators.Context{Ctx: context.Background(), Clientset: clientset} - cov, err := rdmaFabricProbeCoverage(ctx) + cov, err := rdmaFabricProbeCoverage(ctx, helper.AKSRdmaSharedResource) if err != nil { t.Fatalf("rdmaFabricProbeCoverage() error = %v, want nil (only the schedulable RDMA GPU node is required to carry the fabric)", err) } @@ -373,7 +373,7 @@ func TestRDMAFabricProbe_NonUniformCountFails(t *testing.T) { ) ctx := &validators.Context{Ctx: context.Background(), Clientset: clientset} - cov, err := rdmaFabricProbeCoverage(ctx) + cov, err := rdmaFabricProbeCoverage(ctx, helper.AKSRdmaSharedResource) if err == nil { t.Fatal("expected an error on non-uniform fabric counts, got nil") } @@ -396,7 +396,7 @@ func TestRDMAFabricProbe_PassesWhenUniform(t *testing.T) { ) ctx := &validators.Context{Ctx: context.Background(), Clientset: clientset} - cov, err := rdmaFabricProbeCoverage(ctx) + cov, err := rdmaFabricProbeCoverage(ctx, helper.AKSRdmaSharedResource) if err != nil { t.Fatalf("rdmaFabricProbeCoverage() error = %v, want nil (fabric uniform on all RDMA GPU nodes)", err) } diff --git a/validators/deployment/expected_resources_test.go b/validators/deployment/expected_resources_test.go index 58d4399918..91ae41aa91 100644 --- a/validators/deployment/expected_resources_test.go +++ b/validators/deployment/expected_resources_test.go @@ -23,6 +23,7 @@ import ( "github.com/NVIDIA/aicr/pkg/recipe" v1 "github.com/NVIDIA/aicr/pkg/validator/v1" "github.com/NVIDIA/aicr/validators" + "github.com/NVIDIA/aicr/validators/helper" appsv1 "k8s.io/api/apps/v1" corev1 "k8s.io/api/core/v1" apierrors "k8s.io/apimachinery/pkg/api/errors" @@ -1566,7 +1567,7 @@ func TestRDMAFabricProbeCoverage_DisclosesCordoned(t *testing.T) { ) ctx := &validators.Context{Ctx: context.Background(), Clientset: clientset} - cov, err := rdmaFabricProbeCoverage(ctx) + cov, err := rdmaFabricProbeCoverage(ctx, helper.AKSRdmaSharedResource) if err != nil { t.Fatalf("rdmaFabricProbeCoverage() error = %v, want nil (the one schedulable RDMA node carries the fabric)", err) } @@ -1595,7 +1596,7 @@ func TestRDMAFabricProbeCoverage_CountsCordonedOnFailClosed(t *testing.T) { ) ctx := &validators.Context{Ctx: context.Background(), Clientset: clientset} - cov, err := rdmaFabricProbeCoverage(ctx) + cov, err := rdmaFabricProbeCoverage(ctx, helper.AKSRdmaSharedResource) if err == nil { t.Fatal("expected a fail-closed error while the fabric is absent, got nil") } diff --git a/validators/deployment/rdma_fabric_resource.go b/validators/deployment/rdma_fabric_resource.go new file mode 100644 index 0000000000..153d8c3311 --- /dev/null +++ b/validators/deployment/rdma_fabric_resource.go @@ -0,0 +1,213 @@ +// Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package main + +import ( + "context" + "encoding/json" + "fmt" + "sort" + "strings" + + "sigs.k8s.io/yaml" + + "github.com/NVIDIA/aicr/pkg/errors" + "github.com/NVIDIA/aicr/pkg/manifest" + "github.com/NVIDIA/aicr/pkg/recipe" +) + +// Upstream device-plugin default resource prefixes, applied when a +// NicClusterPolicy config entry omits resourcePrefix. Pinned here so a +// prefix-less config still derives the exact resource name the plugin will +// advertise: +// - k8s-rdma-shared-dev-plugin defaults to "rdma" +// (github.com/Mellanox/k8s-rdma-shared-dev-plugin, rdmaHcaResourcePrefix). +// - sriov-network-device-plugin defaults to "intel.com" +// (github.com/k8snetworkplumbingwg/sriov-network-device-plugin, +// resourcePrefix). +const ( + rdmaSharedDefaultResourcePrefix = "rdma" + sriovDefaultResourcePrefix = "intel.com" +) + +// nicClusterPolicyDoc is the minimal NicClusterPolicy shape the fabric gate +// needs: which device-plugin block(s) the policy deploys and their embedded +// plugin config. Both configs arrive as JSON strings, verbatim from the +// plugin's own config-file format. +type nicClusterPolicyDoc struct { + Kind string `json:"kind"` + Spec struct { + RdmaSharedDevicePlugin *nicDevicePluginSpec `json:"rdmaSharedDevicePlugin"` + SriovDevicePlugin *nicDevicePluginSpec `json:"sriovDevicePlugin"` + } `json:"spec"` +} + +type nicDevicePluginSpec struct { + Config string `json:"config"` +} + +// nicResourceEntry is the shared shape of one resource declaration inside +// either plugin's JSON config. +type nicResourceEntry struct { + ResourcePrefix string `json:"resourcePrefix"` + ResourceName string `json:"resourceName"` +} + +// rdmaFabricResource derives the extended-resource name the recipe's +// NicClusterPolicy actually advertises, by rendering the ComponentRef's +// NicClusterPolicy manifest(s) with the component's effective values and +// reading the device-plugin config embedded in the policy spec +// (rdmaSharedDevicePlugin on AKS "rdma/hca_shared_devices_a" and OKE GB200 +// "nvidia.com/mlnxnics"; sriovDevicePlugin on OKE L40S "nvidia.com/mlnxnics"). +// +// Deriving the resource from the manifest — instead of hardcoding a +// per-provider constant — keeps verifyRDMAFabricReady polling for exactly what +// the recipe's own fabric will publish, so a new provider or an edited +// resourceName can never reintroduce the poll-forever mismatch (#2356 review). +// TestRDMAFabricResource_AKSPin pins the AKS manifest's parse to +// helper.AKSRdmaSharedResource so this parser and the NCCL consumer's constant +// cannot drift. +// +// Fail-closed contract: any read, render, or parse failure is an error, and so +// are a marker-matched manifest set that declares no device-plugin resource or +// one that declares more than one distinct resource (the gate polls a single +// uniform resource across the cohort). "Could not derive the fabric resource" +// must never degrade into "skip the fabric gate". +func rdmaFabricResource(goCtx context.Context, ref recipe.ComponentRef) (string, error) { + values, err := recipe.GetComponentValuesWithContext(goCtx, nil, &ref) + if err != nil { + return "", errors.Wrap(errors.ErrCodeInternal, + fmt.Sprintf("failed to resolve effective values for component %s", ref.Name), err) + } + chartName := ref.Chart + if chartName == "" { + chartName = ref.Name + } + renderInput := manifest.RenderInput{ + ComponentName: ref.Name, + Namespace: ref.Namespace, + ChartName: chartName, + ChartVersion: ref.Version, + Values: values, + } + + resources := map[string]struct{}{} + for _, path := range ref.ManifestFiles { + if !strings.Contains(path, nicClusterPolicyManifestMarker) { + continue + } + select { + case <-goCtx.Done(): + return "", errors.Wrap(errors.ErrCodeTimeout, + "deployment validation canceled while deriving the RDMA fabric resource", goCtx.Err()) + default: + } + content, err := recipe.GetManifestContentWithContext(goCtx, nil, path) + if err != nil { + return "", errors.Wrap(errors.ErrCodeInternal, + fmt.Sprintf("failed to load NicClusterPolicy manifest %s for component %s", path, ref.Name), err) + } + rendered, rerr := manifest.Render(content, renderInput) + if rerr != nil { + return "", errors.Wrap(errors.ErrCodeInternal, + fmt.Sprintf("failed to render NicClusterPolicy manifest %s for component %s", path, ref.Name), rerr) + } + found, perr := nicClusterPolicyResources(rendered) + if perr != nil { + return "", errors.Wrap(errors.ErrCodeInternal, + fmt.Sprintf("failed to parse NicClusterPolicy manifest %s for component %s", path, ref.Name), perr) + } + for _, r := range found { + resources[r] = struct{}{} + } + } + + switch len(resources) { + case 1: + for r := range resources { + return r, nil + } + panic("unreachable") + case 0: + return "", errors.New(errors.ErrCodeInternal, + fmt.Sprintf("component %s declares a NicClusterPolicy manifest but no device-plugin "+ + "resource could be derived from it; the RDMA fabric readiness gate cannot run", ref.Name)) + default: + names := make([]string, 0, len(resources)) + for r := range resources { + names = append(names, r) + } + sort.Strings(names) + return "", errors.New(errors.ErrCodeInternal, + fmt.Sprintf("component %s declares multiple distinct RDMA fabric resources %v; "+ + "the readiness gate polls a single uniform resource and cannot arbitrate", ref.Name, names)) + } +} + +// nicClusterPolicyResources extracts every "/" extended resource +// declared by the NicClusterPolicy document(s) in rendered manifest content. +// Non-NicClusterPolicy documents are skipped; a document that fails to decode +// is an error (fail closed — a malformed policy must not read as "no fabric"). +func nicClusterPolicyResources(rendered []byte) ([]string, error) { + var out []string + for _, doc := range strings.Split(string(rendered), "\n---") { + if strings.TrimSpace(doc) == "" { + continue + } + var ncp nicClusterPolicyDoc + if err := yaml.Unmarshal([]byte(doc), &ncp); err != nil { + return nil, errors.Wrap(errors.ErrCodeInternal, "failed to decode manifest document", err) + } + if ncp.Kind != "NicClusterPolicy" { + continue + } + if p := ncp.Spec.RdmaSharedDevicePlugin; p != nil { + var cfg struct { + ConfigList []nicResourceEntry `json:"configList"` + } + if err := json.Unmarshal([]byte(p.Config), &cfg); err != nil { + return nil, errors.Wrap(errors.ErrCodeInternal, + "failed to decode rdmaSharedDevicePlugin config JSON", err) + } + for _, e := range cfg.ConfigList { + out = append(out, qualifiedNICResource(e, rdmaSharedDefaultResourcePrefix)) + } + } + if p := ncp.Spec.SriovDevicePlugin; p != nil { + var cfg struct { + ResourceList []nicResourceEntry `json:"resourceList"` + } + if err := json.Unmarshal([]byte(p.Config), &cfg); err != nil { + return nil, errors.Wrap(errors.ErrCodeInternal, + "failed to decode sriovDevicePlugin config JSON", err) + } + for _, e := range cfg.ResourceList { + out = append(out, qualifiedNICResource(e, sriovDefaultResourcePrefix)) + } + } + } + return out, nil +} + +// qualifiedNICResource joins a config entry into the "/" +// extended-resource form, applying the plugin's documented default prefix +// when the entry omits one. +func qualifiedNICResource(e nicResourceEntry, defaultPrefix string) string { + prefix := e.ResourcePrefix + if prefix == "" { + prefix = defaultPrefix + } + return prefix + "/" + e.ResourceName +} diff --git a/validators/deployment/rdma_fabric_resource_test.go b/validators/deployment/rdma_fabric_resource_test.go new file mode 100644 index 0000000000..4eefc769d3 --- /dev/null +++ b/validators/deployment/rdma_fabric_resource_test.go @@ -0,0 +1,207 @@ +// Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package main + +import ( + "context" + "strings" + "testing" + + "github.com/NVIDIA/aicr/pkg/recipe" + "github.com/NVIDIA/aicr/validators/helper" +) + +// TestRDMAFabricResource_RealManifests derives the fabric resource from the +// actual embedded catalog manifests, so the gate's parser and the shipped +// NicClusterPolicies cannot drift. The AKS case doubles as the pin between +// this parser and helper.AKSRdmaSharedResource (the NCCL consumer's constant). +func TestRDMAFabricResource_RealManifests(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + manifest string + want string + }{ + { + // Pin: manifest-derived == the NCCL consumer's constant. + name: "AKS rdmaSharedDevicePlugin (hca_shared_devices_a)", + manifest: "components/network-operator/manifests/nic-cluster-policy-aks.yaml", + want: helper.AKSRdmaSharedResource, + }, + { + name: "OKE GB200 rdmaSharedDevicePlugin (IB shared HCAs)", + manifest: "components/network-operator/manifests/nic-cluster-policy-oke-gb200.yaml", + want: "nvidia.com/mlnxnics", + }, + { + name: "OKE L40S sriovDevicePlugin (RoCE VFs)", + manifest: "components/network-operator/manifests/nic-cluster-policy-oke-l40s.yaml", + want: "nvidia.com/mlnxnics", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + ref := recipe.ComponentRef{ + Name: networkOperatorComponent, + Namespace: "network-operator", + ManifestFiles: []string{tt.manifest}, + } + got, err := rdmaFabricResource(context.Background(), ref) + if err != nil { + t.Fatalf("rdmaFabricResource(%s) error = %v, want nil", tt.manifest, err) + } + if got != tt.want { + t.Errorf("rdmaFabricResource(%s) = %q, want %q", tt.manifest, got, tt.want) + } + }) + } +} + +// TestRDMAFabricResource_FailsClosed proves derivation failures are errors, +// never a silent skip: a fabric-declaring ref whose policy yields no resource, +// distinct resources across manifests, and an unreadable manifest path all +// fail. +func TestRDMAFabricResource_FailsClosed(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + manifestFiles []string + wantErrSub string + }{ + { + name: "missing manifest path fails closed", + manifestFiles: []string{"components/network-operator/manifests/nic-cluster-policy-nonexistent.yaml"}, + wantErrSub: "failed to load NicClusterPolicy manifest", + }, + { + name: "two manifests with distinct resources fail closed", + manifestFiles: []string{ + "components/network-operator/manifests/nic-cluster-policy-aks.yaml", + "components/network-operator/manifests/nic-cluster-policy-oke-gb200.yaml", + }, + wantErrSub: "multiple distinct RDMA fabric resources", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + ref := recipe.ComponentRef{ + Name: networkOperatorComponent, + Namespace: "network-operator", + ManifestFiles: tt.manifestFiles, + } + _, err := rdmaFabricResource(context.Background(), ref) + if err == nil { + t.Fatalf("rdmaFabricResource() error = nil, want error containing %q", tt.wantErrSub) + } + if !strings.Contains(err.Error(), tt.wantErrSub) { + t.Fatalf("rdmaFabricResource() error = %v, want substring %q", err, tt.wantErrSub) + } + }) + } +} + +// TestNICClusterPolicyResources_ParseShapes exercises the parser on synthetic +// documents: default-prefix fallback for both plugin kinds, non-NCP documents +// skipped, malformed config JSON failing closed, and a policy with neither +// device-plugin block yielding no resources. +func TestNICClusterPolicyResources_ParseShapes(t *testing.T) { + t.Parallel() + + tests := []struct { + name string + rendered string + want []string + wantErrSub string + }{ + { + name: "rdmaShared default prefix applied", + rendered: `apiVersion: mellanox.com/v1alpha1 +kind: NicClusterPolicy +spec: + rdmaSharedDevicePlugin: + config: '{"configList":[{"resourceName":"hca_shared_devices_a"}]}' +`, + want: []string{"rdma/hca_shared_devices_a"}, + }, + { + name: "sriov default prefix applied", + rendered: `kind: NicClusterPolicy +spec: + sriovDevicePlugin: + config: '{"resourceList":[{"resourceName":"vf_pool"}]}' +`, + want: []string{"intel.com/vf_pool"}, + }, + { + name: "non-NCP document skipped, no resources", + rendered: `kind: ConfigMap +metadata: + name: unrelated +`, + want: nil, + }, + { + name: "malformed config JSON fails closed", + rendered: `kind: NicClusterPolicy +spec: + rdmaSharedDevicePlugin: + config: 'not-json' +`, + wantErrSub: "failed to decode rdmaSharedDevicePlugin config JSON", + }, + { + name: "policy without device-plugin blocks yields none", + rendered: `kind: NicClusterPolicy +spec: + ofedDriver: + image: doca-driver +`, + want: nil, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + got, err := nicClusterPolicyResources([]byte(tt.rendered)) + if tt.wantErrSub != "" { + if err == nil { + t.Fatalf("nicClusterPolicyResources() error = nil, want error containing %q", tt.wantErrSub) + } + if !strings.Contains(err.Error(), tt.wantErrSub) { + t.Fatalf("nicClusterPolicyResources() error = %v, want substring %q", err, tt.wantErrSub) + } + return + } + if err != nil { + t.Fatalf("nicClusterPolicyResources() error = %v, want nil", err) + } + if len(got) != len(tt.want) { + t.Fatalf("nicClusterPolicyResources() = %v, want %v", got, tt.want) + } + for i := range got { + if got[i] != tt.want[i] { + t.Errorf("nicClusterPolicyResources()[%d] = %q, want %q", i, got[i], tt.want[i]) + } + } + }) + } +} diff --git a/validators/performance/nccl_all_reduce_bw_constraint.go b/validators/performance/nccl_all_reduce_bw_constraint.go index 92365f519a..f6066446c8 100644 --- a/validators/performance/nccl_all_reduce_bw_constraint.go +++ b/validators/performance/nccl_all_reduce_bw_constraint.go @@ -247,6 +247,10 @@ var supportedNCCLCombinations = map[ncclVariant]map[recipe.CriteriaServiceType][ }, variantNET: { recipe.CriteriaServiceEKS: {recipe.CriteriaAcceleratorGB200}, + // OKE GB200 NVL72: IB east-west (rdma0-3) via the + // rdmaSharedDevicePlugin's nvidia.com/mlnxnics shared HCAs — + // see testdata/gb200/oke/runtime-net.yaml. + recipe.CriteriaServiceOKE: {recipe.CriteriaAcceleratorGB200}, }, variantNVLS: { recipe.CriteriaServiceEKS: {recipe.CriteriaAcceleratorGB200}, @@ -393,11 +397,14 @@ func validateNcclAllReduceBw(ctx *validators.Context, constraint recipe.Constrai } // Preflight cluster-side prerequisites before spending TrainJob time. - // On GB200/EKS the NET variant needs NVreg_GrdmaPciTopoCheckOverride=1 - // on the NVIDIA driver; without it, EFA can't attach dma-buf to GPU HBM - // and NCCL silently falls back to Socket. Preflights key off the - // benchmark target: opting into a profile opts into that profile's - // environment contract, preflights included. + // On GB200/EKS and GB200/OKE the NET variant needs + // NVreg_GrdmaPciTopoCheckOverride=1 on the NVIDIA driver; without it, the + // PCIe-attached NIC (EFA on EKS, ConnectX IB on OKE) can't attach dma-buf + // to GPU HBM and NCCL silently falls back to Socket. Preflights key off + // the benchmark target: opting into a profile opts into that profile's + // environment contract, preflights included. (OKE takes the default + // fabric env here — AICR_NCCL_FABRIC's roce override is an EKS-only + // template concern.) if customRuntime == "" && fabric == fabricEFA && gb200NetPreflightApplies(variant, target.accelerator, target.service) { if pfErr := preflightGB200NetNVregFlag(ctx, gpuConfig.Nodes); pfErr != nil { return "", false, pfErr diff --git a/validators/performance/nccl_benchmark_profile_test.go b/validators/performance/nccl_benchmark_profile_test.go index bedc0f8768..99044c2ac6 100644 --- a/validators/performance/nccl_benchmark_profile_test.go +++ b/validators/performance/nccl_benchmark_profile_test.go @@ -156,7 +156,7 @@ func TestNCCLCombinationSupported(t *testing.T) { {"default B200 any", variantDefault, fabricEFA, target(recipe.CriteriaAcceleratorB200, recipe.CriteriaServiceAny), true}, {"default GB200 EKS not covered", variantDefault, fabricEFA, target(recipe.CriteriaAcceleratorGB200, recipe.CriteriaServiceEKS), false}, {"NET GB200 EKS", variantNET, fabricEFA, target(recipe.CriteriaAcceleratorGB200, recipe.CriteriaServiceEKS), true}, - {"NET GB200 OKE not covered", variantNET, fabricEFA, target(recipe.CriteriaAcceleratorGB200, recipe.CriteriaServiceOKE), false}, + {"NET GB200 OKE (IB via rdmaSharedDevicePlugin)", variantNET, fabricEFA, target(recipe.CriteriaAcceleratorGB200, recipe.CriteriaServiceOKE), true}, {"NVLS GB200 EKS", variantNVLS, fabricEFA, target(recipe.CriteriaAcceleratorGB200, recipe.CriteriaServiceEKS), true}, {"NVLS GB200 OKE", variantNVLS, fabricEFA, target(recipe.CriteriaAcceleratorGB200, recipe.CriteriaServiceOKE), true}, {"unknown service", variantNVLS, fabricEFA, target(recipe.CriteriaAcceleratorGB200, "custom-svc"), false}, diff --git a/validators/performance/nccl_preflight_nvreg.go b/validators/performance/nccl_preflight_nvreg.go index 158c00c78c..fd359aa13b 100644 --- a/validators/performance/nccl_preflight_nvreg.go +++ b/validators/performance/nccl_preflight_nvreg.go @@ -60,23 +60,33 @@ const ( preflightPodNamePrefix = "nccl-nvreg-probe-" // nvregDocsHint is the cluster-operator-facing message the preflight - // emits when the flag is missing. Keeps the fix one `kubectl` away. - nvregDocsHint = `NVreg_GrdmaPciTopoCheckOverride=1 is required on p6e-gb200 EKS nodes so ` + - `the NVIDIA driver allows EFA (a PCIe-attached NIC) to attach dma-buf ` + - `handles for GPU HBM on the Grace CPU topology. Without it, the kernel ` + - `rejects the attach with "NVRM: dma-buf attach failed: topology not ` + - `supported for mapping type FORCE_PCIE" and NCCL silently falls back ` + - `to the Socket transport. Set it via the GPU Operator ClusterPolicy: ` + - `spec.driver.kernelModuleConfig.name → a ConfigMap in gpu-operator ` + - `with data "nvidia.conf: options nvidia NVreg_GrdmaPciTopoCheckOverride=1", ` + - `then delete the nvidia-driver DaemonSet pods to pick up the change.` + // emits when the flag is missing. Keeps the fix one `kubectl` away, and + // names the remediation for BOTH driver-ownership modes: when the GPU + // Operator manages the driver (EKS p6e-gb200; OKE under + // gpuStack=operator-managed) the ConfigMap route applies, and when the + // driver ships in the node image (OKE's default oci-managed profile) the + // image or node bootstrap must set the module parameter itself — there is + // no driver DaemonSet for a ConfigMap to reach. + nvregDocsHint = `NVreg_GrdmaPciTopoCheckOverride=1 is required on GB200 nodes (EKS p6e-gb200 ` + + `and OKE NVL72) so the NVIDIA driver allows a PCIe-attached NIC (EFA on EKS, ` + + `ConnectX IB on OKE) to attach dma-buf handles for GPU HBM on the Grace CPU ` + + `topology. Without it, the kernel rejects the attach with "NVRM: dma-buf ` + + `attach failed: topology not supported for mapping type FORCE_PCIE" and NCCL ` + + `silently falls back to the Socket transport. When the GPU Operator manages ` + + `the driver, set it via the ClusterPolicy: spec.driver.kernelModuleConfig.name ` + + `→ a ConfigMap in gpu-operator with data "nvidia.conf: options nvidia ` + + `NVreg_GrdmaPciTopoCheckOverride=1", then delete the nvidia-driver DaemonSet ` + + `pods to pick up the change. When the driver ships in the node image (OKE ` + + `gpuStack=oci-managed), set the module parameter in the image or node ` + + `bootstrap (e.g. /etc/modprobe.d) and reboot the GPU nodes.` ) // preflightGB200NetNVregFlag verifies that every target GPU node has the // NVIDIA kernel driver loaded with NVreg_GrdmaPciTopoCheckOverride=1. Called -// only for the NET variant on GB200/EKS — this is the knob that determines -// whether EFA GPUDirect RDMA works on the GB200 PCI topology. NVLS (MNNVL) -// traffic stays on NVLink-C2C and does not need it. +// only for the NET variant on GB200/EKS and GB200/OKE — this is the knob that +// determines whether GPUDirect RDMA over a PCIe-attached NIC (EFA, ConnectX +// IB) works on the GB200 PCI topology. NVLS (MNNVL) traffic stays on +// NVLink-C2C and does not need it. // // The check runs one short-lived Pod per target node, pinned via NodeName, // with /proc/driver/nvidia hostPath-mounted read-only. The pod greps for the @@ -285,8 +295,17 @@ func waitForPreflightPodPhase(ctx context.Context, clientset kubernetes.Interfac // gb200NetPreflightApplies reports whether the preflight check should run for // the given (variant, accelerator, service) tuple. Keeps the call site at the // top of validateNcclAllReduceBw uncluttered. +// +// EKS and OKE are the two GB200 NET fabrics that traverse a PCIe-attached NIC +// (EFA and ConnectX IB respectively), so both need the dma-buf module flag. +// On OKE the flag reaches the driver only under gpuStack=operator-managed +// (the leaf's kernel-module-params ConfigMap needs a driver DaemonSet to +// consume it — see recipes/overlays/gb200-oke-training.yaml); under the +// default oci-managed profile the driver ships in the node image, so this +// preflight is the fail-closed gate that catches an image driver missing the +// flag before NCCL silently degrades to Socket (#2356 review). func gb200NetPreflightApplies(variant ncclVariant, accelerator recipe.CriteriaAcceleratorType, service recipe.CriteriaServiceType) bool { return variant == variantNET && accelerator == recipe.CriteriaAcceleratorGB200 && - service == recipe.CriteriaServiceEKS + (service == recipe.CriteriaServiceEKS || service == recipe.CriteriaServiceOKE) } diff --git a/validators/performance/nccl_preflight_nvreg_test.go b/validators/performance/nccl_preflight_nvreg_test.go index 597da8adde..e4ae74b499 100644 --- a/validators/performance/nccl_preflight_nvreg_test.go +++ b/validators/performance/nccl_preflight_nvreg_test.go @@ -94,8 +94,12 @@ func TestGB200NetPreflightApplies(t *testing.T) { variantNET, recipe.CriteriaAcceleratorGB200, recipe.CriteriaServiceGKE, false, }, { - "NET + GB200 + OKE → not required (no EFA on OKE)", - variantNET, recipe.CriteriaAcceleratorGB200, recipe.CriteriaServiceOKE, false, + "NET + GB200 + OKE → check required (ConnectX IB dma-buf on Grace topology)", + variantNET, recipe.CriteriaAcceleratorGB200, recipe.CriteriaServiceOKE, true, + }, + { + "NVLS + GB200 + OKE → not required (NVLink-C2C, no PCIe dma-buf)", + variantNVLS, recipe.CriteriaAcceleratorGB200, recipe.CriteriaServiceOKE, false, }, } for _, tt := range tests { diff --git a/validators/performance/testdata/gb200/oke/runtime-net.yaml b/validators/performance/testdata/gb200/oke/runtime-net.yaml new file mode 100644 index 0000000000..7ae7aea323 --- /dev/null +++ b/validators/performance/testdata/gb200/oke/runtime-net.yaml @@ -0,0 +1,208 @@ +# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# NCCL all-reduce TrainingRuntime for GB200 on OKE, NET/IB-transport variant. +# +# Measures the InfiniBand east-west fabric (rdma0-3, served by the +# rdmaSharedDevicePlugin from the gb200-oke NicClusterPolicy) instead of +# NVLink: NVLS and MNNVL are forced off so NCCL falls back to its built-in +# IB transport over the nvidia.com/mlnxnics shared HCAs. NCCL_DEBUG stays +# INFO — verifyTransportFromLogs confirms the transport by parsing the +# "NCCL INFO Using network" banner from launcher logs. +# +# Composition: the gb200/oke runtime-nvls.yaml scaffold (pytorch image, +# mpirun ssh wiring) with the IMEX resource claims removed and the +# net-variant env from gb200/eks runtime-net.yaml, minus the EFA provider +# knobs. Node pods request nvidia.com/mlnxnics alongside GPUs. +# +# Cluster prerequisites: rdmaSharedDevicePlugin healthy (nvidia.com/mlnxnics +# allocatable on every GPU node — the gb200-oke leaves' NicClusterPolicy), +# and NVreg_GrdmaPciTopoCheckOverride=1 from the leaf's kernel-module-params +# ConfigMap (dma-buf attach over the IB fabric). + +apiVersion: trainer.kubeflow.org/v1alpha1 +kind: TrainingRuntime +metadata: + name: nccl-all-reduce-runtime + namespace: ${NAMESPACE} + labels: + trainer.kubeflow.org/framework: mpi +spec: + mlPolicy: + mpi: + mpiImplementation: OpenMPI + numProcPerNode: ${GPU_COUNT_PER_NODE} + runLauncherAsNode: false + sshAuthMountPath: /tmp/mpi-keys + template: + spec: + network: + enableDNSHostnames: true + publishNotReadyAddresses: true + replicatedJobs: + - name: launcher + replicas: 1 + template: + spec: + template: + spec: + tolerations: + - operator: Exists + initContainers: + - name: fix-ssh-perms + image: nvcr.io/nvidia/pytorch:25.06-py3 + command: + - /bin/sh + - -c + - | + mkdir -p /root/.ssh + cp /tmp/mpi-keys/id_rsa /root/.ssh/id_rsa + cp /tmp/mpi-keys/authorized_keys /root/.ssh/authorized_keys + chmod 700 /root/.ssh + chmod 600 /root/.ssh/id_rsa /root/.ssh/authorized_keys + volumeMounts: + - name: mpi-ssh-auth + mountPath: /tmp/mpi-keys + readOnly: true + - name: ssh-config + mountPath: /root/.ssh + containers: + - name: node + image: nvcr.io/nvidia/pytorch:25.06-py3 + env: + - name: LD_LIBRARY_PATH + value: "/usr/local/nvidia/lib64:/usr/local/cuda/lib64" + command: + - /usr/local/mpi/bin/mpirun + args: + - -np + - "${GPU_COUNT}" + - --allow-run-as-root + - --mca + - plm_rsh_args + - -o StrictHostKeyChecking=no -o ConnectionAttempts=10 + - --mca + - btl + - ^openib + - --mca + - btl_tcp_if_include + - eth0 + - --mca + - oob_tcp_if_include + - eth0 + - -x + - LD_LIBRARY_PATH + - -x + - NCCL_DEBUG=INFO + # Force NCCL onto the NET/IB transport: disable NVLS (NVLink + # SHARP) and MNNVL (multi-node NVLink) so traffic crosses the + # IB fabric; NCCL_NET_PLUGIN=none selects the built-in IB + # verbs transport over the shared mlx5 HCAs. + - -x + - NCCL_NVLS_ENABLE=0 + - -x + - NCCL_MNNVL_ENABLE=0 + - -x + - NCCL_NET_PLUGIN=none + - -x + - NCCL_SOCKET_IFNAME=eth0 + - -x + - NCCL_IGNORE_DISABLED_P2P=1 + - /usr/local/bin/${TEST_TYPE}_mpi + - -b + - ${MIN_MESSAGE_SIZE} + - -e + - ${MAX_MESSAGE_SIZE} + - -f + - "2" + - -g + - "1" + resources: + limits: + cpu: "2" + memory: 128Mi + volumeMounts: + - name: ssh-config + mountPath: /root/.ssh + volumes: + - name: ssh-config + emptyDir: {} + - name: node + template: + spec: + template: + spec: + tolerations: + - operator: Exists + initContainers: + - name: fix-ssh-perms + image: nvcr.io/nvidia/pytorch:25.06-py3 + command: + - /bin/sh + - -c + - | + apt-get update && + apt-get install -y --no-install-recommends openssh-server && + mkdir -p /var/run/sshd && + chmod 0755 /var/run/sshd && + mkdir -p /root/.ssh && + cp /tmp/mpi-keys/authorized_keys /root/.ssh/authorized_keys && + chmod 700 /root/.ssh && + chmod 600 /root/.ssh/authorized_keys + volumeMounts: + - name: mpi-ssh-auth + mountPath: /tmp/mpi-keys + readOnly: true + - name: ssh-config + mountPath: /root/.ssh + containers: + - name: node + image: nvcr.io/nvidia/pytorch:25.06-py3 + command: ["sh", "-c"] + args: + - | + apt-get update && + apt-get install -y --no-install-recommends openssh-server && + mkdir -p /var/run/sshd && + chmod 0755 /var/run/sshd && + mkdir -p /root/.ssh && + cp /tmp/mpi-keys/* /root/.ssh/ && + chmod 700 /root/.ssh && + chmod 600 /root/.ssh/authorized_keys && + /usr/sbin/sshd -De + resources: + limits: + nvidia.com/gpu: ${GPU_COUNT_PER_NODE} + nvidia.com/mlnxnics: "1" + requests: + nvidia.com/gpu: ${GPU_COUNT_PER_NODE} + nvidia.com/mlnxnics: "1" + securityContext: + capabilities: + add: ["IPC_LOCK"] + volumeMounts: + - name: ssh-config + mountPath: /root/.ssh + - name: dshm + mountPath: /dev/shm + volumes: + - name: ssh-config + emptyDir: {} + - name: dshm + emptyDir: + medium: Memory + successPolicy: + operator: All + targetReplicatedJobs: + - launcher From 38a95259ca81180805191fe0fb5166c2de6c8064 Mon Sep 17 00:00:00 2001 From: Atif Mahmood Date: Thu, 3 Sep 2026 11:55:32 -0400 Subject: [PATCH 2/3] fix(validator): guard the RDMA gate's OKE label supply and reject nameless resources Follow-ups from the multi-persona review round: - Pin the RDMA readiness gate's cohort-label supply chain on OKE: the chart's own nvidia-nics-rules NodeFeatureRule (deployNodeFeatureRules default true) labels vendor-15b3 nodes pci-15b3.present - verified rendered from the pinned 26.4.1 chart with the OKE values and live on a BM.GPU.GB300 NVL72 cluster (18/18 nodes labeled). Comments in both OKE values files plus a recipes test now fail immediately if someone copies AKS's deployNodeFeatureRules:false without also attaching a targeted rule manifest. - qualifiedNICResource rejects a config entry without resourceName instead of deriving a phantom "/" the gate would poll until timeout; parser test row added. - Correct the fail-closed test docstring to the cases the table drives (zero-derivable-resources stays covered at the parser level). - Fix a stale generateName comment in the NVreg preflight. Signed-off-by: Atif Mahmood --- .../network-operator/values-oke-gb200.yaml | 7 ++ .../network-operator/values-oke-l40s.yaml | 7 ++ recipes/network_operator_nfr_test.go | 71 +++++++++++++++++++ validators/deployment/rdma_fabric_resource.go | 25 +++++-- .../deployment/rdma_fabric_resource_test.go | 17 ++++- .../performance/nccl_preflight_nvreg.go | 5 +- 6 files changed, 121 insertions(+), 11 deletions(-) create mode 100644 recipes/network_operator_nfr_test.go diff --git a/recipes/components/network-operator/values-oke-gb200.yaml b/recipes/components/network-operator/values-oke-gb200.yaml index 752d60dc1f..1bc4b8a522 100644 --- a/recipes/components/network-operator/values-oke-gb200.yaml +++ b/recipes/components/network-operator/values-oke-gb200.yaml @@ -19,6 +19,13 @@ # served by rdmaSharedDevicePlugin from the post-install NicClusterPolicy manifest, # NOT the chart. deployCR off so that manifest CR is authoritative. # nfd.enabled: false — GPU Operator's NFD is used; no second NFD. +# deployNodeFeatureRules stays at the chart default (true): the chart's +# nvidia-nics-rules NodeFeatureRule labels vendor-15b3 class-0200/0207 nodes +# feature.node.kubernetes.io/pci-15b3.present=true, which the deployment +# validator's RDMA readiness gate uses as its node cohort. Do NOT copy AKS's +# deployNodeFeatureRules: false here — AKS disables it only because it ships +# its own targeted rule manifest; without either source the gate's cohort is +# empty and it fails closed on every deploy. deployCR: false nvIpam: enabled: false diff --git a/recipes/components/network-operator/values-oke-l40s.yaml b/recipes/components/network-operator/values-oke-l40s.yaml index 6f0b4b2a60..14bdfa316e 100644 --- a/recipes/components/network-operator/values-oke-l40s.yaml +++ b/recipes/components/network-operator/values-oke-l40s.yaml @@ -28,6 +28,13 @@ # the only place the OCI VF selectors 101a/101e can be expressed); the operator # reconciles nv-ipam + secondaryNetwork + sriovDevicePlugin from that CR regardless. # nfd.enabled: false — GPU Operator's NFD is used; no second NFD. +# deployNodeFeatureRules stays at the chart default (true): the chart's +# nvidia-nics-rules NodeFeatureRule labels vendor-15b3 class-0200/0207 nodes +# feature.node.kubernetes.io/pci-15b3.present=true, which the deployment +# validator's RDMA readiness gate uses as its node cohort. Do NOT copy AKS's +# deployNodeFeatureRules: false here — AKS disables it only because it ships +# its own targeted rule manifest; without either source the gate's cohort is +# empty and it fails closed on every deploy. deployCR: false nvIpam: enabled: false diff --git a/recipes/network_operator_nfr_test.go b/recipes/network_operator_nfr_test.go new file mode 100644 index 0000000000..c53889116d --- /dev/null +++ b/recipes/network_operator_nfr_test.go @@ -0,0 +1,71 @@ +// Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +package recipes + +import ( + "io/fs" + "testing" + + "sigs.k8s.io/yaml" +) + +// TestOKENetworkOperatorKeepsChartNodeFeatureRule pins the label supply chain +// behind the deployment validator's RDMA readiness gate: the gate's node +// cohort is nodes labeled feature.node.kubernetes.io/pci-15b3.present=true +// (validators/helper.PCIMellanoxPresentLabel). On OKE that label comes from +// the network-operator chart's own nvidia-nics-rules NodeFeatureRule, which +// renders whenever `deployNodeFeatureRules` is left at the chart default +// (true) — the OKE values disable the chart's NFD deployment (nfd.enabled: +// false, GPU Operator's NFD processes the rule) but must NOT disable the +// rule itself. AKS is the deliberate exception: it sets +// deployNodeFeatureRules: false and attaches its own targeted rule manifest +// (nfd-network-rule.yaml) instead. +// +// A contributor copying AKS's `deployNodeFeatureRules: false` into an OKE +// values file — without also attaching a rule manifest — would leave no +// producer for the cohort label, and the RDMA gate would fail closed on +// every OKE fabric deploy ("no schedulable Mellanox RDMA-capable GPU nodes +// observed"). This test turns that mistake into an immediate failure. +func TestOKENetworkOperatorKeepsChartNodeFeatureRule(t *testing.T) { + t.Parallel() + + for _, p := range []string{ + "components/network-operator/values-oke-gb200.yaml", + "components/network-operator/values-oke-l40s.yaml", + } { + t.Run(p, func(t *testing.T) { + t.Parallel() + data, err := fs.ReadFile(FS, p) + if err != nil { + t.Fatalf("read %s: %v", p, err) + } + var values map[string]any + if err := yaml.Unmarshal(data, &values); err != nil { + t.Fatalf("parse %s: %v", p, err) + } + v, present := values["deployNodeFeatureRules"] + if !present { + return // chart default (true) applies — the rule renders + } + enabled, ok := v.(bool) + if !ok || !enabled { + t.Fatalf("%s sets deployNodeFeatureRules=%v: the RDMA readiness gate's "+ + "cohort label (pci-15b3.present) has no other producer on OKE — either "+ + "leave the chart default or attach a targeted NodeFeatureRule manifest "+ + "like AKS does", p, v) + } + }) + } +} diff --git a/validators/deployment/rdma_fabric_resource.go b/validators/deployment/rdma_fabric_resource.go index 153d8c3311..bc6cbca6aa 100644 --- a/validators/deployment/rdma_fabric_resource.go +++ b/validators/deployment/rdma_fabric_resource.go @@ -182,7 +182,11 @@ func nicClusterPolicyResources(rendered []byte) ([]string, error) { "failed to decode rdmaSharedDevicePlugin config JSON", err) } for _, e := range cfg.ConfigList { - out = append(out, qualifiedNICResource(e, rdmaSharedDefaultResourcePrefix)) + r, rerr := qualifiedNICResource(e, rdmaSharedDefaultResourcePrefix) + if rerr != nil { + return nil, rerr + } + out = append(out, r) } } if p := ncp.Spec.SriovDevicePlugin; p != nil { @@ -194,7 +198,11 @@ func nicClusterPolicyResources(rendered []byte) ([]string, error) { "failed to decode sriovDevicePlugin config JSON", err) } for _, e := range cfg.ResourceList { - out = append(out, qualifiedNICResource(e, sriovDefaultResourcePrefix)) + r, rerr := qualifiedNICResource(e, sriovDefaultResourcePrefix) + if rerr != nil { + return nil, rerr + } + out = append(out, r) } } } @@ -203,11 +211,18 @@ func nicClusterPolicyResources(rendered []byte) ([]string, error) { // qualifiedNICResource joins a config entry into the "/" // extended-resource form, applying the plugin's documented default prefix -// when the entry omits one. -func qualifiedNICResource(e nicResourceEntry, defaultPrefix string) string { +// when the entry omits one. An entry with no resourceName is an error — +// "/" is not a resource any plugin will ever advertise, and letting +// it through would make the readiness gate poll a phantom until timeout +// instead of naming the malformed config. +func qualifiedNICResource(e nicResourceEntry, defaultPrefix string) (string, error) { + if e.ResourceName == "" { + return "", errors.New(errors.ErrCodeInternal, + "device-plugin config entry declares no resourceName") + } prefix := e.ResourcePrefix if prefix == "" { prefix = defaultPrefix } - return prefix + "/" + e.ResourceName + return prefix + "/" + e.ResourceName, nil } diff --git a/validators/deployment/rdma_fabric_resource_test.go b/validators/deployment/rdma_fabric_resource_test.go index 4eefc769d3..b64cd59209 100644 --- a/validators/deployment/rdma_fabric_resource_test.go +++ b/validators/deployment/rdma_fabric_resource_test.go @@ -73,9 +73,11 @@ func TestRDMAFabricResource_RealManifests(t *testing.T) { } // TestRDMAFabricResource_FailsClosed proves derivation failures are errors, -// never a silent skip: a fabric-declaring ref whose policy yields no resource, -// distinct resources across manifests, and an unreadable manifest path all -// fail. +// never a silent skip: an unreadable manifest path and distinct resources +// across manifests both fail. (The zero-derivable-resources branch is +// exercised at the parser level in TestNICClusterPolicyResources_ParseShapes +// — no embedded marker-named manifest without a device-plugin block exists +// to drive it end to end.) func TestRDMAFabricResource_FailsClosed(t *testing.T) { t.Parallel() @@ -158,6 +160,15 @@ metadata: `, want: nil, }, + { + name: "config entry without resourceName fails closed", + rendered: `kind: NicClusterPolicy +spec: + rdmaSharedDevicePlugin: + config: '{"configList":[{"resourcePrefix":"rdma"}]}' +`, + wantErrSub: "declares no resourceName", + }, { name: "malformed config JSON fails closed", rendered: `kind: NicClusterPolicy diff --git a/validators/performance/nccl_preflight_nvreg.go b/validators/performance/nccl_preflight_nvreg.go index fd359aa13b..35065dc7cd 100644 --- a/validators/performance/nccl_preflight_nvreg.go +++ b/validators/performance/nccl_preflight_nvreg.go @@ -54,9 +54,8 @@ func parseNVregFromParams(content string) bool { const ( // preflightPodNamePrefix is the generateName seed for the per-node probe - // pods. Short so the full name (including node hash + rand suffix) fits - // inside the 63-character DNS-1123 label limit on all realistic node - // names. + // pods. Short so the full name (with the apiserver's random generateName + // suffix) stays well inside the 63-character DNS-1123 label limit. preflightPodNamePrefix = "nccl-nvreg-probe-" // nvregDocsHint is the cluster-operator-facing message the preflight From b6532b3775d86322db081b270e1731a39a9c9348 Mon Sep 17 00:00:00 2001 From: Atif Mahmood Date: Thu, 3 Sep 2026 15:42:32 -0400 Subject: [PATCH 3/3] fix(recipes): make the OKE NodeFeatureRule guard assert merged effective values The previous guard read deployNodeFeatureRules at the top level of the raw overlay YAML, but the chart key lives under nfd:, so both subtests took the early return and asserted nothing. Rewritten per review: resolves MERGED effective values (base values.yaml -> overlay) via recipe.GetComponentValuesWithContext, reads nfd.deployNodeFeatureRules, and derives the overlay set by glob so future OKE fabric leaves are covered automatically. Mutation-verified: flips at both the overlay level and the shared base values.yaml now fail the test. External test package to avoid the pkg/recipe -> recipes import cycle. Signed-off-by: Atif Mahmood --- docs/user/container-images.md | 8 ++- pkg/bundler/testdata/stock_render_golden.yaml | 4 +- .../testdata/catalog_parity_golden.yaml | 4 +- recipes/network_operator_nfr_test.go | 70 ++++++++++++------- 4 files changed, 55 insertions(+), 31 deletions(-) diff --git a/docs/user/container-images.md b/docs/user/container-images.md index cc8b1084af..e67eea7d48 100644 --- a/docs/user/container-images.md +++ b/docs/user/container-images.md @@ -20,7 +20,7 @@ A machine-readable **CycloneDX 1.6 JSON** companion to this page is produced by ## Summary - Components: **44** -- Unique images: **97** +- Unique images: **101** - Distinct registries: **11** Registries: `602401143452.dkr.ecr.us-west-2.amazonaws.com`, `cr.agentgateway.dev`, `docker.io`, `gcr.io`, `ghcr.io`, `gke.gcr.io`, `nvcr.io`, `public.ecr.aws`, `quay.io`, `registry.k8s.io`, `us-docker.pkg.dev` @@ -56,7 +56,7 @@ _Rendering fidelity:_ `catalog-parity: charts are rendered with the shared recip | kueue | helm | kueue | 0.18.2 | 1 | | mariadb-operator | helm | mariadb-operator | 26.6.0 | 1 | | mariadb-operator-crds | helm | mariadb-operator-crds | 26.6.0 | 0 | -| network-operator | helm | nvidia/network-operator | 26.4.1 | 5 | +| network-operator | helm | nvidia/network-operator | 26.4.1 | 9 | | network-operator-ocp | manifest | — | — | 0 | | network-operator-ocp-olm | manifest | — | — | 0 | | nfd | helm | node-feature-discovery | 0.19.0 | 1 | @@ -238,6 +238,10 @@ _No images extracted._ ### network-operator - `docker.io/library/busybox:1.38.0@sha256:dc2d74b28e4cf8984fa52af1f39bc7c3d9c73760b41a74d629f5d11b1ab28616` +- `ghcr.io/k8snetworkplumbingwg/multus-cni:v4.2.1` +- `ghcr.io/k8snetworkplumbingwg/plugins:v1.6.2-update.1` +- `ghcr.io/k8snetworkplumbingwg/sriov-network-device-plugin:v3.9.0` +- `ghcr.io/mellanox/nvidia-k8s-ipam:v0.2.0` - `nvcr.io/nvidia/cloud-native/network-operator:v26.4.1` - `nvcr.io/nvidia/doca/doca_telemetry:1.22.5-doca3.1.0-host` - `nvcr.io/nvidia/mellanox/doca-driver:doca3.2.0-25.10-1.2.8.0-2` diff --git a/pkg/bundler/testdata/stock_render_golden.yaml b/pkg/bundler/testdata/stock_render_golden.yaml index d981380b8c..aab2e94792 100644 --- a/pkg/bundler/testdata/stock_render_golden.yaml +++ b/pkg/bundler/testdata/stock_render_golden.yaml @@ -17,7 +17,7 @@ gb200-eks-ubuntu-inference-dynamo: 0cb8d348251d816706a3de33c3258cf7571d03fe9651c gb200-eks-ubuntu-training-kubeflow: 9a9bebc7996647e0da42de05f30c7dd126c321fcf9e4642723a2fc1b9dc9aa43 gb200-eks-ubuntu-training-slurm: e0d0e22306d16edaaa61ac2d406bfeee13d3b8738a13500616a6258faba85d28 gb200-oke-ubuntu-inference-dynamo: 0678d70588d81be1ebffda2497075b7e237d73049f6881ecb2cffd34b6e1842d -gb200-oke-ubuntu-training-kubeflow: ac4fb9394bf3849e20413391d84c971daa87077e7311f7912d289fd84829ae3f +gb200-oke-ubuntu-training-kubeflow: b4cdebf0db1373bfe855f9fd80af212e0d9902bff7e3e4efe22d0c9a672887b2 gb300-any: 89784fc824cc7bed3c2af68b5e1d8858d471f9a3116a15e95a975f68747037e1 gb300-eks-ubuntu-inference-dynamo: beead8d2397289ec05ccf41397dac2866d7773cfaa3fc8d851556210fa14198d gb300-eks-ubuntu-training-kubeflow: a56c6fe3dc703ff067b38bb7144aab91b5a4fda1c63e02d50b178a6d735be32f @@ -43,7 +43,7 @@ h200-eks-training: c062ae91b06cd7d6a008269be3a79263b666c62a2407d9b4a5aaa5cc9c2f3 l40-any: d2c4703050a708256e4073721b73bd7fc6947c9af264166c179e78d9f1d36ac6 l40s-any: 37d3fefd82f7f69f77e5d5554600eb1af2b5113707a662823e39c5846a632ccd l40s-oke-inference: c23425edf569c5fc98c6b3277cbe4f5a73a6ada607d760dd5e4d0d1daf516856 -l40s-oke-training: fe534d483969dc6e8ae7f0d619b36b8dcf35fc82c37fcaf1502975551cc99fef +l40s-oke-training: 4274029777e2a396d00cbab17c1c6ee9f13c77325533fd6f5110f869d7508142 monitoring-hpa: 5269e14167c645177fce3d83b2f33d94a615b6ce9c099a84a8aa3f56ab3f00fd ocp-inference-nim: bb3c7fa78c972241b47abedbff3069d4fad35ab0c64d5f0172c633d8ed610b27 ocp-training: 1f457d2c2aed921c2a3281cca1a55b1181d979cc99f07c1b51b9ce827379b401 diff --git a/pkg/recipe/testdata/catalog_parity_golden.yaml b/pkg/recipe/testdata/catalog_parity_golden.yaml index 8cf3747eec..ddc88eefd6 100644 --- a/pkg/recipe/testdata/catalog_parity_golden.yaml +++ b/pkg/recipe/testdata/catalog_parity_golden.yaml @@ -17,7 +17,7 @@ gb200-eks-ubuntu-inference-dynamo: d8fd0d57581a7c826b27eb6dc4ba63d3ee063aa58cf25 gb200-eks-ubuntu-training-kubeflow: 77a7f39257ac4a7abb8567386196949029c740dd910f9bfaa2e010946d3a8261 gb200-eks-ubuntu-training-slurm: cc54ba8b5ab6deb57a642abc8462360d3a8d854d6b9274b6f7f51a9c0b63f0cc gb200-oke-ubuntu-inference-dynamo: 56955bb6fa6086b5d7c1431145c574f99b9550ac29c565006d65a3dacd1c1871 -gb200-oke-ubuntu-training-kubeflow: a227c194b8eb1023b53d7fe52dbde14bcf5d45a90979317340f987e6ff331cbb +gb200-oke-ubuntu-training-kubeflow: 0bb6faa6b2bbd9de2dcf825e36ceca1789e98e9782115ed06575c775155c7f4b gb300-any: 15930cf8c57d0d59b3493c4ec46e8455179d8d6a32c635315b32d89619904ff6 gb300-eks-ubuntu-inference-dynamo: fc61807bfc29c7fff199a6a1d07d7aa6f7be217ca8a7c6b671087567360539f3 gb300-eks-ubuntu-training-kubeflow: eb4ffd9bafec87bf85044d732864f3814937073edf919b5821373d4754dd99ba @@ -43,7 +43,7 @@ h200-eks-training: 4b484e17c89074fa1322b54b5686cea3c1d862ffc3af04730ae760491296f l40-any: 09de30af8e20c64a19f47afcbb0c70d67785e2ce99be20e3ce53a39e2a3abaf7 l40s-any: 66644f104b5b2f0129756763ad89db5deef008ecf1453a6abc9f950870788aca l40s-oke-inference: a639160bec239422ce5f34d0b48674c10f4016e43d48260db07fc9f3f865eb6a -l40s-oke-training: 7395b51b4f8063c2b3fc86eb446dd1be2e3591fd018cc76a7cbd7c7c71463973 +l40s-oke-training: 81a0d7ec5dd5e85cd01099ce44b30131122b3bbef11f39f1b271455030c6a3e6 monitoring-hpa: a9847bb17e9dbd73bd24e7282dcdc66aaa40716dfafc2a4b855af85facea5223 ocp-inference-nim: 1ce82c596d41ad8953eef488fdfddec7dd8f83c0114b1be63a975f59257e646d ocp-training: 88fc2b45c7aabf66fc023c9ed40bf520b0d4e58831435963f19fe4c1b4001ac3 diff --git a/recipes/network_operator_nfr_test.go b/recipes/network_operator_nfr_test.go index c53889116d..ad7952af3b 100644 --- a/recipes/network_operator_nfr_test.go +++ b/recipes/network_operator_nfr_test.go @@ -12,13 +12,17 @@ // See the License for the specific language governing permissions and // limitations under the License. -package recipes +// External test package: pkg/recipe imports this package for the embedded +// catalog FS, so resolving effective values through pkg/recipe from an +// in-package test would be an import cycle. +package recipes_test import ( "io/fs" "testing" - "sigs.k8s.io/yaml" + "github.com/NVIDIA/aicr/pkg/recipe" + "github.com/NVIDIA/aicr/recipes" ) // TestOKENetworkOperatorKeepsChartNodeFeatureRule pins the label supply chain @@ -26,45 +30,61 @@ import ( // cohort is nodes labeled feature.node.kubernetes.io/pci-15b3.present=true // (validators/helper.PCIMellanoxPresentLabel). On OKE that label comes from // the network-operator chart's own nvidia-nics-rules NodeFeatureRule, which -// renders whenever `deployNodeFeatureRules` is left at the chart default +// renders whenever `nfd.deployNodeFeatureRules` is left at the chart default // (true) — the OKE values disable the chart's NFD deployment (nfd.enabled: -// false, GPU Operator's NFD processes the rule) but must NOT disable the +// false; the GPU Operator's NFD processes the rule) but must NOT disable the // rule itself. AKS is the deliberate exception: it sets -// deployNodeFeatureRules: false and attaches its own targeted rule manifest -// (nfd-network-rule.yaml) instead. +// nfd.deployNodeFeatureRules: false and attaches its own targeted rule +// manifest (nfd-network-rule.yaml) instead. // // A contributor copying AKS's `deployNodeFeatureRules: false` into an OKE -// values file — without also attaching a rule manifest — would leave no -// producer for the cohort label, and the RDMA gate would fail closed on -// every OKE fabric deploy ("no schedulable Mellanox RDMA-capable GPU nodes -// observed"). This test turns that mistake into an immediate failure. +// values file — or flipping it in the shared component base values.yaml — +// without also attaching a rule manifest would leave no producer for the +// cohort label, and the RDMA gate would fail closed on every OKE fabric +// deploy ("no schedulable Mellanox RDMA-capable GPU nodes observed"). This +// test turns that mistake into an immediate failure. +// +// It asserts on MERGED effective values (base values.yaml → overlay), exactly +// as bundle rendering resolves them, and derives the overlay set by glob so a +// future OKE fabric leaf is covered without editing this test. func TestOKENetworkOperatorKeepsChartNodeFeatureRule(t *testing.T) { t.Parallel() - for _, p := range []string{ - "components/network-operator/values-oke-gb200.yaml", - "components/network-operator/values-oke-l40s.yaml", - } { - t.Run(p, func(t *testing.T) { + overlays, err := fs.Glob(recipes.FS, "components/network-operator/values-oke-*.yaml") + if err != nil { + t.Fatalf("glob OKE network-operator values overlays: %v", err) + } + if len(overlays) == 0 { + t.Fatal("no components/network-operator/values-oke-*.yaml overlays found; " + + "the glob or the embedded FS layout changed") + } + + for _, overlay := range overlays { + t.Run(overlay, func(t *testing.T) { t.Parallel() - data, err := fs.ReadFile(FS, p) + + ref := recipe.ComponentRef{Name: "network-operator", ValuesFile: overlay} + values, err := recipe.GetComponentValuesWithContext(t.Context(), nil, &ref) if err != nil { - t.Fatalf("read %s: %v", p, err) + t.Fatalf("resolve effective values for %s: %v", overlay, err) } - var values map[string]any - if err := yaml.Unmarshal(data, &values); err != nil { - t.Fatalf("parse %s: %v", p, err) + + nfd, ok := values["nfd"].(map[string]any) + if !ok { + // No nfd block anywhere in the merged values — both chart + // defaults apply, including deployNodeFeatureRules: true. + return } - v, present := values["deployNodeFeatureRules"] + v, present := nfd["deployNodeFeatureRules"] if !present { return // chart default (true) applies — the rule renders } enabled, ok := v.(bool) if !ok || !enabled { - t.Fatalf("%s sets deployNodeFeatureRules=%v: the RDMA readiness gate's "+ - "cohort label (pci-15b3.present) has no other producer on OKE — either "+ - "leave the chart default or attach a targeted NodeFeatureRule manifest "+ - "like AKS does", p, v) + t.Fatalf("effective values for %s set nfd.deployNodeFeatureRules=%v: the RDMA "+ + "readiness gate's cohort label (pci-15b3.present) has no other producer on "+ + "OKE — either leave the chart default or attach a targeted NodeFeatureRule "+ + "manifest like AKS does", overlay, v) } }) }