diff --git a/docs/user/component-catalog.md b/docs/user/component-catalog.md index 37b214cd8f..b2610bb943 100644 --- a/docs/user/component-catalog.md +++ b/docs/user/component-catalog.md @@ -59,7 +59,7 @@ The source of truth is [`recipes/registry.yaml`](https://github.com/NVIDIA/aicr/ | **cert-manager-ocp-olm** | OLM installer for cert-manager on OpenShift. Creates the OperatorGroup and Subscription resources that install the certified cert-manager Operator via the Operator Lifecycle Manager. Paired with `cert-manager-ocp`. OCP-specific. | [cert-manager (Certified)](https://catalog.redhat.com/software/container-stacks/detail/5ec3f5a5eebc3d6acb0ee71c) | | **cert-manager-ocp** | cert-manager CertManager CR for OpenShift. The operand Deployments (controller, cainjector, webhook) land in a hardcoded `cert-manager` namespace regardless of the operator's own namespace. Deployed after `cert-manager-ocp-olm`. OCP-specific. | [cert-manager](https://github.com/cert-manager/cert-manager) | | **prometheus-adapter-ocp** | Prometheus Adapter for OpenShift. Reuses the same upstream chart as `prometheus-adapter`, pointed at OCP's built-in Thanos Querier instead of kube-prometheus-stack (which stays disabled on OCP). No certified OCP operator exists for this component. OCP-specific. | [prometheus-adapter](https://github.com/kubernetes-sigs/prometheus-adapter) | -| **nvidia-dra-driver-gpu-ocp** | NVIDIA DRA GPU driver for OpenShift. Reuses the same upstream chart as `nvidia-dra-driver-gpu`, with an added SCC RoleBinding granting the kubelet-plugin DaemonSet the host device access OCP's default restricted-v2 SCC forbids. No certified OCP operator exists for this component. OCP-specific. Known limitation: some GPU-driver rollout protections and remedy hints do not yet cover the OCP aliases (`gpu-operator-ocp`, `nvidia-dra-driver-gpu-ocp`) — the deployer's stale-NVML migration wait/restart, driver-version annotation injection, and the driver-absent remedy's `gpuoperator:`/`dradriver:` override keys; tracked in [#2136](https://github.com/NVIDIA/aicr/issues/2136). | [NVIDIA DRA Driver](https://github.com/kubernetes-sigs/dra-driver-nvidia-gpu) | +| **nvidia-dra-driver-gpu-ocp** | NVIDIA DRA GPU driver for OpenShift. Reuses the same upstream chart as `nvidia-dra-driver-gpu`, with an added SCC RoleBinding granting the kubelet-plugin DaemonSet the host device access OCP's default restricted-v2 SCC forbids. No certified OCP operator exists for this component. OCP-specific. Known limitation: the driver-version annotation injected onto the DRA pod templates falls back to the `gpu-operator-ocp-olm` Subscription channel, which changes on a channel re-pin but not on every in-channel OLM auto-upgrade — so the stale-NVML rollout gate (#973) can still miss an in-channel driver bump on OCP; tracked in [#2135](https://github.com/NVIDIA/aicr/issues/2135). | [NVIDIA DRA Driver](https://github.com/kubernetes-sigs/dra-driver-nvidia-gpu) | | **k8s-nim-operator-ocp** | NVIDIA NIM Operator for OpenShift. Reuses the same upstream chart as `k8s-nim-operator`, with OCP-specific RBAC. Requires `cert-manager-ocp` for admission-webhook TLS. OCP-specific. | [K8s NIM Operator](https://github.com/NVIDIA/k8s-nim-operator) | ## VR200 Preview coverage diff --git a/pkg/bundler/bundler.go b/pkg/bundler/bundler.go index 283aa3f164..37e255752d 100644 --- a/pkg/bundler/bundler.go +++ b/pkg/bundler/bundler.go @@ -3418,11 +3418,13 @@ const draChartVersionAnnotation = header.Domain + "/gpu-operator-chart-version" // filtered resolved recipe before derived values are written; recipes that // disable either remain untouched. const ( - gpuOperatorComponentName = "gpu-operator" - draComponentName = "nvidia-dra-driver-gpu" - draEvictionEnvName = "NODE_LABEL_FOR_GPU_POD_EVICTION" - draEvictionNodeSelectorPath = "kubeletPlugin.nodeSelector" - gpuOperatorDRAEvictionEnvPath = "driver.manager.env" + gpuOperatorComponentName = "gpu-operator" + gpuOperatorOCPComponentName = "gpu-operator-ocp" + gpuOperatorOCPOLMComponentName = "gpu-operator-ocp-olm" + draComponentName = "nvidia-dra-driver-gpu" + draEvictionEnvName = "NODE_LABEL_FOR_GPU_POD_EVICTION" + draEvictionNodeSelectorPath = "kubeletPlugin.nodeSelector" + gpuOperatorDRAEvictionEnvPath = "driver.manager.env" // draNodeLabelerComponentName is the manifest-only component that mirrors // GFD's nvidia.com/gpu.present onto the eviction label, so the label is @@ -3442,7 +3444,7 @@ const ( ) var ( - gpuOperatorComponentNames = []string{gpuOperatorComponentName, "gpu-operator-ocp"} + gpuOperatorComponentNames = []string{gpuOperatorComponentName, gpuOperatorOCPComponentName} draComponentNames = []string{draComponentName, "nvidia-dra-driver-gpu-ocp"} ) @@ -3972,6 +3974,31 @@ func (b *DefaultBundler) injectDRAChartVersionAnnotation( // is exercised by the disabled-component unit tests. return } + if gpuOperatorComponentName == gpuOperatorOCPComponentName && gpuOperatorVersion == "" { + // gpu-operator-ocp is a ClusterPolicy CR, not a Helm chart, so + // ComponentRef.Version is never populated for it — the empty + // check below would always skip injection on OCP. Fall back to + // the OLM Subscription channel (gpu-operator-ocp-olm) as the + // rollout-trigger value instead. + // + // KNOWN LIMITATION: the channel pin (e.g. "v25.10") only + // changes on a channel re-pin, not on every operator update. + // With installPlanApproval: Automatic (the default — + // components/gpu-operator-ocp-olm/values.yaml), OLM can + // upgrade to newer CSVs inside the same channel — reloading + // the driver — without the channel string changing, so this + // annotation catches bundle-driven operator bumps (a recipe + // regenerated against a different channel) but NOT in-channel + // auto-upgrades. The stale-NVML gap this annotation exists to + // close (#973) remains open for that case on OCP. See #2135. + if olmValues, ok := componentValues[gpuOperatorOCPOLMComponentName]; ok { + if sub, ok := olmValues["subscription"].(map[string]any); ok { + if channel, ok := sub["channel"].(string); ok { + gpuOperatorVersion = channel + } + } + } + } if gpuOperatorVersion == "" { // gpu-operator is enabled but the resolver produced an empty // Version string. This shouldn't happen in normal recipe diff --git a/pkg/bundler/bundler_dra_annotation_test.go b/pkg/bundler/bundler_dra_annotation_test.go index 0b3de87d57..138c734681 100644 --- a/pkg/bundler/bundler_dra_annotation_test.go +++ b/pkg/bundler/bundler_dra_annotation_test.go @@ -193,6 +193,47 @@ func TestInjectDRAChartVersionAnnotation_PreservesExistingValues(t *testing.T) { } } +// TestInjectDRAChartVersionAnnotation_OCPFallbackToOLMChannel pins the +// OCP fallback added for #2135: gpu-operator-ocp is a ClusterPolicy +// CR, not a Helm chart, so ComponentRef.Version is always empty for +// it. Instead of skipping injection (the pre-fix behavior), the +// helper reads the OLM Subscription channel from the +// gpu-operator-ocp-olm component's values and mirrors that onto both +// nvidia-dra-driver-gpu-ocp pod templates. +func TestInjectDRAChartVersionAnnotation_OCPFallbackToOLMChannel(t *testing.T) { + b, err := New() + if err != nil { + t.Fatalf("New() error = %v", err) + } + + const draOCPComponentName = "nvidia-dra-driver-gpu-ocp" + componentValues := map[string]map[string]any{ + gpuOperatorOCPComponentName: {}, + draOCPComponentName: {}, + gpuOperatorOCPOLMComponentName: { + "subscription": map[string]any{ + "channel": "v25.10", + }, + }, + } + rr := &recipe.RecipeResult{ + ComponentRefs: []recipe.ComponentRef{ + {Name: gpuOperatorOCPComponentName, Version: ""}, + {Name: draOCPComponentName, Version: "0.4.1"}, + }, + } + + b.injectDRAChartVersionAnnotation(componentValues, rr) + + for _, podPath := range []string{"controller", "kubeletPlugin"} { + got := dig(componentValues[draOCPComponentName], podPath, "podAnnotations", draChartVersionAnnotation) + if got != "v25.10" { + t.Errorf("podAnnotations[%s][%s] = %v, want v25.10 (OLM channel fallback)", + podPath, draChartVersionAnnotation, got) + } + } +} + // TestInjectDRAChartVersionAnnotation_OverridesUserSet pins the // "internal annotation always reflects the actual chart version" // invariant. A user --set that wrote a stale value into the diff --git a/pkg/bundler/deployer/helm/helm.go b/pkg/bundler/deployer/helm/helm.go index 6a4d9a3cad..d572f9b711 100644 --- a/pkg/bundler/deployer/helm/helm.go +++ b/pkg/bundler/deployer/helm/helm.go @@ -62,6 +62,15 @@ type ComponentData struct { IsOCI bool Tag string // Git ref for Kustomize-typed components (tag/branch/commit) Path string // Path within the repository to the kustomization + + // DriverOperatorManaged is true when the bundle's effective values + // select an operator-managed NVIDIA driver — gpu-operator's or + // gpu-operator-ocp's driver.enabled is true. deploy.sh's DRA + // migration-wait block (see #2135, #973) uses this to tell "driver + // is host-managed" apart from "driver is operator-managed but the + // DaemonSet/node-label migration signal isn't observable yet", + // which live cluster state alone cannot distinguish. + DriverOperatorManaged bool } // compile-time interface check @@ -287,6 +296,42 @@ func (g *Generator) Generate(ctx context.Context, outputDir string) (*deployer.O // buildComponentDataList builds a sorted list of ComponentData from the recipe. // It validates that all component names are safe for use as directory names. +// driverOperatorManaged reports whether this bundle's effective values +// select an operator-managed NVIDIA driver: gpu-operator's or +// gpu-operator-ocp's driver.enabled is true. Checks both component names +// since only one is ever enabled in a given recipe (see +// pkg/bundler/bundler.go's gpuOperatorComponentNames for the canonical +// list this mirrors). +// gpuOperatorComponentName and gpuOperatorOCPComponentName are this +// package's copy of the canonical/OCP gpu-operator component names (a +// 4th duplicate alongside pkg/bundler/bundler.go, pkg/bundler/validations +// /checks.go, and their override-key constants — this package cannot +// import pkg/bundler due to the dependency cycle noted at +// componentOverrideKeys' godoc equivalent). Named here, rather than an +// inline literal, so a `grep gpuOperatorOCPComponentName` across the repo +// surfaces every copy that needs updating together. +const ( + gpuOperatorComponentName = "gpu-operator" + gpuOperatorOCPComponentName = "gpu-operator-ocp" +) + +func (g *Generator) driverOperatorManaged() bool { + for _, name := range []string{gpuOperatorComponentName, gpuOperatorOCPComponentName} { + values, ok := g.ComponentValues[name] + if !ok { + continue + } + driver, ok := values["driver"].(map[string]any) + if !ok { + continue + } + if enabled, ok := driver["enabled"].(bool); ok && enabled { + return true + } + } + return false +} + // Only the fields consumed by the orchestration templates are populated. func (g *Generator) buildComponentDataList() ([]ComponentData, error) { // Sort by deployment order @@ -295,6 +340,8 @@ func (g *Generator) buildComponentDataList() ([]ComponentData, error) { g.RecipeResult.DeploymentOrder, ) + driverOperatorManaged := g.driverOperatorManaged() + components := make([]ComponentData, 0, len(sorted)) for _, ref := range sorted { if !deployer.IsSafePathComponent(ref.Name) { @@ -305,14 +352,15 @@ func (g *Generator) buildComponentDataList() ([]ComponentData, error) { chartName := ref.EffectiveChart() components = append(components, ComponentData{ - Name: ref.Name, - Namespace: ref.Namespace, - Repository: ref.Source, - ChartName: chartName, - Version: ref.Version, - IsOCI: strings.HasPrefix(ref.Source, "oci://"), - Tag: ref.Tag, - Path: ref.Path, + Name: ref.Name, + Namespace: ref.Namespace, + Repository: ref.Source, + ChartName: chartName, + Version: ref.Version, + IsOCI: strings.HasPrefix(ref.Source, "oci://"), + Tag: ref.Tag, + Path: ref.Path, + DriverOperatorManaged: driverOperatorManaged, }) } diff --git a/pkg/bundler/deployer/helm/helm_test.go b/pkg/bundler/deployer/helm/helm_test.go index e76a5a9e8b..99b91d7e50 100644 --- a/pkg/bundler/deployer/helm/helm_test.go +++ b/pkg/bundler/deployer/helm/helm_test.go @@ -389,6 +389,302 @@ func TestGenerate_DeployScriptExecutable(t *testing.T) { } } +// TestGenerate_DeployScript_DRARestartGatedOnDriverOperatorManaged pins the +// fix for #2135's review follow-up: live cluster state alone (absent +// DaemonSet + no labeled node) cannot tell "driver is host-managed" apart +// from "driver is operator-managed but the migration gate hasn't converged +// yet" — the latter must block the DRA kubelet-plugin restart rather than +// running it unguarded, or it reproduces the invalid-CDI/ContainerCreating +// failure (#973). DriverOperatorManaged is derived at bundle time from +// gpu-operator's/gpu-operator-ocp's effective driver.enabled and threaded +// into the rendered script, so this only needs to check the generated +// text — no live cluster required. +func TestGenerate_DeployScript_DRARestartGatedOnDriverOperatorManaged(t *testing.T) { + recipeResult := func() *recipe.RecipeResult { + return &recipe.RecipeResult{ + Kind: "RecipeResult", + APIVersion: "aicr.run/v1alpha2", + Metadata: recipe.RecipeResultMetadata{Version: "v0.1.0"}, + Criteria: &recipe.Criteria{ + Service: "eks", + Accelerator: "h100", + Intent: "training", + }, + ComponentRefs: []recipe.ComponentRef{ + { + Name: "gpu-operator", + Namespace: "gpu-operator", + Chart: "gpu-operator", + Version: "v25.3.3", + Source: "https://helm.ngc.nvidia.com/nvidia", + }, + { + Name: "nvidia-dra-driver-gpu", + Namespace: "nvidia-dra-driver", + Chart: "nvidia-dra-driver-gpu", + Version: "0.4.1", + Source: "https://helm.ngc.nvidia.com/nvidia", + }, + }, + DeploymentOrder: []string{"gpu-operator", "nvidia-dra-driver-gpu"}, + } + } + + tests := []struct { + name string + recipeResultOCP bool // when true, uses OCP component names throughout instead of canonical + componentValues map[string]map[string]any + wantContains []string + wantNotContains []string + }{ + { + name: "operator-managed driver with neither signal observable skips wait like host-managed", + componentValues: map[string]map[string]any{ + "gpu-operator": { + "driver": map[string]any{"enabled": true}, + }, + "nvidia-dra-driver-gpu": {}, + }, + wantContains: []string{ + `SKIP_RESTART="false"`, + `driver DaemonSet not present and no nodes labeled nvidia.com/gpu.deploy.driver=true; skipping migration wait`, + `SKIP_RESTART=true`, + `if [[ -n "${DRA_DS}" && "${SKIP_RESTART}" != "true" ]]; then`, + `no nodes labeled nvidia.com/gpu.deploy.driver=true yet; skipping migration wait and DRA restart`, + `blocking the DRA plugin restart until the migration completes`, + }, + wantNotContains: []string{ + `blocking the DRA plugin restart until the driver rollout is detectable`, + }, + }, + { + name: "host-managed driver also skips the wait without blocking restart", + componentValues: map[string]map[string]any{ + "gpu-operator": { + "driver": map[string]any{"enabled": false}, + }, + "nvidia-dra-driver-gpu": {}, + }, + wantContains: []string{ + `driver DaemonSet not present and no nodes labeled nvidia.com/gpu.deploy.driver=true; skipping migration wait`, + `blocking the DRA plugin restart until the migration completes`, + }, + wantNotContains: []string{ + `blocking the DRA plugin restart until the driver rollout is detectable`, + }, + }, + { + name: "OCP DRA component renders its own guard and skips the wait the same way when neither signal is observable", + recipeResultOCP: true, + componentValues: map[string]map[string]any{ + "gpu-operator-ocp": { + "driver": map[string]any{"enabled": true}, + }, + "nvidia-dra-driver-gpu-ocp": {}, + }, + wantContains: []string{ + `if [[ "${name}" == "nvidia-dra-driver-gpu-ocp" ]]; then`, + `SKIP_RESTART="false"`, + `driver DaemonSet not present and no nodes labeled nvidia.com/gpu.deploy.driver=true; skipping migration wait`, + `SKIP_RESTART=true`, + `blocking the DRA plugin restart until the migration completes`, + }, + wantNotContains: []string{ + `if [[ "${name}" == "nvidia-dra-driver-gpu" ]]; then`, + `blocking the DRA plugin restart until the driver rollout is detectable`, + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + ctx := context.Background() + outputDir := t.TempDir() + + rr := recipeResult() + if tt.recipeResultOCP { + rr = &recipe.RecipeResult{ + Kind: "RecipeResult", + APIVersion: "aicr.run/v1alpha2", + Metadata: recipe.RecipeResultMetadata{Version: "v0.1.0"}, + Criteria: &recipe.Criteria{ + Service: "ocp", + Accelerator: "h100", + Intent: "training", + }, + ComponentRefs: []recipe.ComponentRef{ + { + Name: "gpu-operator-ocp", + Namespace: "gpu-operator", + Chart: "gpu-operator", + Version: "", + Source: "", + }, + { + Name: "nvidia-dra-driver-gpu-ocp", + Namespace: "nvidia-dra-driver", + Chart: "nvidia-dra-driver-gpu", + Version: "0.4.1", + Source: "https://helm.ngc.nvidia.com/nvidia", + }, + }, + DeploymentOrder: []string{"gpu-operator-ocp", "nvidia-dra-driver-gpu-ocp"}, + } + } + + g := &Generator{ + RecipeResult: rr, + ComponentValues: tt.componentValues, + Version: "v1.0.0", + } + + if _, err := g.Generate(ctx, outputDir); err != nil { + t.Fatalf("Generate failed: %v", err) + } + + content, err := os.ReadFile(filepath.Join(outputDir, "deploy.sh")) + if err != nil { + t.Fatalf("failed to read deploy.sh: %v", err) + } + script := string(content) + + for _, want := range tt.wantContains { + if !strings.Contains(script, want) { + t.Errorf("deploy.sh missing %q", want) + } + } + for _, notWant := range tt.wantNotContains { + if strings.Contains(script, notWant) { + t.Errorf("deploy.sh unexpectedly contains %q", notWant) + } + } + }) + } +} + +// TestGenerate_DeployScriptRendersValidBash pins the regression from PR +// #2346's review: the per-component DRA guard's closing `fi` was dropped in +// a restructure, and because the goldens compare rendered bytes rather than +// parsing them, that broke every bundle containing a DRA component (OCP or +// canonical) without failing any existing test. This renders deploy.sh for +// both the canonical and OCP recipe shapes and asserts the result is valid +// bash via `bash -n`, so a reintroduced syntax error fails CI directly +// instead of only showing up at actual deploy time. +func TestGenerate_DeployScriptRendersValidBash(t *testing.T) { + if _, err := exec.LookPath("bash"); err != nil { + t.Skip("bash not available in PATH; skipping syntax check") + } + + canonicalRecipeResult := &recipe.RecipeResult{ + Kind: "RecipeResult", + APIVersion: "aicr.run/v1alpha2", + Metadata: recipe.RecipeResultMetadata{Version: "v0.1.0"}, + Criteria: &recipe.Criteria{ + Service: "eks", + Accelerator: "h100", + Intent: "training", + }, + ComponentRefs: []recipe.ComponentRef{ + { + Name: "gpu-operator", + Namespace: "gpu-operator", + Chart: "gpu-operator", + Version: "v25.3.3", + Source: "https://helm.ngc.nvidia.com/nvidia", + }, + { + Name: "nvidia-dra-driver-gpu", + Namespace: "nvidia-dra-driver", + Chart: "nvidia-dra-driver-gpu", + Version: "0.4.1", + Source: "https://helm.ngc.nvidia.com/nvidia", + }, + }, + DeploymentOrder: []string{"gpu-operator", "nvidia-dra-driver-gpu"}, + } + + ocpRecipeResult := &recipe.RecipeResult{ + Kind: "RecipeResult", + APIVersion: "aicr.run/v1alpha2", + Metadata: recipe.RecipeResultMetadata{Version: "v0.1.0"}, + Criteria: &recipe.Criteria{ + Service: "ocp", + Accelerator: "h100", + Intent: "training", + }, + ComponentRefs: []recipe.ComponentRef{ + { + Name: "gpu-operator-ocp", + Namespace: "gpu-operator", + Chart: "gpu-operator", + }, + { + Name: "nvidia-dra-driver-gpu-ocp", + Namespace: "nvidia-dra-driver", + Chart: "nvidia-dra-driver-gpu", + Version: "0.4.1", + Source: "https://helm.ngc.nvidia.com/nvidia", + }, + }, + DeploymentOrder: []string{"gpu-operator-ocp", "nvidia-dra-driver-gpu-ocp"}, + } + + tests := []struct { + name string + recipeResult *recipe.RecipeResult + componentValues map[string]map[string]any + }{ + { + name: "canonical DRA component, operator-managed driver", + recipeResult: canonicalRecipeResult, + componentValues: map[string]map[string]any{ + "gpu-operator": {"driver": map[string]any{"enabled": true}}, + "nvidia-dra-driver-gpu": {}, + }, + }, + { + name: "canonical DRA component, host-managed driver", + recipeResult: canonicalRecipeResult, + componentValues: map[string]map[string]any{ + "gpu-operator": {"driver": map[string]any{"enabled": false}}, + "nvidia-dra-driver-gpu": {}, + }, + }, + { + name: "OCP DRA component, operator-managed driver", + recipeResult: ocpRecipeResult, + componentValues: map[string]map[string]any{ + "gpu-operator-ocp": {"driver": map[string]any{"enabled": true}}, + "nvidia-dra-driver-gpu-ocp": {}, + }, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + ctx := context.Background() + outputDir := t.TempDir() + + g := &Generator{ + RecipeResult: tt.recipeResult, + ComponentValues: tt.componentValues, + Version: "v1.0.0", + } + + if _, err := g.Generate(ctx, outputDir); err != nil { + t.Fatalf("Generate failed: %v", err) + } + + deployPath := filepath.Join(outputDir, "deploy.sh") + cmd := exec.Command("bash", "-n", deployPath) + out, err := cmd.CombinedOutput() + if err != nil { + t.Errorf("rendered deploy.sh failed bash -n syntax check: %v\noutput:\n%s", err, out) + } + }) + } +} + // --------------------------------------------------------------------------- // Property tests (helpers and data-shape preservation) // --------------------------------------------------------------------------- diff --git a/pkg/bundler/deployer/helm/templates/deploy.sh.tmpl b/pkg/bundler/deployer/helm/templates/deploy.sh.tmpl index 362d634430..566e366180 100644 --- a/pkg/bundler/deployer/helm/templates/deploy.sh.tmpl +++ b/pkg/bundler/deployer/helm/templates/deploy.sh.tmpl @@ -40,6 +40,7 @@ HELM_TIMEOUT="10m" NO_WAIT=false BEST_EFFORT=false FAILED_COMPONENTS="" +NEEDS_RETRY="" MAX_RETRIES=5 while [[ $# -gt 0 ]]; do @@ -521,8 +522,8 @@ for dir in "${SCRIPT_DIR}"/[0-9][0-9][0-9]-*/; do # --- post-install name-matched blocks --- {{- range .Components }} - {{- if eq .Name "nvidia-dra-driver-gpu" }} - if [[ "${name}" == "nvidia-dra-driver-gpu" ]]; then + {{- if or (eq .Name "nvidia-dra-driver-gpu") (eq .Name "nvidia-dra-driver-gpu-ocp") }} + if [[ "${name}" == "{{ .Name }}" ]]; then # gpu-operator's k8s-driver-manager reloads NVIDIA kernel modules # asynchronously per-node after `helm upgrade gpu-operator` returns. # If the DRA kubelet plugin pod re-rolls (via the chart's @@ -548,20 +549,51 @@ for dir in "${SCRIPT_DIR}"/[0-9][0-9][0-9]-*/; do # Waiting on every gpu.present=true node would block until the # 15-min timeout for any GPU node the operator deliberately # excludes. - DRIVER_DS_NS=$(kubectl ${KUBECTL_CONN[@]+"${KUBECTL_CONN[@]}"} get daemonset -A -o jsonpath='{.items[?(@.metadata.name=="nvidia-driver-daemonset")].metadata.namespace}' 2>/dev/null | awk '{print $1}') - if [[ -z "${DRIVER_DS_NS}" ]]; then - echo " gpu-operator nvidia-driver-daemonset not present (host-managed driver); skipping migration wait" - else - MANAGED_NODES=$(kubectl ${KUBECTL_CONN[@]+"${KUBECTL_CONN[@]}"} get nodes -l nvidia.com/gpu.deploy.driver=true -o name 2>/dev/null | wc -l | tr -d ' ') - if [[ "${MANAGED_NODES}" -gt 0 ]]; then - echo " Waiting for gpu-operator driver migration on ${MANAGED_NODES} managed GPU node(s) to reach upgrade-done (ns=${DRIVER_DS_NS})..." - if ! kubectl ${KUBECTL_CONN[@]+"${KUBECTL_CONN[@]}"} wait --for=jsonpath='{.metadata.labels.nvidia\.com/gpu-driver-upgrade-state}=upgrade-done' \ - nodes -l nvidia.com/gpu.deploy.driver=true --timeout=15m; then - echo " WARNING: not all managed GPU nodes reached upgrade-done within 15m; proceeding with restart anyway" - fi - else - echo " No nodes labeled nvidia.com/gpu.deploy.driver=true yet; skipping migration wait" + # Driver-DaemonSet lookup is name-prefix matched, not exact: the + # certified OpenShift build renders it as + # nvidia-driver-daemonset- via the Driver Toolkit, + # so an exact-name match silently fails open on OCP and reports + # "host-managed driver" even when gpu-operator-ocp owns it + # (driver.enabled: true). Node-label presence (gate 2) is checked + # unconditionally and takes priority over the DaemonSet lookup for + # the same reason: the operator applies that label itself, so it + # is a naming-agnostic signal of who owns the driver. + DRIVER_DS_NS=$(kubectl ${KUBECTL_CONN[@]+"${KUBECTL_CONN[@]}"} get daemonset -A --no-headers 2>/dev/null | awk '$2 ~ /^nvidia-driver-daemonset(-|$)/ {print $1; exit}') + MANAGED_NODES=$(kubectl ${KUBECTL_CONN[@]+"${KUBECTL_CONN[@]}"} get nodes -l nvidia.com/gpu.deploy.driver=true -o name 2>/dev/null | wc -l | tr -d ' ') + SKIP_RESTART="false" + if [[ -z "${DRIVER_DS_NS}" && "${MANAGED_NODES}" -eq 0 ]]; then + # Neither signal is observable: either the driver is genuinely + # host-managed, or the operator hasn't started its rollout yet (e.g. + # a fresh install, or any cluster where gpu-operator's driver + # DaemonSet has not been created by the time this step runs, such as + # simulated/KWOK clusters). There is no evidence a migration is + # actually underway, so it's safe to proceed as before: skip the + # wait and restart the DRA plugin normally. The fail-closed gate + # below only applies once a migration is actually in progress — the + # DaemonSet is present but no node is labeled yet, or an observed + # wait times out. + echo " gpu-operator driver DaemonSet not present and no nodes labeled nvidia.com/gpu.deploy.driver=true; skipping migration wait" + elif [[ "${MANAGED_NODES}" -gt 0 ]]; then + echo " Waiting for gpu-operator driver migration on ${MANAGED_NODES} managed GPU node(s) to reach upgrade-done (ns=${DRIVER_DS_NS:-})..." + if ! kubectl ${KUBECTL_CONN[@]+"${KUBECTL_CONN[@]}"} wait --for=jsonpath='{.metadata.labels.nvidia\.com/gpu-driver-upgrade-state}=upgrade-done' \ + nodes -l nvidia.com/gpu.deploy.driver=true --timeout=15m; then + # Fail closed, same reasoning as the unobservable-gate branch + # above: a migration wait that times out means some managed node + # has NOT reached upgrade-done, so restarting the DRA plugin now + # risks the exact mid-migration invalid-CDI/ContainerCreating + # failure (#973) this block exists to prevent. + echo " WARNING: not all managed GPU nodes reached upgrade-done within 15m; blocking the DRA plugin restart until the migration completes (retry the deploy)" + SKIP_RESTART=true + NEEDS_RETRY="${NEEDS_RETRY} {{ .Name }}" fi + else + # DaemonSet present but no node carries the label yet (e.g. still + # scheduling): fail closed and block the DRA restart below — + # restarting DRA before the driver migration gate is reached can + # leave DRA pods stuck against a mid-migration driver. + echo " gpu-operator driver DaemonSet present (ns=${DRIVER_DS_NS}) but no nodes labeled nvidia.com/gpu.deploy.driver=true yet; skipping migration wait and DRA restart" + SKIP_RESTART="true" + NEEDS_RETRY="${NEEDS_RETRY} {{ .Name }}" fi # Best-effort mitigation for kubelet DRA plugin registration drift. # After uninstall/reinstall, kubelet's fsnotify watcher may not detect new @@ -569,7 +601,7 @@ for dir in "${SCRIPT_DIR}"/[0-9][0-9][0-9]-*/; do # This does NOT fix cases where kubelet itself has lost registration state — # a node reboot is required for that. See docs/user/cli-reference.md. DRA_DS=$(kubectl ${KUBECTL_CONN[@]+"${KUBECTL_CONN[@]}"} get daemonset -n {{ .Namespace }} -o name 2>/dev/null | awk '/kubelet-plugin/{print; exit}' || true) - if [[ -n "${DRA_DS}" ]]; then + if [[ -n "${DRA_DS}" && "${SKIP_RESTART}" != "true" ]]; then echo " Restarting DRA kubelet plugin (${DRA_DS##*/}) to ensure registration..." if ! kubectl ${KUBECTL_CONN[@]+"${KUBECTL_CONN[@]}"} rollout restart "${DRA_DS}" -n {{ .Namespace }}; then echo " WARNING: failed to restart DRA kubelet plugin daemonset" @@ -579,6 +611,8 @@ for dir in "${SCRIPT_DIR}"/[0-9][0-9][0-9]-*/; do # DRA plugin socket), not a readiness convenience like --wait. echo " WARNING: DRA kubelet plugin rollout did not complete within 120s" fi + elif [[ "${SKIP_RESTART}" == "true" ]]; then + echo " Skipping DRA kubelet plugin restart (driver migration gate not yet reached)" else echo " WARNING: no DRA kubelet plugin daemonset found in {{ .Namespace }}" fi @@ -592,6 +626,16 @@ if [[ -n "${FAILED_COMPONENTS}" ]]; then else _ok "All components installed successfully." fi +if [[ -n "${NEEDS_RETRY}" ]]; then + # Distinct from FAILED_COMPONENTS/helm_failed so --best-effort semantics + # are unaffected: helm itself succeeded, but the DRA kubelet-plugin + # restart was deliberately withheld because the driver-migration gate + # was not yet observable when this deploy ran (see the WARNING above). + # Surfaced with its own non-zero exit below so automated callers (UAT, + # ArgoCD hooks, CI) get an actionable non-success signal instead of + # reading "All components installed successfully." as fully done. + _warn_line "DRA kubelet plugin restart blocked, retry needed for:${NEEDS_RETRY} — re-run this deploy once the operator has converged (driver DaemonSet present or a node carries nvidia.com/gpu.deploy.driver=true)." +fi echo echo "NOTE: The above status reflects Helm install and manifest apply results," echo "not whether the cluster is ready for GPU workloads. On fresh" @@ -602,3 +646,7 @@ echo " - GPU operator operand rollout (driver, toolkit, device-plugin DS)" echo " - NVIDIA DRA kubelet plugin registration" echo echo "See: https://github.com/NVIDIA/aicr/blob/main/docs/user/cli-reference.md#deploy-script-behavior-deploysh" + +if [[ -n "${NEEDS_RETRY}" ]]; then + exit 2 +fi diff --git a/pkg/bundler/deployer/helm/testdata/kai_scheduler_present/deploy.sh b/pkg/bundler/deployer/helm/testdata/kai_scheduler_present/deploy.sh index a45056c8e3..5c5adca8d4 100644 --- a/pkg/bundler/deployer/helm/testdata/kai_scheduler_present/deploy.sh +++ b/pkg/bundler/deployer/helm/testdata/kai_scheduler_present/deploy.sh @@ -40,6 +40,7 @@ HELM_TIMEOUT="10m" NO_WAIT=false BEST_EFFORT=false FAILED_COMPONENTS="" +NEEDS_RETRY="" MAX_RETRIES=5 while [[ $# -gt 0 ]]; do @@ -581,6 +582,16 @@ if [[ -n "${FAILED_COMPONENTS}" ]]; then else _ok "All components installed successfully." fi +if [[ -n "${NEEDS_RETRY}" ]]; then + # Distinct from FAILED_COMPONENTS/helm_failed so --best-effort semantics + # are unaffected: helm itself succeeded, but the DRA kubelet-plugin + # restart was deliberately withheld because the driver-migration gate + # was not yet observable when this deploy ran (see the WARNING above). + # Surfaced with its own non-zero exit below so automated callers (UAT, + # ArgoCD hooks, CI) get an actionable non-success signal instead of + # reading "All components installed successfully." as fully done. + _warn_line "DRA kubelet plugin restart blocked, retry needed for:${NEEDS_RETRY} — re-run this deploy once the operator has converged (driver DaemonSet present or a node carries nvidia.com/gpu.deploy.driver=true)." +fi echo echo "NOTE: The above status reflects Helm install and manifest apply results," echo "not whether the cluster is ready for GPU workloads. On fresh" @@ -591,3 +602,7 @@ echo " - GPU operator operand rollout (driver, toolkit, device-plugin DS)" echo " - NVIDIA DRA kubelet plugin registration" echo echo "See: https://github.com/NVIDIA/aicr/blob/main/docs/user/cli-reference.md#deploy-script-behavior-deploysh" + +if [[ -n "${NEEDS_RETRY}" ]]; then + exit 2 +fi diff --git a/pkg/bundler/deployer/helm/testdata/manifest_only/deploy.sh b/pkg/bundler/deployer/helm/testdata/manifest_only/deploy.sh index 8be8f4f076..cd9d4e747d 100644 --- a/pkg/bundler/deployer/helm/testdata/manifest_only/deploy.sh +++ b/pkg/bundler/deployer/helm/testdata/manifest_only/deploy.sh @@ -40,6 +40,7 @@ HELM_TIMEOUT="10m" NO_WAIT=false BEST_EFFORT=false FAILED_COMPONENTS="" +NEEDS_RETRY="" MAX_RETRIES=5 while [[ $# -gt 0 ]]; do @@ -581,6 +582,16 @@ if [[ -n "${FAILED_COMPONENTS}" ]]; then else _ok "All components installed successfully." fi +if [[ -n "${NEEDS_RETRY}" ]]; then + # Distinct from FAILED_COMPONENTS/helm_failed so --best-effort semantics + # are unaffected: helm itself succeeded, but the DRA kubelet-plugin + # restart was deliberately withheld because the driver-migration gate + # was not yet observable when this deploy ran (see the WARNING above). + # Surfaced with its own non-zero exit below so automated callers (UAT, + # ArgoCD hooks, CI) get an actionable non-success signal instead of + # reading "All components installed successfully." as fully done. + _warn_line "DRA kubelet plugin restart blocked, retry needed for:${NEEDS_RETRY} — re-run this deploy once the operator has converged (driver DaemonSet present or a node carries nvidia.com/gpu.deploy.driver=true)." +fi echo echo "NOTE: The above status reflects Helm install and manifest apply results," echo "not whether the cluster is ready for GPU workloads. On fresh" @@ -591,3 +602,7 @@ echo " - GPU operator operand rollout (driver, toolkit, device-plugin DS)" echo " - NVIDIA DRA kubelet plugin registration" echo echo "See: https://github.com/NVIDIA/aicr/blob/main/docs/user/cli-reference.md#deploy-script-behavior-deploysh" + +if [[ -n "${NEEDS_RETRY}" ]]; then + exit 2 +fi diff --git a/pkg/bundler/deployer/helm/testdata/mixed_gpu_operator/deploy.sh b/pkg/bundler/deployer/helm/testdata/mixed_gpu_operator/deploy.sh index cb3aade5a0..b53646d725 100644 --- a/pkg/bundler/deployer/helm/testdata/mixed_gpu_operator/deploy.sh +++ b/pkg/bundler/deployer/helm/testdata/mixed_gpu_operator/deploy.sh @@ -40,6 +40,7 @@ HELM_TIMEOUT="10m" NO_WAIT=false BEST_EFFORT=false FAILED_COMPONENTS="" +NEEDS_RETRY="" MAX_RETRIES=5 while [[ $# -gt 0 ]]; do @@ -581,6 +582,16 @@ if [[ -n "${FAILED_COMPONENTS}" ]]; then else _ok "All components installed successfully." fi +if [[ -n "${NEEDS_RETRY}" ]]; then + # Distinct from FAILED_COMPONENTS/helm_failed so --best-effort semantics + # are unaffected: helm itself succeeded, but the DRA kubelet-plugin + # restart was deliberately withheld because the driver-migration gate + # was not yet observable when this deploy ran (see the WARNING above). + # Surfaced with its own non-zero exit below so automated callers (UAT, + # ArgoCD hooks, CI) get an actionable non-success signal instead of + # reading "All components installed successfully." as fully done. + _warn_line "DRA kubelet plugin restart blocked, retry needed for:${NEEDS_RETRY} — re-run this deploy once the operator has converged (driver DaemonSet present or a node carries nvidia.com/gpu.deploy.driver=true)." +fi echo echo "NOTE: The above status reflects Helm install and manifest apply results," echo "not whether the cluster is ready for GPU workloads. On fresh" @@ -591,3 +602,7 @@ echo " - GPU operator operand rollout (driver, toolkit, device-plugin DS)" echo " - NVIDIA DRA kubelet plugin registration" echo echo "See: https://github.com/NVIDIA/aicr/blob/main/docs/user/cli-reference.md#deploy-script-behavior-deploysh" + +if [[ -n "${NEEDS_RETRY}" ]]; then + exit 2 +fi diff --git a/pkg/bundler/deployer/helm/testdata/mixed_with_pre/deploy.sh b/pkg/bundler/deployer/helm/testdata/mixed_with_pre/deploy.sh index cdc3c662de..95ae30603b 100644 --- a/pkg/bundler/deployer/helm/testdata/mixed_with_pre/deploy.sh +++ b/pkg/bundler/deployer/helm/testdata/mixed_with_pre/deploy.sh @@ -40,6 +40,7 @@ HELM_TIMEOUT="10m" NO_WAIT=false BEST_EFFORT=false FAILED_COMPONENTS="" +NEEDS_RETRY="" MAX_RETRIES=5 while [[ $# -gt 0 ]]; do @@ -581,6 +582,16 @@ if [[ -n "${FAILED_COMPONENTS}" ]]; then else _ok "All components installed successfully." fi +if [[ -n "${NEEDS_RETRY}" ]]; then + # Distinct from FAILED_COMPONENTS/helm_failed so --best-effort semantics + # are unaffected: helm itself succeeded, but the DRA kubelet-plugin + # restart was deliberately withheld because the driver-migration gate + # was not yet observable when this deploy ran (see the WARNING above). + # Surfaced with its own non-zero exit below so automated callers (UAT, + # ArgoCD hooks, CI) get an actionable non-success signal instead of + # reading "All components installed successfully." as fully done. + _warn_line "DRA kubelet plugin restart blocked, retry needed for:${NEEDS_RETRY} — re-run this deploy once the operator has converged (driver DaemonSet present or a node carries nvidia.com/gpu.deploy.driver=true)." +fi echo echo "NOTE: The above status reflects Helm install and manifest apply results," echo "not whether the cluster is ready for GPU workloads. On fresh" @@ -591,3 +602,7 @@ echo " - GPU operator operand rollout (driver, toolkit, device-plugin DS)" echo " - NVIDIA DRA kubelet plugin registration" echo echo "See: https://github.com/NVIDIA/aicr/blob/main/docs/user/cli-reference.md#deploy-script-behavior-deploysh" + +if [[ -n "${NEEDS_RETRY}" ]]; then + exit 2 +fi diff --git a/pkg/bundler/deployer/helm/testdata/nodewright_present/deploy.sh b/pkg/bundler/deployer/helm/testdata/nodewright_present/deploy.sh index 3119cf654c..c8dfb87d81 100644 --- a/pkg/bundler/deployer/helm/testdata/nodewright_present/deploy.sh +++ b/pkg/bundler/deployer/helm/testdata/nodewright_present/deploy.sh @@ -40,6 +40,7 @@ HELM_TIMEOUT="10m" NO_WAIT=false BEST_EFFORT=false FAILED_COMPONENTS="" +NEEDS_RETRY="" MAX_RETRIES=5 while [[ $# -gt 0 ]]; do @@ -683,6 +684,16 @@ if [[ -n "${FAILED_COMPONENTS}" ]]; then else _ok "All components installed successfully." fi +if [[ -n "${NEEDS_RETRY}" ]]; then + # Distinct from FAILED_COMPONENTS/helm_failed so --best-effort semantics + # are unaffected: helm itself succeeded, but the DRA kubelet-plugin + # restart was deliberately withheld because the driver-migration gate + # was not yet observable when this deploy ran (see the WARNING above). + # Surfaced with its own non-zero exit below so automated callers (UAT, + # ArgoCD hooks, CI) get an actionable non-success signal instead of + # reading "All components installed successfully." as fully done. + _warn_line "DRA kubelet plugin restart blocked, retry needed for:${NEEDS_RETRY} — re-run this deploy once the operator has converged (driver DaemonSet present or a node carries nvidia.com/gpu.deploy.driver=true)." +fi echo echo "NOTE: The above status reflects Helm install and manifest apply results," echo "not whether the cluster is ready for GPU workloads. On fresh" @@ -693,3 +704,7 @@ echo " - GPU operator operand rollout (driver, toolkit, device-plugin DS)" echo " - NVIDIA DRA kubelet plugin registration" echo echo "See: https://github.com/NVIDIA/aicr/blob/main/docs/user/cli-reference.md#deploy-script-behavior-deploysh" + +if [[ -n "${NEEDS_RETRY}" ]]; then + exit 2 +fi diff --git a/pkg/bundler/deployer/helm/testdata/owns_crds/deploy.sh b/pkg/bundler/deployer/helm/testdata/owns_crds/deploy.sh index 6175947dac..efbe7be163 100644 --- a/pkg/bundler/deployer/helm/testdata/owns_crds/deploy.sh +++ b/pkg/bundler/deployer/helm/testdata/owns_crds/deploy.sh @@ -40,6 +40,7 @@ HELM_TIMEOUT="10m" NO_WAIT=false BEST_EFFORT=false FAILED_COMPONENTS="" +NEEDS_RETRY="" MAX_RETRIES=5 while [[ $# -gt 0 ]]; do @@ -581,6 +582,16 @@ if [[ -n "${FAILED_COMPONENTS}" ]]; then else _ok "All components installed successfully." fi +if [[ -n "${NEEDS_RETRY}" ]]; then + # Distinct from FAILED_COMPONENTS/helm_failed so --best-effort semantics + # are unaffected: helm itself succeeded, but the DRA kubelet-plugin + # restart was deliberately withheld because the driver-migration gate + # was not yet observable when this deploy ran (see the WARNING above). + # Surfaced with its own non-zero exit below so automated callers (UAT, + # ArgoCD hooks, CI) get an actionable non-success signal instead of + # reading "All components installed successfully." as fully done. + _warn_line "DRA kubelet plugin restart blocked, retry needed for:${NEEDS_RETRY} — re-run this deploy once the operator has converged (driver DaemonSet present or a node carries nvidia.com/gpu.deploy.driver=true)." +fi echo echo "NOTE: The above status reflects Helm install and manifest apply results," echo "not whether the cluster is ready for GPU workloads. On fresh" @@ -591,3 +602,7 @@ echo " - GPU operator operand rollout (driver, toolkit, device-plugin DS)" echo " - NVIDIA DRA kubelet plugin registration" echo echo "See: https://github.com/NVIDIA/aicr/blob/main/docs/user/cli-reference.md#deploy-script-behavior-deploysh" + +if [[ -n "${NEEDS_RETRY}" ]]; then + exit 2 +fi diff --git a/pkg/bundler/deployer/helm/testdata/owns_crds_chart_override/deploy.sh b/pkg/bundler/deployer/helm/testdata/owns_crds_chart_override/deploy.sh index 6175947dac..efbe7be163 100644 --- a/pkg/bundler/deployer/helm/testdata/owns_crds_chart_override/deploy.sh +++ b/pkg/bundler/deployer/helm/testdata/owns_crds_chart_override/deploy.sh @@ -40,6 +40,7 @@ HELM_TIMEOUT="10m" NO_WAIT=false BEST_EFFORT=false FAILED_COMPONENTS="" +NEEDS_RETRY="" MAX_RETRIES=5 while [[ $# -gt 0 ]]; do @@ -581,6 +582,16 @@ if [[ -n "${FAILED_COMPONENTS}" ]]; then else _ok "All components installed successfully." fi +if [[ -n "${NEEDS_RETRY}" ]]; then + # Distinct from FAILED_COMPONENTS/helm_failed so --best-effort semantics + # are unaffected: helm itself succeeded, but the DRA kubelet-plugin + # restart was deliberately withheld because the driver-migration gate + # was not yet observable when this deploy ran (see the WARNING above). + # Surfaced with its own non-zero exit below so automated callers (UAT, + # ArgoCD hooks, CI) get an actionable non-success signal instead of + # reading "All components installed successfully." as fully done. + _warn_line "DRA kubelet plugin restart blocked, retry needed for:${NEEDS_RETRY} — re-run this deploy once the operator has converged (driver DaemonSet present or a node carries nvidia.com/gpu.deploy.driver=true)." +fi echo echo "NOTE: The above status reflects Helm install and manifest apply results," echo "not whether the cluster is ready for GPU workloads. On fresh" @@ -591,3 +602,7 @@ echo " - GPU operator operand rollout (driver, toolkit, device-plugin DS)" echo " - NVIDIA DRA kubelet plugin registration" echo echo "See: https://github.com/NVIDIA/aicr/blob/main/docs/user/cli-reference.md#deploy-script-behavior-deploysh" + +if [[ -n "${NEEDS_RETRY}" ]]; then + exit 2 +fi diff --git a/pkg/bundler/deployer/helm/testdata/readiness_gate/deploy.sh b/pkg/bundler/deployer/helm/testdata/readiness_gate/deploy.sh index f183be6fbb..25b264cc2c 100644 --- a/pkg/bundler/deployer/helm/testdata/readiness_gate/deploy.sh +++ b/pkg/bundler/deployer/helm/testdata/readiness_gate/deploy.sh @@ -40,6 +40,7 @@ HELM_TIMEOUT="10m" NO_WAIT=false BEST_EFFORT=false FAILED_COMPONENTS="" +NEEDS_RETRY="" MAX_RETRIES=5 while [[ $# -gt 0 ]]; do @@ -581,6 +582,16 @@ if [[ -n "${FAILED_COMPONENTS}" ]]; then else _ok "All components installed successfully." fi +if [[ -n "${NEEDS_RETRY}" ]]; then + # Distinct from FAILED_COMPONENTS/helm_failed so --best-effort semantics + # are unaffected: helm itself succeeded, but the DRA kubelet-plugin + # restart was deliberately withheld because the driver-migration gate + # was not yet observable when this deploy ran (see the WARNING above). + # Surfaced with its own non-zero exit below so automated callers (UAT, + # ArgoCD hooks, CI) get an actionable non-success signal instead of + # reading "All components installed successfully." as fully done. + _warn_line "DRA kubelet plugin restart blocked, retry needed for:${NEEDS_RETRY} — re-run this deploy once the operator has converged (driver DaemonSet present or a node carries nvidia.com/gpu.deploy.driver=true)." +fi echo echo "NOTE: The above status reflects Helm install and manifest apply results," echo "not whether the cluster is ready for GPU workloads. On fresh" @@ -591,3 +602,7 @@ echo " - GPU operator operand rollout (driver, toolkit, device-plugin DS)" echo " - NVIDIA DRA kubelet plugin registration" echo echo "See: https://github.com/NVIDIA/aicr/blob/main/docs/user/cli-reference.md#deploy-script-behavior-deploysh" + +if [[ -n "${NEEDS_RETRY}" ]]; then + exit 2 +fi diff --git a/pkg/bundler/deployer/helm/testdata/upstream_helm_only/deploy.sh b/pkg/bundler/deployer/helm/testdata/upstream_helm_only/deploy.sh index dee8f10cb1..69940e0e78 100644 --- a/pkg/bundler/deployer/helm/testdata/upstream_helm_only/deploy.sh +++ b/pkg/bundler/deployer/helm/testdata/upstream_helm_only/deploy.sh @@ -40,6 +40,7 @@ HELM_TIMEOUT="10m" NO_WAIT=false BEST_EFFORT=false FAILED_COMPONENTS="" +NEEDS_RETRY="" MAX_RETRIES=5 while [[ $# -gt 0 ]]; do @@ -581,6 +582,16 @@ if [[ -n "${FAILED_COMPONENTS}" ]]; then else _ok "All components installed successfully." fi +if [[ -n "${NEEDS_RETRY}" ]]; then + # Distinct from FAILED_COMPONENTS/helm_failed so --best-effort semantics + # are unaffected: helm itself succeeded, but the DRA kubelet-plugin + # restart was deliberately withheld because the driver-migration gate + # was not yet observable when this deploy ran (see the WARNING above). + # Surfaced with its own non-zero exit below so automated callers (UAT, + # ArgoCD hooks, CI) get an actionable non-success signal instead of + # reading "All components installed successfully." as fully done. + _warn_line "DRA kubelet plugin restart blocked, retry needed for:${NEEDS_RETRY} — re-run this deploy once the operator has converged (driver DaemonSet present or a node carries nvidia.com/gpu.deploy.driver=true)." +fi echo echo "NOTE: The above status reflects Helm install and manifest apply results," echo "not whether the cluster is ready for GPU workloads. On fresh" @@ -591,3 +602,7 @@ echo " - GPU operator operand rollout (driver, toolkit, device-plugin DS)" echo " - NVIDIA DRA kubelet plugin registration" echo echo "See: https://github.com/NVIDIA/aicr/blob/main/docs/user/cli-reference.md#deploy-script-behavior-deploysh" + +if [[ -n "${NEEDS_RETRY}" ]]; then + exit 2 +fi diff --git a/pkg/bundler/testdata/stock_render_golden.yaml b/pkg/bundler/testdata/stock_render_golden.yaml index e60a527022..c338037c8f 100644 --- a/pkg/bundler/testdata/stock_render_golden.yaml +++ b/pkg/bundler/testdata/stock_render_golden.yaml @@ -84,8 +84,8 @@ a100-aks-ubuntu-training-kubeflow: README.md: c59cdceff0e9959d375d3c4e6725eb4df735e8979ccefb79f9bd6d75a60935b5 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: b8757b034c4af4ea4260caf4c3311f2b805a435155323196e8fa30394aff7e1c - checksums.txt: 8f99b2583e2dcf38e6b997da8a5e39d7d3cf86bf410511e2cf58bb76d78f0e02 - deploy.sh: 578de8756d8c7fd3f4cb092af39adaeefd0e3df50e5d248893b42bd1029f4532 + checksums.txt: ec32478c3cc6aa7cba668bb8bfd28e435c79b8b84ebbf8baa5ffb335c91a4052 + deploy.sh: 31482e8aec932784f31615f39314fe85d918687672813eded0a7f0c45f560ffc recipe.yaml: 31de158c477835adc03f540590bfa4e53e366df642faad27e257830a2701ea92 a100-any: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -136,8 +136,8 @@ a100-any: README.md: 9ca6e45a84a07008ffa7cdfa3c51bcaf13a3e992b8069cf6327882d32d8ce806 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: ad4f9557a3b9797a2ffb23f6873c1935e283599143f741c823d83798a200132c - checksums.txt: 467ea1ae826459ae2b5e392009d9da59d4e7379587476f76c00b8faa6b932d6a - deploy.sh: ea70804cf5d8c0bb716ccd59ea4a31a1037672453aa11e2e2bfbd20daf55d901 + checksums.txt: 156c09b7a037318d9bfa8324920b780cc1f86d2b554b6bf4fcd96c1b995448ca + deploy.sh: caebd2467959960c9272cf95670c6b4d85dcf60412f3cfbc74fddd26d62f3550 recipe.yaml: 67ce39f490a229ef7eaf80e15316edd3f19dd077f23b39c12f63a9bb13f0970d a100-eks-ubuntu-training-kubeflow: 001-aws-ebs-csi-driver/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -210,8 +210,8 @@ a100-eks-ubuntu-training-kubeflow: README.md: c2fb67edc7eca8b6988382312a0aefef868066029c2a3cb7e1e8e59cc6566373 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 7d9b1fdd481ae80b6fe9aa8079bbb3604ec1ce9c11002c985f186f91c5659f01 - checksums.txt: 21aa52ca92161d183c6eb6d4d7dd08894aa73299e4d9ec46c1a564c308f4d8bb - deploy.sh: 5b2acd33c456f23e8d1f4d574f94b9ba2bce52147876f165a54e3395e55de09b + checksums.txt: 9dd8a7a08686bef6f22d74ef0783cbddb3ff0aed59a37f45c9b73643d34a9a82 + deploy.sh: 2b5bd0d390a1d9f1d3426916fcd05e957d1aa2397a4821730f1f5ea9ab5b022a recipe.yaml: 9c868eb55b01a98513d2e341d7242e9d0b9b1b5969206c6415d4cecdc2abfa30 a100-gke-cos-training-kubeflow: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -285,8 +285,8 @@ a100-gke-cos-training-kubeflow: README.md: e21a4b1766a05e41473e85415eddc8c3c72fac6d5eb3851c5da02c946cb2dfed UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: e9231504d14acfc247811326fdf17edb8022c38a238c3113ceb4792fffb79ea3 - checksums.txt: f565ee5440332343e50bed71c77ab2b534401260273563b2eacfa13a2a143312 - deploy.sh: 2b0cd173c38bb27ee4cdb7d6c4d1d94d523b56fed94d6889cacd342b0b5a0dfd + checksums.txt: ae5b4211bcf2fed66d9f36709e02c8c5eca22193f247ac20651c53182a6f9404 + deploy.sh: 6eb8bceaaadd38d35e3445f4814dc8b391a6d840f4615cbe706414b08be1efa4 recipe.yaml: 9ba7f55ca95b2b646138e1364ed8cce1ef085c141b6092e9acec1b2e3496a427 a100-oke-ubuntu-training-kubeflow: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -346,8 +346,8 @@ a100-oke-ubuntu-training-kubeflow: README.md: bb7d5d0404f2dc415d3559a7d1e8f1b2288fd6684a76b4758278020a65736424 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 4e734937ad34f76384eaaa848b203ae607df88334ea31ddbe8e489b0c3f52fbe - checksums.txt: a3f4866685d7443a8519e2245e2b7ea3bf693732d0b51eaea77a673d75860684 - deploy.sh: 8ebf4bd801a6d8fd6cba8979dc354e9aeeb8d5cfccbee9e6e7262417b175c6f4 + checksums.txt: 6d66f88a8cf77f0cd59f893744a3f31bae1450375d4fc68c28ca866ea047ea9e + deploy.sh: 0cd9c43e70d4b627cab9b30c96262a95c91b3c0b77aa6dcc7296b09303a80d04 recipe.yaml: 36381c1944cba39d385c0ab8faf89e15bb5cf8c320e384196b56b68e5854d516 aks-ubuntu: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -414,8 +414,8 @@ aks-ubuntu: README.md: 47818d601139c841e2b931899d4b1579acbcf5bcfc6bb6bb064c4f83a0ffeb90 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 052dbe04fd5e7a135bc6e6df50aa560516183938c65b3f3dac132cb143fe0c6f - checksums.txt: 10c3564d5c3e5d3fa518ea444e506669ab8f9b377cd8edd0b58b61f100a5a326 - deploy.sh: 1d5c9d627548eabf502d0ca6ecacddce17627e87ebdbd095333df77e8cdad2df + checksums.txt: 41215af4a0b6bf1f35158cdcca5eea4d888b34b25e73533d6dba0fcc92f7ead7 + deploy.sh: f4d6f3956edc61baa1e242a31ec53890870525ad1e7e83a351d3350d4131a879 recipe.yaml: 1c4ba900739f229c70ecacaf72ff790e7a554880b226655c32f0284c4076eee7 b200-any: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -466,8 +466,8 @@ b200-any: README.md: cf4d7e0922c8821fcac68e37444beff835581194b11b0a11c2ddae7be474c80b UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 678737aa5616f511a57776d0d357073c70ca2bfeb83d29dbd317ef82964aa039 - checksums.txt: 02e447100626305aa65ee3ee6c3502966dfaced063ae8b3c5ecf9b225227af5b - deploy.sh: ea70804cf5d8c0bb716ccd59ea4a31a1037672453aa11e2e2bfbd20daf55d901 + checksums.txt: 085c58eee060014083dd686b77fa4438c3b1e9a4511962526db96a7d45e5663d + deploy.sh: caebd2467959960c9272cf95670c6b4d85dcf60412f3cfbc74fddd26d62f3550 recipe.yaml: fec141debfbc24173ddcc5e05f3678d11fc936f6092f6f4d24dab9893218fd6a b200-gke-cos-inference-dynamo: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -559,8 +559,8 @@ b200-gke-cos-inference-dynamo: README.md: 5703af26dd5ce9982bd5949da24bef64cd5da4ffe5f2f73d3978e52a48c3ae25 UPGRADING.md: d27ec18c5dbe463bc50dfdde771821d0a717e350be36a89f87297ada543ba81c bundle-info.yaml: 4a472490a261894fc68b4a5b67f482ed3fea56d64abe501043f3c342b2c3fb2a - checksums.txt: ed5a971694464d9108a69611b1005affe45648b91a73443894312d10ae6600fd - deploy.sh: 62b3ab91044223177c74d3aee7241f4aeae64e6bdb31481f7da642dbf4556e9a + checksums.txt: 52b0b4fc967b57ddeee187e625a0f9143fc9746e2f4240131fec4ecc4068a1f7 + deploy.sh: f9ae0011756a0dc2acd2b02a3682da24ed0902a2a4d1bed1d8c9059dcdc14dca recipe.yaml: e9db226befc5008771e8e32c6ea5fd998501fd003e502005fbaf322e13fd196a b200-gke-cos-training-kubeflow: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -634,8 +634,8 @@ b200-gke-cos-training-kubeflow: README.md: 35ee900dad87a1d7a0ea53ca7f9422f602fbcdfff8884086977f599620f7a686 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 910e640f7e7442cb6e8537a11d86b7307096564795f52461a21341f2a175415f - checksums.txt: b061ce0b6c797ec2d3b140b32b32778ccb27c0464e9783d9064809c4b125c283 - deploy.sh: 2b0cd173c38bb27ee4cdb7d6c4d1d94d523b56fed94d6889cacd342b0b5a0dfd + checksums.txt: 02fc68dfa9a710febbfca1252b2d47280626fdf8eb363e0d8f9730ab82f28387 + deploy.sh: 6eb8bceaaadd38d35e3445f4814dc8b391a6d840f4615cbe706414b08be1efa4 recipe.yaml: fbffcad95923a554f5c6ca4d9833cdec38be012ca146a8b1c265487ca531a915 bcm-inference: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -710,8 +710,8 @@ bcm-inference: README.md: 8cf61b75e92017ac8ad83f09b58b101e0506883369e7b31155bc4c4b61492dff UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 62faebc11da34bc66e56ba5e8a6d493ee7300c1b4a49ea60685281d3496d2781 - checksums.txt: 2e783d80116d6eecada96939f62a220361697b37875ad4ec13b85b10406c8c91 - deploy.sh: 0e40b3c44ac83591b443c18609eb82ddbdb3f5d410cb8eaff31a5533f9348d6e + checksums.txt: 1af30661730428134d1c0f56f2f4dc0a55a9798942605504b30cc7b4a357c991 + deploy.sh: 13ff76c4d3e5bf550ec6ba53a61c37fce6acf16a5b71ec4475ae9cdf966f9229 recipe.yaml: 8953e615d4d84bd0a860b1e8aea84c00fa1edbf64028c3c1d7db37807357dfa7 eks-ubuntu: 001-aws-ebs-csi-driver/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -770,8 +770,8 @@ eks-ubuntu: README.md: f73fdb0f62f176b0cc9b735d3ded5bfeca4539daa2c9e97167b3da7055d573d5 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 866da806c394a3ed15397e03afc376e6be7012328971ce88ce84039ff478d58a - checksums.txt: ab0b7daaa49c1cfcee81c3b107e08aaadc2758eefeb578aef0f80d45a3f96566 - deploy.sh: 2fd097b74541a045e1a07751042298f7d295f8071c60bbd9be76029f9ac34959 + checksums.txt: e4f41818650bcaeb4fe26d7acdccff6a569a9b1918c2ebb1f775c80208b7a459 + deploy.sh: 9b9c2cbefccc5f2db6423c9c2fd1ea0576b1dd44349535d2630950128562f55e recipe.yaml: afeb379a07ad5bcf87007bb99a0aab395959e4a61ef92ed5b46092730f5cdd6e gb200-any: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -822,8 +822,8 @@ gb200-any: README.md: 7b43d55c87cb3312c54d1a52f083ccec7e9f9182ae49aeb2790159acd927eda3 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 42c13acd0638fc09d941d4cafd2b4e9c4a99f618e1ebf427aaf3956b686a722a - checksums.txt: 7264b1623b4a9482ba592c63ce9ecec40949f0bd3c0f717ef5cd58be8e83747c - deploy.sh: ea70804cf5d8c0bb716ccd59ea4a31a1037672453aa11e2e2bfbd20daf55d901 + checksums.txt: f2ba14784f7e062d28299aac30565c7c1cc38644bc60ace2dd55a5650d0bc7c9 + deploy.sh: caebd2467959960c9272cf95670c6b4d85dcf60412f3cfbc74fddd26d62f3550 recipe.yaml: 0f7d86bd9c759548abf944cab2d28b2e2e9416670ed6c10c240dca14967474a1 gb200-eks-ubuntu-inference-dynamo: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -919,8 +919,8 @@ gb200-eks-ubuntu-inference-dynamo: README.md: 7d812ea5dcf6bd12128bdf94ebd300306eccb146814de5ff0a3998833f151a0e UPGRADING.md: d27ec18c5dbe463bc50dfdde771821d0a717e350be36a89f87297ada543ba81c bundle-info.yaml: 441da0c5ae961abe27cba8c07bcc12d51fc9869b1159ace189e2ca7a8729a63f - checksums.txt: 043f454553883ffa5adf31a96535ef71f59e6d0acc2f3d867ad6b6684661a0fb - deploy.sh: 7a305353d31349834943e912d9c671f1e44a3b9ec914d8862017d7e56dee9a96 + checksums.txt: 84776b8271ac8673ad48f76e9ab9789511a0d9ad88067bc8c28bc37e6eae9d2e + deploy.sh: 739eecc9fba2e7d0a20615a1c65f774050df2dec22e21bd6f314a573bd315aa1 recipe.yaml: 99016877d03d3f8badd84640e47e5f1b4dff71d45e2fef3da6c2769064805175 gb200-eks-ubuntu-training-kubeflow: 001-aws-ebs-csi-driver/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -998,8 +998,8 @@ gb200-eks-ubuntu-training-kubeflow: README.md: 43af1d677a691a4e8131ef67fa50deb7921d9fedbe0aae6a941a990cb2fe0a44 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 2321ed37c5fa7add302079333778e6c7eba3d6f152d59603d69389a0617e5857 - checksums.txt: bdfb35553255177fe8a7b05916068c0a9f64390da25603acbb50ba76b5e006d2 - deploy.sh: 5b2acd33c456f23e8d1f4d574f94b9ba2bce52147876f165a54e3395e55de09b + checksums.txt: 6cd8e760922a6dd116d942e28e81d650e35086aeb7b6eef41cc8d5f4ec0ed04f + deploy.sh: 2b5bd0d390a1d9f1d3426916fcd05e957d1aa2397a4821730f1f5ea9ab5b022a recipe.yaml: 4fe166f7e3189c7a81eee37ba61056a17f722ed79501c376c5ec4dd65014f956 gb200-eks-ubuntu-training-slurm: 001-aws-ebs-csi-driver/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -1086,8 +1086,8 @@ gb200-eks-ubuntu-training-slurm: README.md: cd743008c0568e64613a3621351bdaa2504bf9f365c7702065f27bc064ebc1fa UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 9905ee9008d86ae458d7f11ace060abf90c8c9199e9bbbbe9b82b72bf31f730a - checksums.txt: 3f05044bd2bbd368e50d408f415f27f4df38aff8cf29231ca9eb766760043697 - deploy.sh: 3bf1475ef39efa155a2cdff6c99ffeba55d931adf38871f8bc74467845584eae + checksums.txt: 00caaedbac3e49e9722d973224957f564f307889b84a24494fef3b63fdf2bc67 + deploy.sh: c922970112eaaa2cf22192c2d74e0978b74fbe5c7eb7d18cf67e689c843f7d2f recipe.yaml: e0d3e329aff972080d3f76604ad6d18971fe685190ceb7e6753e06473640bf8e gb200-gke-cos-inference-dynamo: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -1190,8 +1190,8 @@ gb200-gke-cos-inference-dynamo: README.md: 0f883b8cf45ab50d6e046d71ca483e90a9133c0a344b384328d4bbd9a9ae0da5 UPGRADING.md: d27ec18c5dbe463bc50dfdde771821d0a717e350be36a89f87297ada543ba81c bundle-info.yaml: eac5f1b4ca396596923b1a92a4cdd3769773fa41d47168210946188eec55e03e - checksums.txt: d88489ee25679ea0091e9b4d87c11c731a21f1bf6a3adc3e8abc86bf9022f989 - deploy.sh: c27046f23c79d92f9cdcc9725b9ef25978758818975807ba376dc5e6f9fda214 + checksums.txt: c9815b6652bfb1b9e3be984ac655c7ebf2cd19f182d4892567fe61469767f279 + deploy.sh: 66b402290c1e082ed0753fa4c361f71d352252f8ec9fdfaefd8ad5da1708559b recipe.yaml: 9cac37bba449e4309e64f0926ea3b48296310c0dd6f5002760697b9621c33576 gb200-gke-cos-training-kubeflow: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -1272,8 +1272,8 @@ gb200-gke-cos-training-kubeflow: README.md: 9f72f7c976888ac46558f3a26424107314227af1a20ecc23e28fd9ea212af575 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 8c5b3d8800aeeff6bc4ddf607b0ff9ef49b20eb95159a3641226ffea26063677 - checksums.txt: 7d5b7b2fc417a8d1f6cd1137cece802fbc5a44b39e0f8a54342f05313f2ad652 - deploy.sh: 7d1e5335991335299686c41cb44d9609d4ed3b5aa59506a38dd64949a2553c54 + checksums.txt: c004932a691826ce1c5ab7bd2ff7214e837893b0db67e212198838709e9c057f + deploy.sh: 7484a478b2bf9686300d92b9aedefe090624f48c57b77432ba4ed5df73641e37 recipe.yaml: c2fbdbbe659d84d23e21f9da54825a1ec83e37139e26feb9c2592fef5af42ed0 gb200-gke-cos-training-slurm: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -1362,8 +1362,8 @@ gb200-gke-cos-training-slurm: README.md: 77013cccf74249773878eb911d4af2c86db0de11f2a0a186dcc14a61d73792fd UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: d2be07c4bd13419a9c0f7e71a0e3abcad9329f9a51a7aed9765d6997f690e0de - checksums.txt: 161f7c24292ba1caef47d090a14a43dd1c5bf2067571453ee1ca1f576ee29223 - deploy.sh: ef955644171896a717b59c2582b0fbef5b6cab0d2f67c89924656b3aee005c73 + checksums.txt: b5153020e05e5422d17379867df6b0d66a08c54824341a2d5de147fdefa9ed95 + deploy.sh: 4a3da78e97fe16bae346878ec541650f20de876b3e05039420acdd87f4612c02 recipe.yaml: 0cf270c23abb5c143982232de0b6115c9502e29392acfaab8de683edc8398f18 gb200-oke-ubuntu-inference-dynamo: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -1441,8 +1441,8 @@ gb200-oke-ubuntu-inference-dynamo: README.md: 6242517b67ea219b0acbe46ed9c8f1569eba35255f82532c25446030a7eea85d UPGRADING.md: d27ec18c5dbe463bc50dfdde771821d0a717e350be36a89f87297ada543ba81c bundle-info.yaml: f369d8bf5acb55840e9179d05230aabd42c25e87b718adc49fb54adb531f8cdb - checksums.txt: 1aa423dd42ad6833b8de79a9b2788f570cea2a16c9647ea2980f1be3fe59aa0c - deploy.sh: 6c9407779b830bf0997644af6ced980afe20317b576ffc024df41eb719b8d949 + checksums.txt: f5a83d32233269f637c9b58c391f8166dd6bc9acde657489bbd509428a92df19 + deploy.sh: 370fc74831dad01fbe7bbe0749c890bf6c1c4c9be06056703d2bdd6e553f0b2c recipe.yaml: e3a1bb44834e5b242e5e6b2b7be04c29b4d9b106e292d90ce47d52a1ae582243 gb200-oke-ubuntu-training-kubeflow: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -1516,8 +1516,8 @@ gb200-oke-ubuntu-training-kubeflow: README.md: eeb441534e6dac00ae0c4262b729b23be1148be66f9602bd85cf82aba605fe7f UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: bd28c90092298d3cb527df155396b31adb575214fd90e8e381ef888cdbdf400d - checksums.txt: 25a4383f42ce818758b79c89bbe9d63b71dd3c1adf9096ea8c445fe4e816d50a - deploy.sh: 83fa4613633138d6b344997098cc0b2f77082f6fc1ba1cca6cef729b1a316cd3 + checksums.txt: 1664161995e1914cc6ae77149dd076ef3b30924a19c61bd07911981fd976141f + deploy.sh: 262206678bd377d4dc89fd1a8afb2a9f0e9622b661ed46a7ff7d6ba47cc85265 recipe.yaml: 533380c82da0b4dbda99f3b5d744057c82fb7ccdd0a470b05c65bc29671059b9 gb300-any: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -1568,8 +1568,8 @@ gb300-any: README.md: 3ef7ab2055833c836afe728785e2d1c477e22def8a9087b5ac64f60e5593eb6c UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 812ec5168e9be198310acde61285650d43964ac9bc0bb3dc7d5c336e87eb658f - checksums.txt: 3cda8f96de5948165e5ac5d771f3a265af09da16f8cdc4e8dd5770e60c506198 - deploy.sh: ea70804cf5d8c0bb716ccd59ea4a31a1037672453aa11e2e2bfbd20daf55d901 + checksums.txt: 35907b345012385562144388af33097bdefcd847ee31591b48fab3dd921d330b + deploy.sh: caebd2467959960c9272cf95670c6b4d85dcf60412f3cfbc74fddd26d62f3550 recipe.yaml: b77d5c4ef355d75f0c9b4999f0d57b3d5ff426f78cab48012d0eb1815a037394 gb300-eks-ubuntu-inference-dynamo: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -1665,8 +1665,8 @@ gb300-eks-ubuntu-inference-dynamo: README.md: a511d3e61c3f67f76df9ab26a2a7f5d4cc5343993dc96374b2de60d3b1ea3e33 UPGRADING.md: d27ec18c5dbe463bc50dfdde771821d0a717e350be36a89f87297ada543ba81c bundle-info.yaml: 616d719558836e403f23137366f5a49ee112bc02ca55410f0376dabf126dbb49 - checksums.txt: 5c25e708e55ffbd9d51e814c765f7c89996d51631261156517d157255878ad1d - deploy.sh: 7a305353d31349834943e912d9c671f1e44a3b9ec914d8862017d7e56dee9a96 + checksums.txt: 86e07b6c6359ab8bd664800299637d9c0beaef54701ddc06f9accc1e0518ac47 + deploy.sh: 739eecc9fba2e7d0a20615a1c65f774050df2dec22e21bd6f314a573bd315aa1 recipe.yaml: 4aa6f0d612c3e364603fa25a8e3e3527c203f81736e5cecc718828a9c2c98d60 gb300-eks-ubuntu-training-kubeflow: 001-aws-ebs-csi-driver/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -1744,8 +1744,8 @@ gb300-eks-ubuntu-training-kubeflow: README.md: 6e06975e1ff735dd78f986ffce4e16742dfd8b114f8427bfb149570312465292 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 1d30e69141e264a6cffbbfc0199baec719fe3694f29e1af99c55a222eb52af73 - checksums.txt: c6d0235d2666496aed8fd54cb2919df529e455483a31113a4697031b5d4f3927 - deploy.sh: 5b2acd33c456f23e8d1f4d574f94b9ba2bce52147876f165a54e3395e55de09b + checksums.txt: a6ed3aa022393190e91f9ce549e9ea3bde1070b44c0ab451fbd66b731af4685e + deploy.sh: 2b5bd0d390a1d9f1d3426916fcd05e957d1aa2397a4821730f1f5ea9ab5b022a recipe.yaml: e017dd3bf35f86f1e94d252ae6dd3ad95726071726d3c99530a6899e3a412b66 gb300-eks-ubuntu-training-slurm: 001-aws-ebs-csi-driver/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -1832,8 +1832,8 @@ gb300-eks-ubuntu-training-slurm: README.md: 7ec5178e47106fc572219ead6a6e1a8c91177ab6c48edeedfc37f699a64b3067 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: e6ec5054f3016420f86e78c3473cc0786b01d7e7437389b28700226bf968ffa3 - checksums.txt: 5e0f22b09af08c40532c3f8dfb61b88c654cd4cb22c62935c6c76b85f5310b60 - deploy.sh: 3bf1475ef39efa155a2cdff6c99ffeba55d931adf38871f8bc74467845584eae + checksums.txt: 57ac3ad40cadaacae73861765fdf639093a77a502a75c609139b022465c676a8 + deploy.sh: c922970112eaaa2cf22192c2d74e0978b74fbe5c7eb7d18cf67e689c843f7d2f recipe.yaml: c711fe632ed29fa531c91eed1bf43f0b04c985168f3f07df85030a7cce6b4f08 gb300-generic-ubuntu-training: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -1898,8 +1898,8 @@ gb300-generic-ubuntu-training: README.md: 220ee0f077375a93701aa20ad10fc325d5a2be5c6cb5ec0d56310de4c5a7f2d6 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 01c4c17f8cfb3a1bc3b5739ec81c5922e2ade03f6115fb447ebb4361741f10b9 - checksums.txt: 261a4892c756ba9b26d130309d520c04b0d86b59411d321fc544fe4125b073d1 - deploy.sh: a4407de071637cc7bcd2e67d43c3d51bc767423e6200467a435e1898f67b0627 + checksums.txt: 9b856eb9ca2b8cea9cb3981eb35e6441af8685371623f84c79c7b81f4142a019 + deploy.sh: e3370b30f74fc04aeaafddd24ebb6178b6cdd45899091fa8a5e5c6f4b4d60b5e recipe.yaml: 4d69760384d18a3291c6dc0a5a20f8e0e3ecff9fe814160861beda5c0b984376 h100-aks-ubuntu-inference-dynamo: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -1998,8 +1998,8 @@ h100-aks-ubuntu-inference-dynamo: README.md: 8ffffac5dffea4fc5493a5d2144bcc4f819304c71f4b0e753b2dfcc53d88c224 UPGRADING.md: d27ec18c5dbe463bc50dfdde771821d0a717e350be36a89f87297ada543ba81c bundle-info.yaml: 5dceea0a0f68d6247bdd12ae8a53cacc738a855579503a44235d58c27ed68e6e - checksums.txt: 9e134c598f981da5d91058a3bd4834ea965b0decc37aec4b003cc5ccf8ac2b6a - deploy.sh: 8fe244abd3dd37ba57ffd0c551fe092c6c2c7fe033799da927170855d0104583 + checksums.txt: 1227514acbe024d179c739ff0a97221c87bb7f60b8002437d702870925a15e0f + deploy.sh: 1cc5a5735dc63ca50b6c2d8c0b724c1ed1635f20dd53b3e60d26f973996acceb recipe.yaml: 91155a3b152a7efc742884990e01075f0a0a42379f75fa92d08dee0e9b89fab2 h100-aks-ubuntu-training-kubeflow: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -2080,8 +2080,8 @@ h100-aks-ubuntu-training-kubeflow: README.md: 62edf7818cb89a0d4d40a9c6ad451df97b7664ea01b23bee64403d3c47a8b406 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: a4bc974b5a744dbd68200b2c840b8c21feefefd36afd3a676fc9419878383579 - checksums.txt: 3912eec5a32e70a5b21298ac12a2272303845b8453913e31e184010b2a2e1941 - deploy.sh: 578de8756d8c7fd3f4cb092af39adaeefd0e3df50e5d248893b42bd1029f4532 + checksums.txt: 39bab440be619bafa91dcf62dc3fdd7dae8fd5efb2eba5153b025e88892b780c + deploy.sh: 31482e8aec932784f31615f39314fe85d918687672813eded0a7f0c45f560ffc recipe.yaml: ab935ef631c32798d5680c1c2e2c770cb4eff3534e19a87efd41a544e5d86ef2 h100-aks-ubuntu-training-slurm: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -2170,8 +2170,8 @@ h100-aks-ubuntu-training-slurm: README.md: af73cb4c0d378ce9f520a8d8b8eca27b9bab58e563867071147136cb6030ab36 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 916c3459e3eb96504a80c75513b36d27705f5cf0ad3d7134baed19967cf76e4c - checksums.txt: c51103017f51436cc2a1f168845508c481520316ba160c4da29ce8dfa9c3b05c - deploy.sh: 9b4d0619afbaa1425c76e4035df5a7a41bb8c10f35fd0892b7e552a3bc72d66d + checksums.txt: e17eea74a6db3765c36a1c12c8fe06039537bdd88c4d93f8fd9a7d56b02a65a7 + deploy.sh: ec2c0385303670cfb2dff6d610fc9bef61dc2ae0cc07ace890223a294d4e5035 recipe.yaml: e1fbd4c2236e574456d9ee80cfd4bb61dec01b33e293edfc858de6bb87d4bebb h100-any: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -2222,8 +2222,8 @@ h100-any: README.md: c9c17733e6903e0c364122d669a050b87e7a5ed1aab5ec25cf28806e7216ebfd UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 5ae21a9ee0637c62e0c4b4bb907a63fdccf6f999ccccd67264c22f48cd423746 - checksums.txt: 54927b14ee2d9b142f34cb8a446eaf763fb2b2f59a3bf7f708e8563fe1579709 - deploy.sh: ea70804cf5d8c0bb716ccd59ea4a31a1037672453aa11e2e2bfbd20daf55d901 + checksums.txt: 0e5f77a991414bf1e5f1e13a752bbbca54bfc1926f8272f3ffb232cdb1d6aa8c + deploy.sh: caebd2467959960c9272cf95670c6b4d85dcf60412f3cfbc74fddd26d62f3550 recipe.yaml: 2e0d488d24e987f116cb37ceb665dfef4df87ecc0018a0777337cf9e052ffd4c h100-bcm-ubuntu-training-kubeflow: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -2288,8 +2288,8 @@ h100-bcm-ubuntu-training-kubeflow: README.md: b948a5b9a3a419f5208de7f79d9ae81a73cbdcdced0e23b67b14ea9784207d5a UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 81601dd90ae55fa09407c5bef15c53280feb7852fabd037406171e594902f262 - checksums.txt: c10e4f159d384f1502e338566072573ba55f10c9d83de8f0945966a5beea8777 - deploy.sh: 942b1194469de9dfc1e8b55420980e47fb7cafa917ef628d29af8817dfbd9d79 + checksums.txt: 9ba376faf88dfa457c2afe2df9474908e0d042f61c4a722bd3da3e4f1a361434 + deploy.sh: 6e75d5ca28d404548d2899fac36da6bb7bd27c75b3bf0342aa47f684d1ce1e89 recipe.yaml: 438a504c7b781356509eae930652b5db7413d8fd188b0f553fc6cb7a063c561c h100-eks-ubuntu-inference-dynamo: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -2380,8 +2380,8 @@ h100-eks-ubuntu-inference-dynamo: README.md: af590efaa7ca1d2760af737fa973ee54e7cfc273529f94b7fb08170f1ce70c4c UPGRADING.md: d27ec18c5dbe463bc50dfdde771821d0a717e350be36a89f87297ada543ba81c bundle-info.yaml: 6fffe7c7863286b5589c835af53914ca05afccd20596e185cd799e18980c1fb2 - checksums.txt: 131b82d42932e40287d6a61e17196aed80d169241505d7ca6382bef58ba1299c - deploy.sh: 7a305353d31349834943e912d9c671f1e44a3b9ec914d8862017d7e56dee9a96 + checksums.txt: a5adbf116d9e849151b6b2bb4c21c21f3367126b1524e7df76c84ea91eeacecb + deploy.sh: 739eecc9fba2e7d0a20615a1c65f774050df2dec22e21bd6f314a573bd315aa1 recipe.yaml: 8162d79cda156236cb4078ad712d927ec88f954fadf3d2c60edbb77deae6e688 h100-eks-ubuntu-inference-nim: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -2468,8 +2468,8 @@ h100-eks-ubuntu-inference-nim: README.md: b65217878628ebcba2f35f8006ee24ccf705434f0a794d749edb384c367f25da UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 7c006a7f70694e0539479a2f137418328306c4e73d4ab016f7b52f4e86c7c013 - checksums.txt: 558b8bdc0a4bb5510976c4a2dbf9fe8bdc7d2b1b79de4fec92385478c8786c6f - deploy.sh: 038181d6586cb77d4ab6094a7ddd94e9db4b9c73160da7aa5f13d7389f5166e0 + checksums.txt: 4f8b14fc803041fe537050c45f414c73c809c4827bf1321fc574cc519b02c1a3 + deploy.sh: 06f0da8008542e252d8be432f4b7b7cbe7ee14639cdd3ff382ef4a0e13ecd688 recipe.yaml: 231dd0717f699e2db1b686a097dac7b637783b50410d3361cb73dd73bad3b8a1 h100-eks-ubuntu-training-kubeflow: 001-aws-ebs-csi-driver/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -2542,8 +2542,8 @@ h100-eks-ubuntu-training-kubeflow: README.md: 4993cbc00e197f69746b9d2688e3e3f3afffaced1053d030997a7b71564a014b UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 9e27217baf0328163053412a9ec906133043721593b1123c1fecd6ce33e58881 - checksums.txt: df791557f6a8717b23a1fabc9b65f862ee31f8ddb5385a21c57fe3bd119fa5ad - deploy.sh: 5b2acd33c456f23e8d1f4d574f94b9ba2bce52147876f165a54e3395e55de09b + checksums.txt: 2582c134d1944849fb04ec264a704c4cb14010d902d8672674d2dab6d4338a4e + deploy.sh: 2b5bd0d390a1d9f1d3426916fcd05e957d1aa2397a4821730f1f5ea9ab5b022a recipe.yaml: df5b2f780667e7d3d5987ad6aa76cc470355b84800dbf5d1f6ea950bdad4b027 h100-eks-ubuntu-training-slurm: 001-aws-ebs-csi-driver/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -2624,8 +2624,8 @@ h100-eks-ubuntu-training-slurm: README.md: 7a2c2b2631c320f17839dc9f06dc5a1b206e05619ed84c151b08079b848d9d7e UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 0365df9937fab0521f97faabb994fd738a5fa3168a29d3cb163f6e78f267c169 - checksums.txt: d5de8542d78559f6e1b286da28e5d26cb80baf3c929725cb441e45f86d009208 - deploy.sh: 3bf1475ef39efa155a2cdff6c99ffeba55d931adf38871f8bc74467845584eae + checksums.txt: 9dfe05f9f1b268464ee6da2fa64e0d3cca05caad75bd88ddaf5d3cf5c24fedf4 + deploy.sh: c922970112eaaa2cf22192c2d74e0978b74fbe5c7eb7d18cf67e689c843f7d2f recipe.yaml: bbbac2a32846177e3d7d9ce205fa12ffb50e95d4556de7651a9540c93eddec9d h100-gke-cos-inference-dynamo: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -2717,8 +2717,8 @@ h100-gke-cos-inference-dynamo: README.md: 0816d3e944d74c218fcafa501024c8fe7cabd7a6834db08f58094e4d62e3f7e2 UPGRADING.md: d27ec18c5dbe463bc50dfdde771821d0a717e350be36a89f87297ada543ba81c bundle-info.yaml: aee2fd047e41143b9e2d7b18704875d0f40a77bfaedfa17cdbea2a399eb28d50 - checksums.txt: 86221bd5d33eeb9847be560a82a0f252ddf4701445df45b517921c6e95ccdc02 - deploy.sh: 62b3ab91044223177c74d3aee7241f4aeae64e6bdb31481f7da642dbf4556e9a + checksums.txt: 0ff99ec091dc0a3970f5bac4df1ab08ff75c9a3ab484caf04193fdae4fc7150f + deploy.sh: f9ae0011756a0dc2acd2b02a3682da24ed0902a2a4d1bed1d8c9059dcdc14dca recipe.yaml: b1134834618a00ab52c1450588179c15228c089715e79d1b45fe325c4e7d2165 h100-gke-cos-training-kubeflow: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -2799,8 +2799,8 @@ h100-gke-cos-training-kubeflow: README.md: 720cf3194fa2d25b9f7c4f751a09225a836b8b3fbaba9129a12636043eae6db6 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 0eccb5f38da442d462ab76de778688a955067a424fe503b3d308e9fc4e7edfac - checksums.txt: 1e243c5ae49b1809d026f35ff2d236b56961796a5383bc942785524c153f1367 - deploy.sh: fdb5b08014a89062f4b304cce41d91410d4a690f2c5a723cdb4b08028097d92d + checksums.txt: 06402865c89864b5bc872c905e97af383e7c8ecdcb7a83e3a5b54f227cdc1350 + deploy.sh: 15312acd64caca0bcbfe29b348b9796ef7ab2ffb188b9da7028cad022b5cb037 recipe.yaml: 778f4f38cf2bda39db6c70b3f77bc9cef2a3459980789175b7a25dbce77657ce h100-gke-cos-training-slurm: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -2892,8 +2892,8 @@ h100-gke-cos-training-slurm: README.md: 9ffec06121fc2e14304bef09888c1c94eba7c70d3b696eb31013fb412cde9ff1 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 5f04c03c47b0a86fede255c77ef02c137cb9bc4240fb53f3848ec1ac8a718f15 - checksums.txt: a3645ca6b803c8244e84003102be37331aff09ab76ab3ec160357d6e72872003 - deploy.sh: 26b5bbeba08ab0830741d513d86fb9b24a204d48968b95c88fb5eadc9fe67cba + checksums.txt: 0382b6454980a02b6ac23921c502287bbdc576a970bb056085167e0fb9ba7c5c + deploy.sh: d3fdfa0890a45fa68af2f3f6296fbd7d2b3c634d08b72a2e7bb4db6e50671482 recipe.yaml: 5337077a0192ff123d1987616ce4d8963679cf2b8328b12f291ffe4488ccee1d h100-kind-inference-dynamo: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -2975,8 +2975,8 @@ h100-kind-inference-dynamo: README.md: ab9eb42e715795deb1f8ed539ee4a8a7a47772beb80fee3fe11ee30c298f664e UPGRADING.md: d27ec18c5dbe463bc50dfdde771821d0a717e350be36a89f87297ada543ba81c bundle-info.yaml: 3b1624a1f0e84de11b081953baa4ba2d15d386ffeb6424227b04e2e4832d0bc8 - checksums.txt: e4543bd82ce721f567fab1b1f2bf6c41df514eda1762ba3d00758a07e0612c6a - deploy.sh: 18827849e66bfdd7d103aeab9f6374a7018500f6b57a3e3a68fc789bf40ec5b4 + checksums.txt: 95bf44eb39552d853b0ff0c0430ab7c1eb353f3790bc486c48431058752c2950 + deploy.sh: 8284cacb245e963bd0861271d8d43c557d09a6e3751dda924445b5125eeef125 recipe.yaml: 8547fce134e27a15ffc64eda6bb79398e2d7f5a23a3d9a6d98829654909383c0 h100-kind-training-kubeflow: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -3040,8 +3040,8 @@ h100-kind-training-kubeflow: README.md: bf5195c04e85d5dfab9768cf5ee5be7a8f5370a0272557dfa1782b84ef65bf94 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 5d8907879020323f5cff687edb1fdfe98b6c26f7166a62517c4cfeabde9ffd46 - checksums.txt: 54e0f136ade2c2af0a3290ff89ed1fdc7017365b41b1929a2c19064259f1d03d - deploy.sh: 83fa4613633138d6b344997098cc0b2f77082f6fc1ba1cca6cef729b1a316cd3 + checksums.txt: 4e0f674079dae904d94db784713e010a3a73d6254cea488b957840d90459323e + deploy.sh: 262206678bd377d4dc89fd1a8afb2a9f0e9622b661ed46a7ff7d6ba47cc85265 recipe.yaml: e25f52da6bcb17d0d098ca48b6eaf83fde3d5030554f934870e6f4e252ba59c6 h100-kind-training-slurm: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -3117,8 +3117,8 @@ h100-kind-training-slurm: README.md: 9e75e15c8f365e9420fa72d715062ee018329edd610099734138753fa2c23146 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: a93cb191edf0c4e9a18dbf165832a44a2e6bddb9dcb3a724ca21741884347fb0 - checksums.txt: 1bd46eb553298268a254522feb66ad2761a31339363457422ebe562fa9637ec8 - deploy.sh: a9268f292354d1038ae6c7401d39c2fd190aafd3e2f0f7a3820ee740e685255b + checksums.txt: 05ea62ed11da5d3e3a3807f796e0673507070d15ed09143cea9063f8805a5461 + deploy.sh: 9e51c4667eb2874d80d5e9b168d1bdee0a2dddd8cbf593e31e87ae26cd574acc recipe.yaml: 385bac65d13d3a64c7de7f8f300fe0d6811914d2e65206072d3bf8606d642fc5 h200-any: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -3169,8 +3169,8 @@ h200-any: README.md: b7491c1e6319aa4363c353d8759e50cf4551e524917cca63f6298707bccc11f0 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 514837028e4e12b7355ad78abc8ec7e8375263349badd4b06e9e21513c83188b - checksums.txt: 64c76eb41171549ee79d9fffdbcf3e3a0038c11de90c0baf3f229da0bf2c602f - deploy.sh: ea70804cf5d8c0bb716ccd59ea4a31a1037672453aa11e2e2bfbd20daf55d901 + checksums.txt: 42bb4e6a80e7e52e52c8130b0eb1af792fd2f6af692083d4c9888ad95d819744 + deploy.sh: caebd2467959960c9272cf95670c6b4d85dcf60412f3cfbc74fddd26d62f3550 recipe.yaml: d64ffd6cf71899c321776323620b7599b641ed934fb21a32d09e9427645f8af7 h200-eks-inference: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -3253,8 +3253,8 @@ h200-eks-inference: README.md: e5bfff423abbef94b1784b7d6557f04ff53715ba2f86dd90bbf04824ed0c5d45 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: c04691fb7b2a50ff090141c7e9349eb1d9599c28ea05f763695d12b8f1ff27b0 - checksums.txt: fe62743ace57e986333b62630bb8ac7295ef869b376317adbda7296a4b67c661 - deploy.sh: e149421402115939cf19fcde8aabd5386462e32d4c422b8bea3cbef73a9633e9 + checksums.txt: 5659a389ee7750c930a397bf526c0a874d3ff835c1a2b94cfac1777378bc367c + deploy.sh: 5c75ba098f339bab6507c5c95ade89161cdc5270e93621ff439e56753fd2d661 recipe.yaml: 1c68a28cff255ff927ffb43aa0f016f768980cad3d87020dc8914eea7b85d8f6 h200-eks-training-kubeflow: 001-aws-ebs-csi-driver/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -3327,8 +3327,8 @@ h200-eks-training-kubeflow: README.md: f759b48a96d5d551eef4943dea02a35745980642d599167e354cf4a2e9fade24 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 7da9c570b61105e630eb03ef7b7950d432c04dcdfa14d3f5f4b03d490fead274 - checksums.txt: 21c88c5219c1b5958231732984aef152305ceb7256781d67cae0dbd5651d5c3b - deploy.sh: 5b2acd33c456f23e8d1f4d574f94b9ba2bce52147876f165a54e3395e55de09b + checksums.txt: a3057f1d554d793b541653ddeda17837d3c62c91105cf439b77c3e9a1c44c81d + deploy.sh: 2b5bd0d390a1d9f1d3426916fcd05e957d1aa2397a4821730f1f5ea9ab5b022a recipe.yaml: a306d46ce21d404b811e5c4d896e9b08cb066b939b323f32b4dcd31ab1408b1e h200-k0s-ubuntu-training: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -3379,8 +3379,8 @@ h200-k0s-ubuntu-training: README.md: 108a41246f6812282fff9686c82a9dd8b5298e424d98034adbae462b0838381f UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: e1e276ddcf93d161ac458f60ccfd053898a0c447345a8eb4601d3455cb0777b7 - checksums.txt: a454a3eb0889c1841f8e4f96646542123d9ecdf09e735fe8f9aa242d91bcbd6d - deploy.sh: ea70804cf5d8c0bb716ccd59ea4a31a1037672453aa11e2e2bfbd20daf55d901 + checksums.txt: 84d9c1adedc2116cadcfdf08b8ef4e0bcbdad411349f358bc310f489df562862 + deploy.sh: caebd2467959960c9272cf95670c6b4d85dcf60412f3cfbc74fddd26d62f3550 recipe.yaml: a832789daf7f642c3859e7b4250c48775b4909ad6ebab1e28fe35da6b039d7fd l40-any: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -3431,8 +3431,8 @@ l40-any: README.md: e37d0a2de4211cdab60ee38ed8238d108299b48a24b3486597aa513d3e7dd6a0 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 8bf89ddb5d08fa480102fd5b94cfc503136b6e32dc9112670d34e88be35e512b - checksums.txt: 73064a3827687cadf478e01f205199ab2cebec4b6ba2a87639bafcfea5f165ac - deploy.sh: ea70804cf5d8c0bb716ccd59ea4a31a1037672453aa11e2e2bfbd20daf55d901 + checksums.txt: 7a16b4f770ebadc6e0ed625d5c272c234102915310ff5a2fbb905aa2e9ae08e3 + deploy.sh: caebd2467959960c9272cf95670c6b4d85dcf60412f3cfbc74fddd26d62f3550 recipe.yaml: c345318265e9156cb210d47e6481ce4d106c458072658fcb215b7e1a8af3775a l40s-any: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -3483,8 +3483,8 @@ l40s-any: README.md: 845cef314022dcb5eeca65b550104b817046d099937cfadcf711cd300485e0cb UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 5efaf3183b4bba279b84f3d5100d73d558eade1efb9d0881f7a591f8cb1ed11e - checksums.txt: c5c6510bb8087ccfbb89061a0010095fd8f3e444544cada5374ff3bdbb5a58a3 - deploy.sh: ea70804cf5d8c0bb716ccd59ea4a31a1037672453aa11e2e2bfbd20daf55d901 + checksums.txt: 475b55aa26a210cca80331235f6f24bbadb870022e3e737ed4def1597171632b + deploy.sh: caebd2467959960c9272cf95670c6b4d85dcf60412f3cfbc74fddd26d62f3550 recipe.yaml: eb12a1fc5f7544b4d1ef4a6fe3519bcc40b86d89f5ab45a8d3b5b250e772cbb3 l40s-oke-inference: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -3554,8 +3554,8 @@ l40s-oke-inference: README.md: 1d8048fb45f088cce8cc0a6c19fdecdb9a6c7ae544b2710a80e19899d4c3db00 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: b8aa88d25e096ab8e45f06784d774d41f409528a2da6ce49a4302d321715b6a6 - checksums.txt: 8d73a9fb56f2641c36011e5bb1f8a5b211cf0d5a171625192a466aa31118247b - deploy.sh: f5e448117266a5341c361f7468d2b4b0460be002aeb1dbe550a89e24a4689240 + checksums.txt: 6acd1c011bd526112ac51d581904f9fdc9894505fff5f902a99d02a741f04268 + deploy.sh: 75767d3a8f6260fecfb8ea17fe3ab4855c81e2cba101ee4f5b90330f59cf3f4a recipe.yaml: 71537d8ac7d90713062ea4fa3eede3a657948522970e42ba8fc957807c4793b9 l40s-oke-training-kubeflow: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -3624,8 +3624,8 @@ l40s-oke-training-kubeflow: README.md: c28b433be87a8cdb602607be8189d565ba7b30f21848ccf2d6c59514499f5869 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: a66fb62309da40de238b26ca0fd93c63fa29ad7b2ab7aafd715cf3c14d32ccd4 - checksums.txt: 75f1eb31fe54571404147cb4f096ff87d174f68711d5c29743ad4dd124457993 - deploy.sh: 83fa4613633138d6b344997098cc0b2f77082f6fc1ba1cca6cef729b1a316cd3 + checksums.txt: 7aa0a787254e6b24de7b0d30c04a8afdc03ff9e454e36f98c002221a1b0ba133 + deploy.sh: 262206678bd377d4dc89fd1a8afb2a9f0e9622b661ed46a7ff7d6ba47cc85265 recipe.yaml: 452940cced0ab1323605544acbedd1e6b79cc5e388aeb3c57334d8fe5ca49db5 monitoring-hpa: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -3676,8 +3676,8 @@ monitoring-hpa: README.md: 3d899242cdd95f1b3eecae98bf74d77ec53b68fc8a6e6569114e49a055444ab0 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: a20831b311ca30312daed6109af9023f25ddbd565139a531c4ceeab0e16d93c7 - checksums.txt: fdf572c819d4b79490af4fd28f3766f63e4d282940dc3797db53eba2e4ec9ee4 - deploy.sh: ea70804cf5d8c0bb716ccd59ea4a31a1037672453aa11e2e2bfbd20daf55d901 + checksums.txt: 183cc876d679307e9aa8fbac76479fe4cdd728cd869e882ea3dfaf7aa7c7d21e + deploy.sh: caebd2467959960c9272cf95670c6b4d85dcf60412f3cfbc74fddd26d62f3550 recipe.yaml: 24ab49f88b9ed8b9cb3b1efdf9e9b609a1e6cbfbaa83686dbc5a29fb0f59a1dc ocp-inference-nim: 001-nfd-ocp-olm/Chart.yaml: af13850294d68e9730e11e9b15c329a1935088f4e07c0ffb65a6d8a5915b78e0 @@ -3738,11 +3738,11 @@ ocp-inference-nim: 011-nvidia-dra-driver-gpu-ocp-pre/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 011-nvidia-dra-driver-gpu-ocp-pre/install.sh: f7cf353a0afe5f92a56f6b0f26b5aed4210d53fc31ca501ddfd64be3b943720c 011-nvidia-dra-driver-gpu-ocp-pre/templates/scc-rolebinding.yaml: 9d02598d6dbb28e2e6749bb6fda329fc288e35d07ee5d11ffb44a4fc5a5227a0 - 011-nvidia-dra-driver-gpu-ocp-pre/values.yaml: 539fc42e25e4c707a6031d17e76202375aa535492f4ebc22e2d7a7ee6458c110 + 011-nvidia-dra-driver-gpu-ocp-pre/values.yaml: 60ec4253bd0348cf2e33c27e957ca81128e554ff1b38eb05d61f490304f4f7df 012-nvidia-dra-driver-gpu-ocp/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 012-nvidia-dra-driver-gpu-ocp/install.sh: 1c7faa915324579119dc908dfdc4e1b058da0751be08b8652b9a3c771be60055 012-nvidia-dra-driver-gpu-ocp/upstream.env: 2b81f487848614f9b0167e48f2f8c65a057048f4c1e7b99b34a1ddc899c90ff0 - 012-nvidia-dra-driver-gpu-ocp/values.yaml: 539fc42e25e4c707a6031d17e76202375aa535492f4ebc22e2d7a7ee6458c110 + 012-nvidia-dra-driver-gpu-ocp/values.yaml: 60ec4253bd0348cf2e33c27e957ca81128e554ff1b38eb05d61f490304f4f7df 013-prometheus-adapter-ocp-pre/Chart.yaml: ff9ca8d7b6965365a6e738d6a3adfa2193133f5361e16a905e84867903e3b477 013-prometheus-adapter-ocp-pre/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 013-prometheus-adapter-ocp-pre/install.sh: c60ca285984ddd9c7eb0a15af2679207efe611be7232abb5aa37435ffd6920c9 @@ -3760,8 +3760,8 @@ ocp-inference-nim: 015-prometheus-adapter-ocp-post/values.yaml: 4b424b0499653496b19f128ee249ae6ec84d113c87786af22a73315056e558c9 README.md: d5734584eeb2d506ee617b3d6daab561771b3dbe289d5ff300456a25781b3a54 bundle-info.yaml: 867dcd5523ffcac5983948d5951a92566cef561b4f4712333375f303616f5394 - checksums.txt: 676bbdad74afef55707680cac36443a1f1b8a02c3fb1c9d7b2f80bcee8892348 - deploy.sh: 24cf97959d5e99177292e20d3494d221cbfe6b75a6e9420954eb2d255b5a41e9 + checksums.txt: 0ae95bc6e0bf106825cbaae9bb9ccf79a28ed0917c3bfddbefce38c4e68086b7 + deploy.sh: 5d2da2041eede71b64bd56f1ec2657dea93536e5804d8c3064eb72ef1621deb6 recipe.yaml: b4b646b39d7705617b4bbf4c272eac65a02e2781ca5da593c6ab2715c95fb54a ocp-training: 001-nfd-ocp-olm/Chart.yaml: af13850294d68e9730e11e9b15c329a1935088f4e07c0ffb65a6d8a5915b78e0 @@ -3813,11 +3813,11 @@ ocp-training: 009-nvidia-dra-driver-gpu-ocp-pre/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 009-nvidia-dra-driver-gpu-ocp-pre/install.sh: f7cf353a0afe5f92a56f6b0f26b5aed4210d53fc31ca501ddfd64be3b943720c 009-nvidia-dra-driver-gpu-ocp-pre/templates/scc-rolebinding.yaml: 9d02598d6dbb28e2e6749bb6fda329fc288e35d07ee5d11ffb44a4fc5a5227a0 - 009-nvidia-dra-driver-gpu-ocp-pre/values.yaml: 539fc42e25e4c707a6031d17e76202375aa535492f4ebc22e2d7a7ee6458c110 + 009-nvidia-dra-driver-gpu-ocp-pre/values.yaml: 60ec4253bd0348cf2e33c27e957ca81128e554ff1b38eb05d61f490304f4f7df 010-nvidia-dra-driver-gpu-ocp/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 010-nvidia-dra-driver-gpu-ocp/install.sh: 1c7faa915324579119dc908dfdc4e1b058da0751be08b8652b9a3c771be60055 010-nvidia-dra-driver-gpu-ocp/upstream.env: 2b81f487848614f9b0167e48f2f8c65a057048f4c1e7b99b34a1ddc899c90ff0 - 010-nvidia-dra-driver-gpu-ocp/values.yaml: 539fc42e25e4c707a6031d17e76202375aa535492f4ebc22e2d7a7ee6458c110 + 010-nvidia-dra-driver-gpu-ocp/values.yaml: 60ec4253bd0348cf2e33c27e957ca81128e554ff1b38eb05d61f490304f4f7df 011-prometheus-adapter-ocp-pre/Chart.yaml: ff9ca8d7b6965365a6e738d6a3adfa2193133f5361e16a905e84867903e3b477 011-prometheus-adapter-ocp-pre/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 011-prometheus-adapter-ocp-pre/install.sh: c60ca285984ddd9c7eb0a15af2679207efe611be7232abb5aa37435ffd6920c9 @@ -3835,8 +3835,8 @@ ocp-training: 013-prometheus-adapter-ocp-post/values.yaml: 4b424b0499653496b19f128ee249ae6ec84d113c87786af22a73315056e558c9 README.md: 85af5865e9553756b98eb2e59149c8f1cee5069810174582c89d448f699e30c1 bundle-info.yaml: 5896360f8e8f55c3284676d13799c7c13c8aff8709e2c9ada7932d7b4caf3d5b - checksums.txt: d025c06d7a4b12a0d6826c4b90b9506dc5bbba0c76eaff19f47e2a6a6a118ae7 - deploy.sh: 57db5c15a9f544ccf2f58e52e0de14b990672c2cc82af04cb6f12b83a46b94f1 + checksums.txt: 7a46cc674c6668b3a4cc8799ab4f5fb366f9bb5c370b3083939074f6437ffce6 + deploy.sh: a721f970aafef8502b7f05abd098c0c60e392af08d9bdabf0c80d8d0112176aa recipe.yaml: be82f14aba09224250a0cdfba0af51aadb4719d346f35bd3ac14b8e0724de710 rtx-pro-6000-any: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -3887,8 +3887,8 @@ rtx-pro-6000-any: README.md: 34f29655c21d9afcc0522ad8e0fa0ae3dc6fa4d6437f71bdb9b8bdde8411b2bb UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: eefab36fd4795211a800193574a9257548c39c449b56050f1e7cecbb88c88b57 - checksums.txt: 819541252dbbc3ff0214e2ec32f9011e502d7b35aab28fc2f575db569530c2ec - deploy.sh: ea70804cf5d8c0bb716ccd59ea4a31a1037672453aa11e2e2bfbd20daf55d901 + checksums.txt: 4bc4a82e01b3de5bbb42594c43480de3b039f3ff8c2d2db058c40ff3791cecfc + deploy.sh: caebd2467959960c9272cf95670c6b4d85dcf60412f3cfbc74fddd26d62f3550 recipe.yaml: 32b380c2cd938bb330972b086538fbb275b350af83d0ca97fe5813a4f2db6dbd rtx-pro-6000-eks-ubuntu-inference-dynamo: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -3979,8 +3979,8 @@ rtx-pro-6000-eks-ubuntu-inference-dynamo: README.md: 4bfdda900cf0d4b4c257a275350390c565a2c25cb57884c5fc72a3ab6d5556c2 UPGRADING.md: d27ec18c5dbe463bc50dfdde771821d0a717e350be36a89f87297ada543ba81c bundle-info.yaml: 6f400007a5f7ee13ac7e76b43954dd1ad8110b46b96d3fcf2fccf70b272d435a - checksums.txt: b78183464390709453062df610bd30d1c28ce314103e9dc34a26a3222e158821 - deploy.sh: 7a305353d31349834943e912d9c671f1e44a3b9ec914d8862017d7e56dee9a96 + checksums.txt: 5f3793c76e667e24ba3cb5ea6151b5c0de24802b4cb60186627bc91d8799f148 + deploy.sh: 739eecc9fba2e7d0a20615a1c65f774050df2dec22e21bd6f314a573bd315aa1 recipe.yaml: 4db38877fde150873df88175456a635e4838c46139cffc06a1ea6c91dd8afd52 rtx-pro-6000-eks-ubuntu-inference-nim: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -4067,8 +4067,8 @@ rtx-pro-6000-eks-ubuntu-inference-nim: README.md: 661463f3a9658588dddef8934f676c7bd67f5cd785be1e5ae643d296924e8b79 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 6204dcb7f80da4d3a4527810e266489ac872aa415044cddcb824e30ca8425a4d - checksums.txt: fc08342e3dcbd31e8cf84aa563dd3efbc1b4baf49403c0917092cd9afc99603f - deploy.sh: 038181d6586cb77d4ab6094a7ddd94e9db4b9c73160da7aa5f13d7389f5166e0 + checksums.txt: ce93f94824a327e9c814304b2cca7f0ee6d0fa2db9d329548bf0349493f2337f + deploy.sh: 06f0da8008542e252d8be432f4b7b7cbe7ee14639cdd3ff382ef4a0e13ecd688 recipe.yaml: 21c7bf1e9aab60439f6dc9d84178045fc74187d858eec18bc82be7cbaf5a5855 rtx-pro-6000-eks-ubuntu-training-kubeflow: 001-aws-ebs-csi-driver/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -4141,8 +4141,8 @@ rtx-pro-6000-eks-ubuntu-training-kubeflow: README.md: a0984a093f9000cad08c5ecbbd7088ecd4b54ab3fb35f401d0e2ca531ed2edfc UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: b049397b787a94d946349995bb2f1002fefd2552e479bef7ec019b5a2eca2be6 - checksums.txt: 6843362c2aac4e41c84b499d141c691a5d0e6fba6b3e31f103be97b4a33be15d - deploy.sh: 5b2acd33c456f23e8d1f4d574f94b9ba2bce52147876f165a54e3395e55de09b + checksums.txt: 36a4811417bd99b0aa38c4432482f54402403de9343c93ac0b5eb07132313375 + deploy.sh: 2b5bd0d390a1d9f1d3426916fcd05e957d1aa2397a4821730f1f5ea9ab5b022a recipe.yaml: 20afdda5c4355624e51050d78973b4f016e83cb9bce0b4bbf25e83c1e5f2514e rtx-pro-6000-lke-ubuntu-inference: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -4212,8 +4212,8 @@ rtx-pro-6000-lke-ubuntu-inference: README.md: 2bf65a4c61206273ba292c2cbaea904102487ef248887eeba67114d9a13cdca3 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: f2b647f4d5a54286cb6a4fb7860e51e49231535ac15b6536dbd13d262f2aa0d1 - checksums.txt: c6ba60cd0d11227022dfcfaea0d4c7716a71c4388f9b2cafa0fc2af85de257cb - deploy.sh: f5e448117266a5341c361f7468d2b4b0460be002aeb1dbe550a89e24a4689240 + checksums.txt: 3a2754aa1b6a9e0c4255bd425ad9023d5e5e83577ee079f7560f6fbb5b8b3a8e + deploy.sh: 75767d3a8f6260fecfb8ea17fe3ab4855c81e2cba101ee4f5b90330f59cf3f4a recipe.yaml: 2637f0fd131e88d5adafa415d7de0be4e9bfdd38a17288f880dd40f3d616eedf rtx-pro-6000-lke-ubuntu-training-kubeflow: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -4273,8 +4273,8 @@ rtx-pro-6000-lke-ubuntu-training-kubeflow: README.md: 7c35d1cc30c8e737e00bb5316bd85226e888834ba942cf1acec838e7cd435000 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 3b51439df6f782b0664fb14b73c7decb36cd2d943ff11a39478eb26d9348723a - checksums.txt: 8e1207dc77a0b1cc087633248bd58507ee1e8e7944cf3356fa709d13c354fca5 - deploy.sh: 8ebf4bd801a6d8fd6cba8979dc354e9aeeb8d5cfccbee9e6e7262417b175c6f4 + checksums.txt: af4bbb98a3dbe5cd3899384864fab933463d9cc2862eceb641a1cac8d4ac9c30 + deploy.sh: 0cd9c43e70d4b627cab9b30c96262a95c91b3c0b77aa6dcc7296b09303a80d04 recipe.yaml: 953b93749485bc38a22acf5d0ac2b0e458df9d02d9e087c3e095403df7f8e84a vr200-rke2-ubuntu-inference-dynamo: 001-agentgateway-crds/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -4363,8 +4363,8 @@ vr200-rke2-ubuntu-inference-dynamo: README.md: b0d4dc1341565474394f7c338e1187f97aaa84f2506642d2c91a212f57ba1669 UPGRADING.md: d27ec18c5dbe463bc50dfdde771821d0a717e350be36a89f87297ada543ba81c bundle-info.yaml: 3614ecc79f449848f257e036fb15a0b8fb7cf24fcad50d0079468a98cc92fda3 - checksums.txt: e89cc487af73cecd7dee840f690c6cc66a1c41f27f337eb4eb00506336fd09d5 - deploy.sh: 62b3ab91044223177c74d3aee7241f4aeae64e6bdb31481f7da642dbf4556e9a + checksums.txt: fe7593dcbac5717b0658a9f7a3d2d3a0ee48dac8538e1971e81118a35d64fc3e + deploy.sh: f9ae0011756a0dc2acd2b02a3682da24ed0902a2a4d1bed1d8c9059dcdc14dca recipe.yaml: e3ca2966624bc02a472b86a93d10a8786a869e74fb461420ebf4d8fc0b208f87 vr200-rke2-ubuntu-training-kubeflow: 001-cert-manager/cluster-values.yaml: 49f560d440b0b58c4ec35c7f72fc187f23b94a69f5f7135482fa7070310bae26 @@ -4434,6 +4434,6 @@ vr200-rke2-ubuntu-training-kubeflow: README.md: 80e547edb47741f31f7b21d02cbf9b0fa9a17a99b2a7320cbdff82d856d1cb81 UPGRADING.md: ac50a3b91fe4d02988ef696661f8ac7d5e6854e49419d573361ad4f2b280c327 bundle-info.yaml: 5bfe00acd931aef4736d88e05805e651f6bbb6d1afa830294329f81d2a74c7f5 - checksums.txt: 5ca2e27c469f913d24310d8fcfeda43dfa41610d63be12718d6c4a105dae9db6 - deploy.sh: 2b0cd173c38bb27ee4cdb7d6c4d1d94d523b56fed94d6889cacd342b0b5a0dfd + checksums.txt: ff9be24c05dc66c8bf69d9e3a56eaa90d2c3fcb727165e4a0560615e8d598a4a + deploy.sh: 6eb8bceaaadd38d35e3445f4814dc8b391a6d840f4615cbe706414b08be1efa4 recipe.yaml: eae45aaf8edf5ea05bc9e583440d75ffcdf378166ca86e5e469cd3e572b00e07 diff --git a/pkg/bundler/validations/checks.go b/pkg/bundler/validations/checks.go index 1f216745fe..4d616b63a3 100644 --- a/pkg/bundler/validations/checks.go +++ b/pkg/bundler/validations/checks.go @@ -539,6 +539,10 @@ const gpuOperatorManagedOverrideSet = "--set gpuoperator:driver.enabled=true " + "--set gpuoperator:operator.runtimeClass=nvidia " + "--set dradriver:nvidiaDriverRoot=/run/nvidia/driver" +const ocpGPUOperatorManagedOverrideSet = "--set gpuoperatorocp:driver.enabled=true " + + "--set gpuoperatorocp:toolkit.enabled=true " + + "--set dradriverocp:nvidiaDriverRoot=/run/nvidia/driver" + // gkeGPUOperatorManagedOverrideSet extends the override tuple for GKE // remedies. GKE preinstalled-driver profiles (Google driver installer, // documented for both COS and Ubuntu node images) pin @@ -567,20 +571,25 @@ func legacyRecipeAlternativeRemedy(service recipe.CriteriaServiceType, os recipe "driver, so the GPU-Operator-managed override set is not available there; if the " + "GPU nodes use the GKE-managed driver install, retarget the DRA driver root " + "instead: --set dradriver:nvidiaDriverRoot=" + gkeManagedDriverRootPath + "." - if service != recipe.CriteriaServiceGKE { - return "Or supply the full GPU-Operator-managed override set: " + - gpuOperatorManagedOverrideSet + "." - } - switch os { //nolint:exhaustive // COS and Ubuntu are the only GKE node images with specific wording; everything else (unknown, any, or an OS GKE does not offer) gets both supported GKE paths - case recipe.CriteriaOSCOS: - return gkeCOSAlternative - case recipe.CriteriaOSUbuntu: + switch service { //nolint:exhaustive // only GKE and OCP need dedicated override-key wording; every other service takes the generic gpuOperatorManagedOverrideSet default + case recipe.CriteriaServiceOCP: return "Or supply the full GPU-Operator-managed override set: " + - gkeGPUOperatorManagedOverrideSet + "." + ocpGPUOperatorManagedOverrideSet + "." + case recipe.CriteriaServiceGKE: + switch os { //nolint:exhaustive // COS and Ubuntu are the only GKE node images with specific wording; everything else (unknown, any, or an OS GKE does not offer) gets both supported GKE paths + case recipe.CriteriaOSCOS: + return gkeCOSAlternative + case recipe.CriteriaOSUbuntu: + return "Or supply the full GPU-Operator-managed override set: " + + gkeGPUOperatorManagedOverrideSet + "." + default: + return gkeCOSAlternative + " On GKE Ubuntu node images the GPU Operator can manage " + + "the driver, so those may instead supply the full GPU-Operator-managed " + + "override set: " + gkeGPUOperatorManagedOverrideSet + "." + } default: - return gkeCOSAlternative + " On GKE Ubuntu node images the GPU Operator can manage " + - "the driver, so those may instead supply the full GPU-Operator-managed " + - "override set: " + gkeGPUOperatorManagedOverrideSet + "." + return "Or supply the full GPU-Operator-managed override set: " + + gpuOperatorManagedOverrideSet + "." } } @@ -598,7 +607,7 @@ func legacyRecipeAlternativeRemedy(service recipe.CriteriaServiceType, os recipe // (see gpuOperatorManagedOverrideSet above for why the duplication // exists). func driverAbsentRemedy(service recipe.CriteriaServiceType, os recipe.CriteriaOSType, profiled bool) string { - switch service { //nolint:exhaustive // only AKS and GKE have provider-specific wording; every other service takes the generic default + switch service { //nolint:exhaustive // only AKS, GKE, and OCP have provider-specific wording; every other service takes the generic default case recipe.CriteriaServiceAKS: if !profiled { // Legacy pre-profile artifact: the ownership lock does not @@ -651,6 +660,10 @@ func driverAbsentRemedy(service recipe.CriteriaServiceType, os recipe.CriteriaOS "driver, so those may bundle in GPU-Operator-managed mode: " + gkeGPUOperatorManagedOverrideSet + "." } + case recipe.CriteriaServiceOCP: + return "Either reprovision the GPU nodes with a platform-installed " + + "NVIDIA driver, or bundle in GPU-Operator-managed mode: " + + ocpGPUOperatorManagedOverrideSet + "." default: return "Either reprovision the GPU nodes with a platform-installed " + "NVIDIA driver, or bundle in GPU-Operator-managed mode: " + @@ -1144,7 +1157,7 @@ func draLockstepViolations(ctx context.Context, recipeResult *recipe.RecipeResul "populates that path when the operator does not manage the driver. This is "+ "commonly the signature of a recipe generated before the preinstalled-driver "+ "default flip: regenerate the recipe (aicr recipe ...) for this AICR version. %s", - componentName, draDriverComponentName, operatorContainerDriverRoot, + componentName, draRef.Name, operatorContainerDriverRoot, legacyRecipeAlternativeRemedy(service, osCriteria))) } return msgs, nil diff --git a/pkg/bundler/validations/checks_test.go b/pkg/bundler/validations/checks_test.go index 1bd73b037b..25136489bd 100644 --- a/pkg/bundler/validations/checks_test.go +++ b/pkg/bundler/validations/checks_test.go @@ -932,6 +932,36 @@ func TestCheckConditions(t *testing.T) { } } +// TestDriverAbsentRemedy_OCP pins the OCP branch added for #2135: +// driverAbsentRemedy must use the OCP-specific override keys +// (gpuoperatorocp:/dradriverocp:) rather than falling through to the +// generic gpuoperator:/dradriver: default, since those keys don't +// resolve against an OCP recipe's registry aliases. Also asserts the +// generic-default keys are NOT present, so a future edit that +// accidentally reuses gpuOperatorManagedOverrideSet for OCP fails +// loudly instead of silently. +func TestDriverAbsentRemedy_OCP(t *testing.T) { + got := driverAbsentRemedy(recipe.CriteriaServiceOCP, "", false) + + wantContains := []string{ + "--set gpuoperatorocp:driver.enabled=true", + "--set gpuoperatorocp:toolkit.enabled=true", + "--set dradriverocp:nvidiaDriverRoot=/run/nvidia/driver", + } + for _, want := range wantContains { + if !strings.Contains(got, want) { + t.Errorf("driverAbsentRemedy(OCP) missing %q:\n%s", want, got) + } + } + + dontWant := []string{"gpuoperator:", "dradriver:"} + for _, unwanted := range dontWant { + if strings.Contains(got, unwanted) { + t.Errorf("driverAbsentRemedy(OCP) unexpectedly contains generic-default key %q:\n%s", unwanted, got) + } + } +} + // TestCheckDriverOwnershipCoherence covers the bundle-time // driver-ownership gate on FINAL effective values: Rule 1 (a recipe whose // snapshot observed no NVIDIA driver on the sampled GPU node — @@ -997,6 +1027,7 @@ func TestCheckDriverOwnershipCoherence(t *testing.T) { recipeResult *recipe.RecipeResult bundlerConfig *config.Config conditions map[string][]string + componentName string // defaults to "gpu-operator" when empty; set to test an OCP-alias row through the exact-match componentName entrypoint wantMsgs int wantContains []string wantErrs int @@ -1280,6 +1311,26 @@ func TestCheckDriverOwnershipCoherence(t *testing.T) { wantMsgs: 1, wantContains: []string{"nothing", "regenerate the recipe", "--set gpuoperator:driver.enabled=true"}, }, + { + // Same Rule-2 legacy signature, but through the OCP aliases: + // gpu-operator-ocp/nvidia-dra-driver-gpu-ocp. Verifies the + // remedy names the resolved OCP component (draRef.Name, not + // the canonical constant) and the OCP override keys — the + // generic gpuoperator:/dradriver: set would hard-fail on OCP + // since recipes/overlays/ocp.yaml disables the canonical + // gpu-operator component. + name: "Rule 2: OCP alias lockstep names the OCP component and OCP override keys", + componentName: "gpu-operator-ocp", + recipeResult: result("", recipe.CriteriaServiceOCP, + recipe.ComponentRef{Name: "gpu-operator-ocp", Overrides: driverOff()}, + recipe.ComponentRef{Name: "nvidia-dra-driver-gpu-ocp", Overrides: rootAt("/run/nvidia/driver")}), + wantMsgs: 1, + wantContains: []string{ + "nvidia-dra-driver-gpu-ocp", + "--set gpuoperatorocp:driver.enabled=true", + "--set dradriverocp:nvidiaDriverRoot=/run/nvidia/driver", + }, + }, { name: "Rule 2: driver off + DRA root=/ (preinstalled profile) → passes", recipeResult: result("", aks, @@ -1801,8 +1852,12 @@ func TestCheckDriverOwnershipCoherence(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { + componentName := tt.componentName + if componentName == "" { + componentName = "gpu-operator" + } msgs, errs := CheckDriverOwnershipCoherence( - context.Background(), "gpu-operator", tt.recipeResult, tt.bundlerConfig, tt.conditions) + context.Background(), componentName, tt.recipeResult, tt.bundlerConfig, tt.conditions) if len(errs) != tt.wantErrs { t.Fatalf("hard errors = %d (%v), want %d", len(errs), errs, tt.wantErrs) } diff --git a/pkg/client/v1/gpu_driver_state.go b/pkg/client/v1/gpu_driver_state.go index 59801164d2..62dd565c81 100644 --- a/pkg/client/v1/gpu_driver_state.go +++ b/pkg/client/v1/gpu_driver_state.go @@ -51,6 +51,15 @@ const gpuOperatorManagedOverrideSet = "--set gpuoperator:driver.enabled=true " + "--set gpuoperator:operator.runtimeClass=nvidia " + "--set dradriver:nvidiaDriverRoot=/run/nvidia/driver" +// ocpGPUOperatorManagedOverrideSet is gpuOperatorManagedOverrideSet's OCP +// counterpart — see the sibling constant of the same name in +// pkg/bundler/validations/checks.go for why the keys and fields differ +// (gpuoperatorocp:/dradriverocp: aliases, no operator.runtimeClass on +// the ClusterPolicy CR). Keep both copies in sync. +const ocpGPUOperatorManagedOverrideSet = "--set gpuoperatorocp:driver.enabled=true " + + "--set gpuoperatorocp:toolkit.enabled=true " + + "--set dradriverocp:nvidiaDriverRoot=/run/nvidia/driver" + // gkeGPUOperatorManagedOverrideSet extends the override tuple for GKE // remedies. GKE preinstalled-driver profiles (Google driver installer, // documented for both COS and Ubuntu node images) pin @@ -73,7 +82,7 @@ const gkeGPUOperatorManagedOverrideSet = gpuOperatorManagedOverrideSet + // anything else gets the generic reprovision wording plus the override // set. func driverAbsentRemedy(service recipe.CriteriaServiceType, os recipe.CriteriaOSType, profiled bool) string { - switch service { //nolint:exhaustive // only AKS and GKE have provider-specific wording; every other service takes the generic default + switch service { //nolint:exhaustive // only AKS, GKE, and OCP have provider-specific wording; every other service takes the generic default case recipe.CriteriaServiceAKS: if !profiled { // Legacy pre-profile artifact: the ownership lock does not @@ -125,6 +134,10 @@ func driverAbsentRemedy(service recipe.CriteriaServiceType, os recipe.CriteriaOS "driver, so those may bundle in GPU-Operator-managed mode: " + gkeGPUOperatorManagedOverrideSet + "." } + case recipe.CriteriaServiceOCP: + return "Either reprovision the GPU nodes with a platform-installed " + + "NVIDIA driver, or bundle in GPU-Operator-managed mode: " + + ocpGPUOperatorManagedOverrideSet + "." default: return "Either reprovision the GPU nodes with a platform-installed " + "NVIDIA driver, or bundle in GPU-Operator-managed mode: " + diff --git a/pkg/client/v1/gpu_driver_state_test.go b/pkg/client/v1/gpu_driver_state_test.go index 65a0c32e9a..eb2e2c1bcf 100644 --- a/pkg/client/v1/gpu_driver_state_test.go +++ b/pkg/client/v1/gpu_driver_state_test.go @@ -676,19 +676,22 @@ func TestDriverAbsentRemedyBranches(t *testing.T) { os recipe.CriteriaOSType profiled bool want string + notWant string // optional: substring that must NOT appear; skipped when empty }{ {"aks legacy keeps the four-flag tuple", recipe.CriteriaServiceAKS, recipe.CriteriaOSUbuntu, false, - "bundle in GPU-Operator-managed mode"}, + "bundle in GPU-Operator-managed mode", ""}, {"aks profiled points at --profile", recipe.CriteriaServiceAKS, recipe.CriteriaOSUbuntu, true, - "--profile gpuStack=operator-managed"}, + "--profile gpuStack=operator-managed", ""}, {"gke cos forbids operator install", recipe.CriteriaServiceGKE, recipe.CriteriaOSCOS, false, - "GPU Operator cannot install the driver"}, + "GPU Operator cannot install the driver", ""}, {"gke ubuntu allows operator mode", recipe.CriteriaServiceGKE, recipe.CriteriaOSUbuntu, false, - "GKE Ubuntu node images the GPU Operator can manage"}, + "GKE Ubuntu node images the GPU Operator can manage", ""}, {"gke unknown os presents both paths", recipe.CriteriaServiceGKE, recipe.CriteriaOSAny, false, - "those may bundle in GPU-Operator-managed mode"}, + "those may bundle in GPU-Operator-managed mode", ""}, {"generic service gets the platform wording", recipe.CriteriaServiceEKS, recipe.CriteriaOSUbuntu, false, - "reprovision the GPU nodes with a platform-installed"}, + "reprovision the GPU nodes with a platform-installed", ""}, + {"ocp uses the ocp-aliased override keys", recipe.CriteriaServiceOCP, recipe.CriteriaOSAny, false, + "gpuoperatorocp:driver.enabled=true", "gpuoperator:"}, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { @@ -697,6 +700,10 @@ func TestDriverAbsentRemedyBranches(t *testing.T) { t.Errorf("driverAbsentRemedy(%s,%s,%v) = %q, want substring %q", tt.service, tt.os, tt.profiled, got, tt.want) } + if tt.notWant != "" && strings.Contains(got, tt.notWant) { + t.Errorf("driverAbsentRemedy(%s,%s,%v) = %q, unexpectedly contains %q", + tt.service, tt.os, tt.profiled, got, tt.notWant) + } }) } // The AKS legacy and profiled remedies must be distinct: the lock only