diff --git a/gpu-operator/life-cycle-policy.rst b/gpu-operator/life-cycle-policy.rst index c4c084d8c..4a15b619c 100644 --- a/gpu-operator/life-cycle-policy.rst +++ b/gpu-operator/life-cycle-policy.rst @@ -91,9 +91,10 @@ Refer to :ref:`Upgrading the NVIDIA GPU Operator` for more information. - ${version} * - NVIDIA GPU Driver |ki|_ - - | `580.65.06 `_ (recommended) + - | `580.82.07 `_ (default, recommended) + | `580.65.06 `_ | `575.57.08 `_ - | `570.172.08 `_ (default) + | `570.172.08 `_ | `570.158.01 `_ | `570.148.08 `_ | `535.261.03 `_ @@ -101,38 +102,38 @@ Refer to :ref:`Upgrading the NVIDIA GPU Operator` for more information. | `535.247.01 `_ * - NVIDIA Driver Manager for Kubernetes - - `v0.8.0 `__ + - `v0.8.1 `__ * - NVIDIA Container Toolkit - `1.17.8 `__ * - NVIDIA Kubernetes Device Plugin - - `0.17.3 `__ + - `0.17.4 `__ * - DCGM Exporter - - `4.2.3-4.1.3 `__ + - `4.3.1-4.4.0 `__ * - Node Feature Discovery - `v0.17.3 `__ * - | NVIDIA GPU Feature Discovery | for Kubernetes - - `0.17.3 `__ + - `0.17.4 `__ * - NVIDIA MIG Manager for Kubernetes - - `0.12.2 `__ + - `0.12.3 `__ * - DCGM - - `4.2.3 `__ + - `4.3.1 `__ * - Validator for NVIDIA GPU Operator - ${version} * - NVIDIA KubeVirt GPU Device Plugin - - `v1.3.1 `__ + - `v1.4.0 `__ * - NVIDIA vGPU Device Manager - - `v0.3.0 `__ + - `v0.4.0 `__ * - NVIDIA GDS Driver |gds|_ - `2.20.5 `__ @@ -145,7 +146,7 @@ Refer to :ref:`Upgrading the NVIDIA GPU Operator` for more information. - v0.1.1 * - NVIDIA GDRCopy Driver - - `v2.5.0 `__ + - `v2.5.1 `__ .. _known-issue: diff --git a/gpu-operator/platform-support.rst b/gpu-operator/platform-support.rst index 95e449d2e..c39a2973b 100644 --- a/gpu-operator/platform-support.rst +++ b/gpu-operator/platform-support.rst @@ -148,6 +148,8 @@ The following NVIDIA data center GPUs are supported on x86 based platforms: | NVIDIA RTX PRO 6000 | NVIDIA Blackwell | | Blackwell Server Edition| | +-------------------------+------------------------+ + | NVIDIA RTX PRO 6000D | NVIDIA Blackwell | + +-------------------------+------------------------+ | NVIDIA RTX A6000 | NVIDIA Ampere /Ada | +-------------------------+------------------------+ | NVIDIA RTX A5000 | NVIDIA Ampere | @@ -468,6 +470,9 @@ See the :doc:`precompiled-drivers` page for more information about using precomp | Ubuntu 22.04 | Generic, NVIDIA, Azure | 5.15 | R535, R550, R570 | | | AWS, Oracle | | | +----------------------------+------------------------+----------------+---------------------+ +| Ubuntu 22.04 | Generic, NVIDIA, Azure | 6.8 | R535, R570 | +| | AWS, Oracle | | | ++----------------------------+------------------------+----------------+---------------------+ | Ubuntu 24.04 | Generic, NVIDIA, Azure | 6.8 | R550, R570 | | | AWS, Oracle | | | +----------------------------+------------------------+----------------+---------------------+ @@ -508,8 +513,8 @@ Operating System Kubernetes KubeVirt OpenShift Virtual \ \ | GPU vGPU | GPU vGPU | Passthrough | Passthrough ================ =========== ============= ========= ============= =========== -Ubuntu 20.04 LTS 1.23---1.29 0.36+ 0.59.1+ -Ubuntu 22.04 LTS 1.23---1.29 0.36+ 0.59.1+ +Ubuntu 20.04 LTS 1.23---1.33 0.36+ 0.59.1+ +Ubuntu 22.04 LTS 1.23---1.33 0.36+ 0.59.1+ Red Hat Core OS 4.12---4.19 4.13---4.19 ================ =========== ============= ========= ============= =========== @@ -524,6 +529,8 @@ Refer to :ref:`GPU Operator with KubeVirt` or :ref:`NVIDIA GPU Operator with Ope KubeVirt and OpenShift Virtualization with NVIDIA vGPU is supported on the following devices: +- RTX Pro 6000 Blackwell Server Edition + - H200NVL - H100 diff --git a/gpu-operator/release-notes.rst b/gpu-operator/release-notes.rst index a0900cb9a..afd2bf6de 100644 --- a/gpu-operator/release-notes.rst +++ b/gpu-operator/release-notes.rst @@ -33,6 +33,38 @@ See the :ref:`GPU Operator Component Matrix` for a list of software components a ---- +.. _v25.3.3: + +25.3.3 +====== + +.. _v25.3.3-new-features: + +New Features +------------ + +* Supports these NVIDIA Data Center GPU Driver versions: + + - 580.82.07 (default, recommended) + +* Added support for additional features: + + - RTX Pro 6000 Blackwell Server Edition + + - MIG profiles support + - KubeVirt and OpenShift Virtualization: VM with GPU passthrough (Ubuntu 22.04 only) + - KubeVirt and OpenShift Virtualization: VM with time-slice vGPU (Ubuntu 22.04 only) + + - RTX Pro 6000D + + - KubeVirt and OpenShift Virtualization: VM with GPU passthrough (Ubuntu 22.04 only) + +Fixed Issues +------------- + +* Fixed an issue where user-supplied environment variables configured in ClusterPolicy were not getting set in the rendered DaemonSet. + User-supplied environment variables now take precedence over environment variables set by the ClusterPolicy controller. + .. _v25.3.2: 25.3.2 diff --git a/gpu-operator/troubleshooting.rst b/gpu-operator/troubleshooting.rst index 71ddb9c34..73ea364f8 100644 --- a/gpu-operator/troubleshooting.rst +++ b/gpu-operator/troubleshooting.rst @@ -26,6 +26,29 @@ If you are facing an issue that is not covered by this page, please file an issu `NVIDIA GPU Operator GitHub repository `_. +************************************************** +The ``nouveau`` driver fails to initialize the GPU +************************************************** + +.. rubric:: Observation + :class: h4 + +- The GPU driver fails to initialize the GPU with the error ``Failed to enable MSI-X`` in the system journal logs. +- All GPU Operator pods become stuck in the ``init`` state. + +.. rubric:: Root Cause + :class: h4 + +- The ``nouveau`` Linux kernel module is loaded. + +.. rubric:: Action + :class: h4 + +The ``nouveau`` driver must be denylisted when using NVIDIA vGPU. + +Follow the instructions in the `NVIDIA AI Enterprise: VMware Deployment Guide `_ +to disable ``nouveau`` on your OS/distro to resolve this issue. + *********************************** GPU Operator pods are stuck in Init *********************************** diff --git a/gpu-operator/versions.json b/gpu-operator/versions.json index 3dc444905..b033883fe 100644 --- a/gpu-operator/versions.json +++ b/gpu-operator/versions.json @@ -1,24 +1,24 @@ { - "latest": "24.9.2", + "latest": "25.3.3", "versions": [ { - "version": "24.9.2" + "version": "25.3.3" }, { - "version": "24.9.1" + "version": "25.3.2" }, { - "version": "24.9.0" + "version": "25.3.1" }, { - "version": "24.6.2" + "version": "25.3.0" }, { - "version": "24.6.1" + "version": "24.9.2" }, { - "version": "24.6.0" + "version": "24.9.1" } ] } diff --git a/gpu-operator/versions1.json b/gpu-operator/versions1.json index 86c7cd6ff..33d939b27 100644 --- a/gpu-operator/versions1.json +++ b/gpu-operator/versions1.json @@ -1,6 +1,10 @@ [ { "preferred": "true", + "url": "../25.3.3", + "version": "25.3.3" + }, + { "url": "../25.3.2", "version": "25.3.2" }, @@ -19,9 +23,5 @@ { "url": "../24.9.1", "version": "24.9.1" - }, - { - "url": "../24.9.0", - "version": "24.9.0" } -] \ No newline at end of file +] diff --git a/repo.toml b/repo.toml index a9f9f1b76..2ce711246 100644 --- a/repo.toml +++ b/repo.toml @@ -166,8 +166,8 @@ output_format = "linkcheck" docs_root = "${root}/gpu-operator" project = "gpu-operator" name = "NVIDIA GPU Operator" -version = "25.3.2" -source_substitutions = { version = "v25.3.2", recommended = "580.65.06" } +version = "25.3.3" +source_substitutions = { version = "v25.3.3", recommended = "580.82.07" } copyright_start = 2020 sphinx_exclude_patterns = [ "life-cycle-policy.rst",