From 4747bf4dab7852c05cf31f9db86dc04cdce8b836 Mon Sep 17 00:00:00 2001 From: Manuel Huber Date: Sat, 18 Jul 2026 11:29:09 +0200 Subject: [PATCH 1/6] nvidia/runtime-rs: use block-plain emptyDir MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Set the non-confidential NVIDIA runtime-rs emptyDir mode to block-plain through a dedicated Makefile default. This prepares the runtime class for configurations where filesystem sharing is disabled and emptyDir volumes need to be backed by guest-mounted block devices instead of shared-fs. Also add the shared configure_nvidia_runtime_rs_shared_fs_dropin() test helper, which keeps the NVIDIA runtime-rs Docker/nerdctl smoke tests on virtio-fs while the shared_fs=none + EROFS snapshotter path is exercised by Kubernetes CI. Signed-off-by: Manuel Huber Signed-off-by: Fabiano Fidêncio Assisted-by: OpenAI Codex --- src/runtime-rs/Makefile | 2 ++ tests/common.bash | 25 +++++++++++++++++++++++++ 2 files changed, 27 insertions(+) diff --git a/src/runtime-rs/Makefile b/src/runtime-rs/Makefile index 613b85bd79..6bd8f71138 100644 --- a/src/runtime-rs/Makefile +++ b/src/runtime-rs/Makefile @@ -204,6 +204,7 @@ DEFENABLEANNOTATIONS_COCO := [\"enable_iommu\", \"kernel_params\", \"kernel_veri DEFDISABLEGUESTSECCOMP := true DEFEMPTYDIRMODE := shared-fs DEFEMPTYDIRMODE_COCO := block-encrypted +DEFEMPTYDIRMODE_NV := block-plain ##VAR DEFAULTEXPFEATURES=[features] Default experimental features enabled DEFAULTEXPFEATURES := [] DEFDISABLESELINUX := false @@ -751,6 +752,7 @@ USER_VARS += DEFNETWORKMODEL_QEMU USER_VARS += DEFNETWORKMODEL_FC USER_VARS += DEFEMPTYDIRMODE USER_VARS += DEFEMPTYDIRMODE_COCO +USER_VARS += DEFEMPTYDIRMODE_NV USER_VARS += DEFDISABLEGUESTSECCOMP USER_VARS += DEFDISABLESELINUX USER_VARS += DEFDISABLEGUESTSELINUX diff --git a/tests/common.bash b/tests/common.bash index 36f15a72a6..708cbc7ae0 100644 --- a/tests/common.bash +++ b/tests/common.bash @@ -807,6 +807,31 @@ function enabling_hypervisor() { export KATA_CONFIG_PATH="${DEST_KATA_CONFIG}" } +# Docker and nerdctl smoke tests exercise Kata through the default overlayfs +# snapshotter path. Keep NVIDIA runtime-rs on virtio-fs for those tests; the +# shared_fs=none + EROFS snapshotter path is covered by Kubernetes CI instead. +function configure_nvidia_runtime_rs_shared_fs_dropin() { + case "${KATA_HYPERVISOR:-}" in + qemu-nvidia-cpu-runtime-rs) ;; + *) return 0 ;; + esac + + local -r cfg="${KATA_CONFIG_PATH:-}" + [[ -z "${cfg}" || ! -e "${cfg}" ]] && return 0 + + local -r dropin_dir="$(dirname "${cfg}")/config.d" + local -r dropin_path="${dropin_dir}/99-nvidia-runtime-rs-shared-fs.toml" + + info "Configuring NVIDIA runtime-rs shared-fs smoke test via ${dropin_path}" + sudo mkdir -p "${dropin_dir}" + sudo tee "${dropin_path}" >/dev/null < Date: Sat, 18 Jul 2026 11:31:50 +0200 Subject: [PATCH 2/6] nvidia/runtime-rs: use EROFS for CPU tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Enable the EROFS snapshotter for the non-confidential NVIDIA CPU-only runtime-rs handler and move it to a guest-owned storage model. The CPU runtime-rs QEMU configuration now disables filesystem sharing and uses block-plain emptyDir. The NVIDIA CPU Kubernetes workflows (free-runner and arm64) select the EROFS snapshotter with memory-backed writable layers and dm-verity only for qemu-nvidia-cpu-runtime-rs, while the Go CPU runtime keeps the default snapshotter path. nerdctl smoke tests are kept on virtio-fs through the shared configure_nvidia_runtime_rs_shared_fs_dropin() opt-out, since they do not exercise the Kubernetes EROFS snapshotter mapping. NVIDIA CI hosts lack ext4 fs-verity support, so the erofs containerd drop-in disables fsverity for the CPU handler. Signed-off-by: Manuel Huber Signed-off-by: Fabiano Fidêncio Assisted-by: OpenAI Codex --- .github/workflows/run-k8s-tests-on-free-runner.yaml | 3 ++- .github/workflows/run-k8s-tests-on-nvidia-cpu-arm64.yaml | 7 +++++-- .../configuration-qemu-nvidia-cpu-runtime-rs.toml.in | 4 ++-- tests/gha-run-k8s-common.sh | 5 +++++ tests/integration/docker/gha-run.sh | 1 + tests/integration/kubernetes/gha-run.sh | 4 ++++ tests/integration/nerdctl/gha-run.sh | 1 + .../helm-chart/kata-deploy/try-kata-nvidia-cpu.values.yaml | 4 ++-- 8 files changed, 22 insertions(+), 7 deletions(-) diff --git a/.github/workflows/run-k8s-tests-on-free-runner.yaml b/.github/workflows/run-k8s-tests-on-free-runner.yaml index 153ec53b5b..0d79d463e3 100644 --- a/.github/workflows/run-k8s-tests-on-free-runner.yaml +++ b/.github/workflows/run-k8s-tests-on-free-runner.yaml @@ -53,7 +53,7 @@ jobs: { vmm: clh-runtime-rs, containerd_version: latest }, { vmm: clh-runtime-rs, containerd_version: minimum }, { vmm: qemu-nvidia-cpu, containerd_version: latest }, - { vmm: qemu-nvidia-cpu-runtime-rs, containerd_version: latest }, + { vmm: qemu-nvidia-cpu-runtime-rs, containerd_version: latest, snapshotter: erofs, erofs_mode: memory, erofs_dmverity: dmverity }, ] concurrency: group: ${{ github.workflow }}-${{ github.job }}-${{ github.event.pull_request.number || github.ref }}-free-runner-${{ toJSON(matrix) }} @@ -74,6 +74,7 @@ jobs: CONTAINER_ENGINE_VERSION: ${{ matrix.environment.containerd_version }} SNAPSHOTTER: ${{ matrix.environment.snapshotter }} EROFS_SNAPSHOTTER_MODE: ${{ matrix.environment.erofs_mode }} + EROFS_DMVERITY: ${{ matrix.environment.erofs_dmverity }} EROFS_MERGE_MODE: ${{ matrix.environment.erofs_merge_mode }} GH_TOKEN: ${{ github.token }} steps: diff --git a/.github/workflows/run-k8s-tests-on-nvidia-cpu-arm64.yaml b/.github/workflows/run-k8s-tests-on-nvidia-cpu-arm64.yaml index 0c67b51966..98847322b4 100644 --- a/.github/workflows/run-k8s-tests-on-nvidia-cpu-arm64.yaml +++ b/.github/workflows/run-k8s-tests-on-nvidia-cpu-arm64.yaml @@ -38,8 +38,8 @@ jobs: fail-fast: false matrix: environment: [ - { name: nvidia-cpu, vmm: qemu-nvidia-cpu, runner: arm64-nvidia-gh200 }, - { name: nvidia-cpu-runtime-rs, vmm: qemu-nvidia-cpu-runtime-rs, runner: arm64-nvidia-gh200 }, + { name: nvidia-cpu, vmm: qemu-nvidia-cpu, runner: arm64-nvidia-gh200, snapshotter: "" }, + { name: nvidia-cpu-runtime-rs, vmm: qemu-nvidia-cpu-runtime-rs, runner: arm64-nvidia-gh200, snapshotter: "erofs" }, ] concurrency: group: ${{ github.workflow }}-${{ github.job }}-${{ github.event.pull_request.number || github.ref }}-${{ toJSON(matrix) }} @@ -54,6 +54,9 @@ jobs: KUBERNETES: kubeadm K8S_TEST_HOST_TYPE: baremetal TARGET_ARCH: aarch64 + SNAPSHOTTER: ${{ matrix.environment.snapshotter }} + EROFS_SNAPSHOTTER_MODE: ${{ matrix.environment.snapshotter == 'erofs' && 'memory' || '' }} + EROFS_DMVERITY: ${{ matrix.environment.snapshotter == 'erofs' && 'dmverity' || '' }} steps: - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 with: diff --git a/src/runtime-rs/config/configuration-qemu-nvidia-cpu-runtime-rs.toml.in b/src/runtime-rs/config/configuration-qemu-nvidia-cpu-runtime-rs.toml.in index 5e7318109d..e14cc533f9 100644 --- a/src/runtime-rs/config/configuration-qemu-nvidia-cpu-runtime-rs.toml.in +++ b/src/runtime-rs/config/configuration-qemu-nvidia-cpu-runtime-rs.toml.in @@ -195,7 +195,7 @@ disable_block_device_use = @DEFDISABLEBLOCK@ # - virtio-fs (default) # - virtio-fs-nydus # - none -shared_fs = "@DEFSHAREDFS_QEMU_VIRTIOFS@" +shared_fs = "none" # Path to vhost-user-fs daemon. virtio_fs_daemon = "@DEFVIRTIOFSDAEMON@" @@ -808,7 +808,7 @@ vfio_mode = "@DEFVFIOMODE_NV@" # - block-plain # Plugs a block device to be mounted directly in the guest. # -emptydir_mode = "@DEFEMPTYDIRMODE@" +emptydir_mode = "@DEFEMPTYDIRMODE_NV@" # Enabled experimental feature list, format: ["a", "b"]. # Experimental features are features not stable enough for production, diff --git a/tests/gha-run-k8s-common.sh b/tests/gha-run-k8s-common.sh index ec32e70357..b8fc2d62ed 100644 --- a/tests/gha-run-k8s-common.sh +++ b/tests/gha-run-k8s-common.sh @@ -880,6 +880,11 @@ function helm_helper() { HELM_CONTAINERD_USER_DROP_IN="[plugins.'io.containerd.snapshotter.v1.erofs']"$'\n' HELM_CONTAINERD_USER_DROP_IN+=" default_size = \"${erofs_default_size}\"" + # NVIDIA CI hosts are not configured with ext4 fs-verity support. + if [[ "${KATA_HYPERVISOR:-}" == *"nvidia-cpu"* ]]; then + HELM_CONTAINERD_USER_DROP_IN+=$'\n'" enable_fsverity = false" + fi + HELM_CONTAINERD_USER_DROP_IN="${HELM_CONTAINERD_USER_DROP_IN}" \ yq -i '.containerd.userDropIn = strenv(HELM_CONTAINERD_USER_DROP_IN)' "${values_yaml}" diff --git a/tests/integration/docker/gha-run.sh b/tests/integration/docker/gha-run.sh index 09693052cc..2dffd7e6e3 100755 --- a/tests/integration/docker/gha-run.sh +++ b/tests/integration/docker/gha-run.sh @@ -92,6 +92,7 @@ function run() { info "Running docker smoke test tests using ${KATA_HYPERVISOR} hypervisor" enabling_hypervisor + configure_nvidia_runtime_rs_shared_fs_dropin enable_kata_debug info "Running docker with runc" diff --git a/tests/integration/kubernetes/gha-run.sh b/tests/integration/kubernetes/gha-run.sh index 274d1b6a76..fa6059e511 100755 --- a/tests/integration/kubernetes/gha-run.sh +++ b/tests/integration/kubernetes/gha-run.sh @@ -200,6 +200,10 @@ function deploy_kata() { EXPERIMENTAL_FORCE_GUEST_PULL=false fi + if [[ "${KATA_HYPERVISOR}" == "qemu-nvidia-cpu-runtime-rs" ]] && [[ -z "${SNAPSHOTTER}" ]]; then + SNAPSHOTTER="erofs" + fi + ANNOTATIONS="default_vcpus" if [[ "${KATA_HYPERVISOR}" == *azure* ]]; then ANNOTATIONS="image kernel default_vcpus cc_init_data" diff --git a/tests/integration/nerdctl/gha-run.sh b/tests/integration/nerdctl/gha-run.sh index 5790cae3de..5a4dadde88 100755 --- a/tests/integration/nerdctl/gha-run.sh +++ b/tests/integration/nerdctl/gha-run.sh @@ -101,6 +101,7 @@ function run() { sudo nerdctl network create "${net2}" enabling_hypervisor + configure_nvidia_runtime_rs_shared_fs_dropin if [[ -n "${GITHUB_ENV:-}" ]]; then start_time=$(date '+%Y-%m-%d %H:%M:%S') diff --git a/tools/packaging/kata-deploy/helm-chart/kata-deploy/try-kata-nvidia-cpu.values.yaml b/tools/packaging/kata-deploy/helm-chart/kata-deploy/try-kata-nvidia-cpu.values.yaml index 991c96cb7b..a9384ff6f6 100644 --- a/tools/packaging/kata-deploy/helm-chart/kata-deploy/try-kata-nvidia-cpu.values.yaml +++ b/tools/packaging/kata-deploy/helm-chart/kata-deploy/try-kata-nvidia-cpu.values.yaml @@ -15,7 +15,7 @@ debug: false deploymentMode: job snapshotter: - setup: [] + setup: ["erofs"] # Enable NVIDIA CPU shims # Disable all shims at once, then enable only the ones we need. @@ -38,7 +38,7 @@ shims: - arm64 allowedHypervisorAnnotations: [] containerd: - snapshotter: "" + snapshotter: "erofs" # Default shim per architecture. defaultShim: From 2d56b83411c911771763c22457183e0fc62c3546 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Fabiano=20Fid=C3=AAncio?= Date: Sat, 18 Jul 2026 11:53:30 +0200 Subject: [PATCH 3/6] nvidia/runtime-rs: disable EROFS fs-verity in NVIDIA CPU profile MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit EROFS fs-verity depends on the backing filesystem supporting fs-verity, which cannot be guaranteed on an arbitrary node. Rather than gate the NVIDIA CPU profile on a host feature we do not control, disable fs-verity so it works on the widest possible range of setups; we take that hit deliberately to keep the profile generic. Move this out of the test harness and into the kata-deploy NVIDIA CPU profile: try-kata-nvidia-cpu.values.yaml now ships a containerd erofs snapshotter drop-in that disables fs-verity and pins the memory-backed default_size, and selects memory-backed rw layers with dm-verity for the lower (image) layers. Because containerd.userDropIn is loaded after kata-deploy's generated config, it overrides the built-in enable_fsverity default. The CI helm helper no longer special-cases NVIDIA CPU to inject enable_fsverity=false. Instead it honors an erofs snapshotter drop-in already provided by the base values file and only synthesizes the default default_size drop-in when the profile does not provide its own. Note that this drop-in targets containerd's built-in EROFS snapshotter, available only on containerd >= 2.2.0 (config version 3, conf.d auto-import), which is the same minimum kata-deploy already enforces for the EROFS snapshotter. Signed-off-by: Fabiano Fidêncio Assisted-by: Cursor with Claude Opus 4.8 --- tests/gha-run-k8s-common.sh | 46 +++++++++++-------- .../try-kata-nvidia-cpu.values.yaml | 28 +++++++++++ 2 files changed, 54 insertions(+), 20 deletions(-) diff --git a/tests/gha-run-k8s-common.sh b/tests/gha-run-k8s-common.sh index b8fc2d62ed..23adc07ef6 100644 --- a/tests/gha-run-k8s-common.sh +++ b/tests/gha-run-k8s-common.sh @@ -864,30 +864,36 @@ function helm_helper() { die "EROFS_SNAPSHOTTER_MODE is only supported with SNAPSHOTTER=erofs" fi - local erofs_default_size - case "${EROFS_SNAPSHOTTER_MODE}" in - disk) - erofs_default_size="10G" - ;; - memory) - erofs_default_size="0" - ;; - *) - die "Unsupported EROFS_SNAPSHOTTER_MODE: ${EROFS_SNAPSHOTTER_MODE}" - ;; - esac + # Honor an erofs snapshotter drop-in already shipped by the base + # values file (e.g. try-kata-nvidia-cpu.values.yaml pins the + # memory-backed default_size and disables fs-verity, which cannot + # be guaranteed on an arbitrary node's backing filesystem). Only + # synthesize the default drop-in when the profile does not provide + # its own. + local existing_dropin + existing_dropin="$(yq -r '.containerd.userDropIn // ""' "${values_yaml}")" - HELM_CONTAINERD_USER_DROP_IN="[plugins.'io.containerd.snapshotter.v1.erofs']"$'\n' - HELM_CONTAINERD_USER_DROP_IN+=" default_size = \"${erofs_default_size}\"" + if [[ -z "${existing_dropin//[[:space:]]/}" ]]; then + local erofs_default_size + case "${EROFS_SNAPSHOTTER_MODE}" in + disk) + erofs_default_size="10G" + ;; + memory) + erofs_default_size="0" + ;; + *) + die "Unsupported EROFS_SNAPSHOTTER_MODE: ${EROFS_SNAPSHOTTER_MODE}" + ;; + esac - # NVIDIA CI hosts are not configured with ext4 fs-verity support. - if [[ "${KATA_HYPERVISOR:-}" == *"nvidia-cpu"* ]]; then - HELM_CONTAINERD_USER_DROP_IN+=$'\n'" enable_fsverity = false" + HELM_CONTAINERD_USER_DROP_IN="[plugins.'io.containerd.snapshotter.v1.erofs']"$'\n' + HELM_CONTAINERD_USER_DROP_IN+=" default_size = \"${erofs_default_size}\"" + + HELM_CONTAINERD_USER_DROP_IN="${HELM_CONTAINERD_USER_DROP_IN}" \ + yq -i '.containerd.userDropIn = strenv(HELM_CONTAINERD_USER_DROP_IN)' "${values_yaml}" fi - HELM_CONTAINERD_USER_DROP_IN="${HELM_CONTAINERD_USER_DROP_IN}" \ - yq -i '.containerd.userDropIn = strenv(HELM_CONTAINERD_USER_DROP_IN)' "${values_yaml}" - # Propagate rwlayer backing mode to kata-deploy. yq -i ".snapshotter.erofsSnapshotterMode = \"${EROFS_SNAPSHOTTER_MODE}\"" "${values_yaml}" fi diff --git a/tools/packaging/kata-deploy/helm-chart/kata-deploy/try-kata-nvidia-cpu.values.yaml b/tools/packaging/kata-deploy/helm-chart/kata-deploy/try-kata-nvidia-cpu.values.yaml index a9384ff6f6..0d1340b26b 100644 --- a/tools/packaging/kata-deploy/helm-chart/kata-deploy/try-kata-nvidia-cpu.values.yaml +++ b/tools/packaging/kata-deploy/helm-chart/kata-deploy/try-kata-nvidia-cpu.values.yaml @@ -16,6 +16,34 @@ deploymentMode: job snapshotter: setup: ["erofs"] + # The NVIDIA CPU runtime-rs handler runs with shared_fs=none and uses the + # EROFS snapshotter with memory-backed rw layers and dm-verity for lower + # (image) layer integrity. + erofsSnapshotterMode: "memory" + erofsDmverity: true + +# EROFS fs-verity requires the backing filesystem to support fs-verity, and we +# cannot guarantee that on an arbitrary node. Rather than gate this profile on a +# host feature we do not control, we disable fs-verity so it works on the widest +# possible range of setups; we take that hit deliberately to keep the profile +# generic. Combined with memory-backed EROFS rw layers, ship this as the default +# containerd erofs snapshotter drop-in for the NVIDIA CPU profile so it does not +# need to be injected out-of-band. It is loaded after kata-deploy's generated +# config, so it overrides the built-in enable_fsverity default. +# +# NOTE: This drop-in targets containerd's built-in EROFS snapshotter, which is +# only available on containerd >= 2.2.0 (config version 3, using the +# auto-imported conf.d drop-in mechanism). This is the same minimum kata-deploy +# enforces for the EROFS snapshotter (see check_containerd_erofs_version_support: +# "In order to use erofs-snapshotter containerd must be 2.2.0 or newer"). On +# older containerd releases the erofs snapshotter plugin does not exist and this +# specific "[plugins.'io.containerd.snapshotter.v1.erofs']" key layout will not +# be recognized, so do not expect this profile to work there. +containerd: + userDropIn: | + [plugins.'io.containerd.snapshotter.v1.erofs'] + enable_fsverity = false + default_size = "0" # Enable NVIDIA CPU shims # Disable all shims at once, then enable only the ones we need. From b6449fca2ba2a26661873aabc6a594f7da883df4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Fabiano=20Fid=C3=AAncio?= Date: Sat, 18 Jul 2026 11:53:53 +0200 Subject: [PATCH 4/6] nvidia/runtime-rs: default the NVIDIA CPU profile to runtime-rs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Make qemu-nvidia-cpu-runtime-rs the default shim on amd64 and arm64 for the kata-deploy NVIDIA CPU profile, so installs from try-kata-nvidia-cpu.values.yaml select the Rust runtime by default instead of the Go qemu-nvidia-cpu shim. Signed-off-by: Fabiano Fidêncio Assisted-by: Cursor with Claude Opus 4.8 --- .../helm-chart/kata-deploy/try-kata-nvidia-cpu.values.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tools/packaging/kata-deploy/helm-chart/kata-deploy/try-kata-nvidia-cpu.values.yaml b/tools/packaging/kata-deploy/helm-chart/kata-deploy/try-kata-nvidia-cpu.values.yaml index 0d1340b26b..d7505e55cd 100644 --- a/tools/packaging/kata-deploy/helm-chart/kata-deploy/try-kata-nvidia-cpu.values.yaml +++ b/tools/packaging/kata-deploy/helm-chart/kata-deploy/try-kata-nvidia-cpu.values.yaml @@ -70,8 +70,8 @@ shims: # Default shim per architecture. defaultShim: - amd64: qemu-nvidia-cpu - arm64: qemu-nvidia-cpu + amd64: qemu-nvidia-cpu-runtime-rs + arm64: qemu-nvidia-cpu-runtime-rs runtimeClasses: enabled: true From 271a797600d67e32e4b3d8ed5ef14dd6889f1a6a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Fabiano=20Fid=C3=AAncio?= Date: Sat, 18 Jul 2026 12:31:29 +0200 Subject: [PATCH 5/6] tests: run k8s-termination-log on the NVIDIA CPU runtime-rs handler MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The termination-log tests exercise termination-message propagation over the agent GetDiagnosticData RPC, which only matters when the host cannot read guest files directly, i.e. on shared_fs=none configurations. They previously gated on is_confidential_runtime_class and therefore skipped on the NVIDIA CPU runtime-rs handler, which now also runs with shared_fs=none. Add an is_shared_fs_none_runtime_class helper (confidential runtime classes plus qemu-nvidia-cpu-runtime-rs) and use it for the setup and teardown gating so the tests run on the NVIDIA CPU runtime-rs handler as well. The plain Go qemu-nvidia-cpu class still uses virtio-fs and is intentionally excluded. Keep the "blocked by default CoCo policy" assertion confidential-only, since non-confidential shared_fs=none classes do not ship a default policy that denies the RPC. Signed-off-by: Fabiano Fidêncio Assisted-by: Cursor with Claude Opus 4.8 --- tests/hypervisor_helpers.sh | 15 +++++++++++++ .../kubernetes/k8s-termination-log.bats | 22 ++++++++++++++----- 2 files changed, 31 insertions(+), 6 deletions(-) diff --git a/tests/hypervisor_helpers.sh b/tests/hypervisor_helpers.sh index db60c976dc..209ecbcf15 100644 --- a/tests/hypervisor_helpers.sh +++ b/tests/hypervisor_helpers.sh @@ -122,6 +122,21 @@ function is_confidential_runtime_class() { fi } +# Runtime classes that boot with shared_fs=none, where the host cannot read +# guest files directly. Data such as a container's termination log must then be +# retrieved over the agent GetDiagnosticData RPC instead of a shared filesystem. +# This covers the confidential runtime classes and the non-confidential NVIDIA +# CPU runtime-rs handler. The plain qemu-nvidia-cpu (Go) class still uses +# virtio-fs, so it is intentionally excluded. +function is_shared_fs_none_runtime_class() { + local hypervisor="${1:-${KATA_HYPERVISOR}}" + if is_confidential_runtime_class "${hypervisor}"; then + return 0 + fi + [[ "${hypervisor}" == "qemu-nvidia-cpu-runtime-rs" ]] && return 0 + return 1 +} + # Runtime classes that boot a measured (dm-verity) rootfs: the confidential # classes plus the CPU-only NVIDIA classes, which boot the verity-backed # nvidia base image without being confidential. diff --git a/tests/integration/kubernetes/k8s-termination-log.bats b/tests/integration/kubernetes/k8s-termination-log.bats index 2c398f518b..3c0ae2ab5c 100644 --- a/tests/integration/kubernetes/k8s-termination-log.bats +++ b/tests/integration/kubernetes/k8s-termination-log.bats @@ -5,8 +5,9 @@ # SPDX-License-Identifier: Apache-2.0 # # Test termination log propagation via GetDiagnosticData RPC. -# These tests target shared_fs=none configurations (e.g. qemu-coco-dev) -# where the host cannot directly read guest files. +# These tests target shared_fs=none configurations (e.g. qemu-coco-dev or the +# NVIDIA CPU runtime-rs handler) where the host cannot directly read guest +# files. load "${BATS_TEST_DIRNAME}/lib.sh" load "${BATS_TEST_DIRNAME}/../../common.bash" @@ -14,9 +15,11 @@ load "${BATS_TEST_DIRNAME}/tests_common.sh" load "${BATS_TEST_DIRNAME}/confidential_common.sh" setup() { - # These tests only make sense on CoCo platforms (shared_fs=none). - if ! is_confidential_runtime_class; then - skip "Test requires a CoCo runtime class (shared_fs=none)" + # These tests only make sense on shared_fs=none configurations, where the + # host cannot read guest files directly and the termination log has to be + # fetched over the agent GetDiagnosticData RPC. + if ! is_shared_fs_none_runtime_class; then + skip "Test requires a shared_fs=none runtime class" fi setup_common || die "setup_common failed" @@ -95,6 +98,13 @@ wait_for_pod_terminated() { } @test "Termination log: request blocked by default CoCo policy" { + # The default-deny behaviour is specific to the confidential runtime + # classes; non-confidential shared_fs=none classes (e.g. NVIDIA CPU + # runtime-rs) do not ship a default policy that blocks the RPC. + if ! is_confidential_runtime_class; then + skip "Default-deny policy check only applies to CoCo runtime classes" + fi + if ! auto_generate_policy_enabled; then echo "# Skipping default CoCo policy check: requires AUTO_GENERATE_POLICY=yes" >&3 return 0 @@ -128,7 +138,7 @@ wait_for_pod_terminated() { } teardown() { - if ! is_confidential_runtime_class; then + if ! is_shared_fs_none_runtime_class; then return fi From fc81c79cabe5dfc28dfdbc67eb31fdb6b75ec00d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Fabiano=20Fid=C3=AAncio?= Date: Sat, 18 Jul 2026 13:04:58 +0200 Subject: [PATCH 6/6] runtime-rs: do not require OVMF for qemu-nvidia-cpu-runtime-rs on arm64 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Commit e8bb619b1ba6 ("runtime: do not require OVMF for qemu-nvidia-cpu on arm64") dropped the generic OVMF/AAVMF firmware inheritance for the Go qemu-nvidia-cpu runtime, but assumed runtime-rs already did the right thing by default. It didn't: the runtime-rs profile still pointed at the generic @FIRMWAREPATH@ / @FIRMWAREVOLUMEPATH@ and still pulled in the ovmf component on aarch64. Mirror the fix on the runtime-rs side by introducing dedicated empty FIRMWAREPATH_NV_CPU / FIRMWAREVOLUMEPATH_NV_CPU variables, wiring them into the qemu-nvidia-cpu-runtime-rs configuration, and dropping ovmf from its aarch64 shim components. Signed-off-by: Fabiano Fidêncio Assisted-by: Cursor with Claude Opus 4.8 --- src/runtime-rs/Makefile | 8 ++++++++ .../configuration-qemu-nvidia-cpu-runtime-rs.toml.in | 4 ++-- tools/packaging/kata-deploy/shim-components.json | 2 +- 3 files changed, 11 insertions(+), 3 deletions(-) diff --git a/src/runtime-rs/Makefile b/src/runtime-rs/Makefile index 6bd8f71138..5250633d4e 100644 --- a/src/runtime-rs/Makefile +++ b/src/runtime-rs/Makefile @@ -149,6 +149,12 @@ ifeq ($(ARCH), aarch64) FIRMWAREPATH_NV := $(PREFIXDEPS)/share/$(EDK2_NAME)/AAVMF_CODE.fd endif +# The CPU-only NVIDIA runtime class does not need UEFI firmware (no OVMF/AAVMF). +# Use dedicated empty variables so it does not inherit the generic firmware +# paths; parity with src/runtime/Makefile FIRMWAREPATH_NV_CPU. +FIRMWAREPATH_NV_CPU := +FIRMWAREVOLUMEPATH_NV_CPU := + KERNELVERITYPARAMS ?= "" # TDX @@ -704,7 +710,9 @@ USER_VARS += KERNELPATH USER_VARS += KERNELVIRTIOFSPATH USER_VARS += FIRMWAREPATH USER_VARS += FIRMWAREPATH_NV +USER_VARS += FIRMWAREPATH_NV_CPU USER_VARS += FIRMWAREVOLUMEPATH +USER_VARS += FIRMWAREVOLUMEPATH_NV_CPU USER_VARS += MACHINEACCELERATORS USER_VARS += CPUFEATURES USER_VARS += DEFMACHINETYPE_CLH diff --git a/src/runtime-rs/config/configuration-qemu-nvidia-cpu-runtime-rs.toml.in b/src/runtime-rs/config/configuration-qemu-nvidia-cpu-runtime-rs.toml.in index e14cc533f9..899e469169 100644 --- a/src/runtime-rs/config/configuration-qemu-nvidia-cpu-runtime-rs.toml.in +++ b/src/runtime-rs/config/configuration-qemu-nvidia-cpu-runtime-rs.toml.in @@ -65,13 +65,13 @@ kernel_verity_params = "@KERNELVERITYPARAMS_NV@" # Path to the firmware. # If you want that qemu uses the default firmware leave this option empty -firmware = "@FIRMWAREPATH@" +firmware = "@FIRMWAREPATH_NV_CPU@" # Path to the firmware volume. # firmware TDVF or OVMF can be split into FIRMWARE_VARS.fd (UEFI variables # as configuration) and FIRMWARE_CODE.fd (UEFI program image). UEFI variables # can be customized per each user while UEFI code is kept same. -firmware_volume = "@FIRMWAREVOLUMEPATH@" +firmware_volume = "@FIRMWAREVOLUMEPATH_NV_CPU@" # Machine accelerators # comma-separated list of machine accelerators to pass to the hypervisor. diff --git a/tools/packaging/kata-deploy/shim-components.json b/tools/packaging/kata-deploy/shim-components.json index 3ef05a82b2..085aa8ed06 100644 --- a/tools/packaging/kata-deploy/shim-components.json +++ b/tools/packaging/kata-deploy/shim-components.json @@ -45,7 +45,7 @@ }, "qemu-nvidia-cpu-runtime-rs": { "x86_64": ["shim-v2-rust", "qemu", "virtiofsd", "kernel-nvidia-gpu", "rootfs-image-nvidia"], - "aarch64": ["shim-v2-rust", "qemu", "virtiofsd", "kernel-nvidia-gpu", "rootfs-image-nvidia", "ovmf"] + "aarch64": ["shim-v2-rust", "qemu", "virtiofsd", "kernel-nvidia-gpu", "rootfs-image-nvidia"] }, "qemu-nvidia-gpu-snp": { "x86_64": ["shim-v2-go", "qemu-snp-experimental", "kernel-nvidia-gpu", "rootfs-image-nvidia-gpu-confidential", "ovmf-sev"]