From 10664bf57f302a255dbd9e470e174be03713cdc1 Mon Sep 17 00:00:00 2001 From: eabdullin Date: Mon, 22 Dec 2025 06:35:30 +0000 Subject: [PATCH] import OL qemu-kvm-9.1.0-29.el9_7.3 --- SOURCES/kvm-Enable-amd-iommu-device.patch | 38 + ...mmu-Add-support-for-pass-though-mode.patch | 141 ++ ...md_iommu-Check-APIC-ID-255-for-XTSup.patch | 66 + ...ommu-Rename-variable-mmio-to-mr_mmio.patch | 94 ++ ...otification-when-invalidate-interrup.patch | 81 ++ ...ared-memory-region-for-Interrupt-Rem.patch | 105 ++ ..._virt_compat_set-to-apply-the-compat.patch | 53 + ...vm-report-registers-we-failed-to-set.patch | 154 ++ ...d-new-bdrv_co_is_all_zeroes-function.patch | 145 ++ ...block-status-mode-from-bool-to-flags.patch | 689 +++++++++ ...o_is_zero_fast-consolidate-adjacent-.patch | 90 ++ ...io-skip-head-tail-requests-on-EINVAL.patch | 10 +- ...atic-zero-init-of-large-array-in-ioq.patch | 10 +- ...skip-automatic-zero-init-of-large-ar.patch | 10 +- ...-skip-automatic-zero-init-of-large-a.patch | 10 +- ...ket-skip-automatic-zero-init-of-larg.patch | 10 +- ...cpu_dirty-when-guest_state_protected.patch | 48 + ...Remove-nr_cores-from-struct-CPUState.patch | 76 + ...cros-for-hash-algorithm-digest-lengt.patch | 74 + SOURCES/kvm-docs-Add-TDX-documentation.patch | 222 +++ ...-Document-reset-expectations-for-DMA.patch | 53 + ...le-posix-Define-DM_MPATH_PROBE_PATHS.patch | 10 +- ...x-Fix-crash-on-discard_granularity-0.patch | 10 +- ...-paths-and-retry-SG_IO-on-potential-.patch | 10 +- ...er-Handle-zero-block-status-hint-bet.patch | 64 + ...-discard-alignment-on-Linux-block-de.patch | 10 +- ...nitions-from-UEFI-spec-for-volumes-r.patch | 233 +++ ...-arm-smmuv3-Move-reset-to-exit-phase.patch | 123 ++ ...ip-automatic-zero-init-of-large-arra.patch | 10 +- ...-skip-automatic-zero-init-of-large-a.patch | 10 +- ...skip-automatic-zero-init-of-large-ar.patch | 10 +- ...p-automatic-zero-init-of-large-array.patch | 10 +- ..._88w8618-skip-automatic-zero-init-of.patch | 10 +- ...ip-automatic-zero-init-of-large-arra.patch | 10 +- ...7-skip-automatic-zero-init-of-large-.patch | 10 +- ...ole-lm-skip-automatic-zero-init-of-l.patch | 10 +- ...e_vga-skip-automatic-zero-init-of-la.patch | 10 +- ...dma-skip-automatic-zero-init-of-larg.patch | 10 +- ...-skip-automatic-zero-init-of-large-a.patch | 10 +- ...-i386-Fix-machine-type-compatibility.patch | 18 +- ...u-Allow-migration-when-explicitly-cr.patch | 117 ++ ...u-Assign-pci-id-0x1419-for-the-AMD-I.patch | 57 + ...u-Isolate-AMDVI-PCI-from-amd-iommu-d.patch | 267 ++++ ...intel-iommu-Migrate-to-3-phase-reset.patch | 96 ++ ...ace-skip-automatic-zero-init-of-larg.patch | 10 +- ...kip-automatic-zero-init-of-large-arr.patch | 10 +- ...p-automatic-zero-init-of-large-array.patch | 10 +- ...t-skip-automatic-zero-init-of-large-.patch | 10 +- ...p-automatic-zero-init-of-large-array.patch | 10 +- ...p-automatic-zero-init-of-large-array.patch | 12 +- ...sic-support-for-PCI-power-management.patch | 242 ++++ ...m-hw-pci-Rename-has_power-to-enabled.patch | 130 ++ ..._proxy-skip-automatic-zero-init-of-l.patch | 10 +- ...-device-Convert-to-three-phase-reset.patch | 63 + ...tio-ccw-Convert-to-three-phase-reset.patch | 92 ++ ...ice-Fix-memory-leak-in-loadparm-sett.patch | 47 + ...5a-skip-automatic-zero-init-of-large.patch | 10 +- ...skip-automatic-zero-init-of-large-ar.patch | 10 +- ...p-automatic-zero-init-of-large-array.patch | 10 +- ...skip-automatic-zero-init-of-large-ar.patch | 10 +- ...dd-a-trace-point-in-vfio_reset_handl.patch | 61 + .../kvm-hw-vfio-pci-Re-order-pre-reset.patch | 74 + ...nclude-md-stubs-in-case-CONFIG_VIRTI.patch | 59 + ...-avoid-cost-of-ftrivial-auto-var-ini.patch | 10 +- ...irtio-iommu-Migrate-to-3-phase-reset.patch | 96 ++ .../kvm-i386-Introduce-tdx-guest-object.patch | 214 +++ ...ed-parameter-uint32_t-bit-in-feature.patch | 62 + ...-i386-apic-Skip-kvm_apic_put-for-TDX.patch | 63 + ...ce-x86_confidential_guest_check_feat.patch | 80 ++ ...mask_cpuid_features-to-adjust_cpuid_.patch | 116 ++ ...u-Cleanup-host_cpu_max_instance_init.patch | 49 + ...date-the-helper-to-get-Host-s-vendor.patch | 78 + ...-Drop-cores_per_pkg-in-cpu_x86_cpuid.patch | 55 + ...e-check-of-phys_bits-in-host_cpu_rea.patch | 79 ++ ...e-variable-smp_cores-and-smp_threads.patch | 68 + ...-a-common-fucntion-to-setup-value-of.patch | 112 ++ ...heck-of-CPUID_EXT3_TOPOEXT-against-t.patch | 80 ++ ...ce-enable_cpuid_0x1f-to-force-exposi.patch | 105 ++ ...justment-of-CPUID_EXT_PDCM-before-fe.patch | 59 + ...6_ext_save_areas-initialization-to-..patch | 91 ++ ...enable_cpuid_0x1f-to-force_cpuid_0x1.patch | 73 + ...-track-CPUID_EXT3_CMP_LEG-in-env-fea.patch | 69 + ...CPUID_HT-in-x86_cpu_expand_features-.patch | 59 + ...-X86CPUTopoInfo-directly-in-CPUX86St.patch | 301 ++++ ...ce-x86_confidential_guest_cpu_instan.patch | 87 ++ ...m-i386-tdvf-Fix-build-on-32-bit-host.patch | 55 + ...duce-function-to-parse-TDVF-metadata.patch | 315 +++++ ...F-memory-via-KVM_TDX_INIT_MEM_REGION.patch | 106 ++ ...-TDX-fixed1-bits-to-supported-CPUIDs.patch | 249 ++++ ...-tdx-Add-XFD-to-supported-bit-of-TDX.patch | 60 + ...perty-sept-ve-disable-for-tdx-guest-.patch | 113 ++ ...ported-CPUID-bits-related-to-TD-Attr.patch | 144 ++ ...supported-CPUID-bits-relates-to-XFAM.patch | 220 +++ ...M_TDX_INIT_VCPU-to-initialize-TDX-vc.patch | 66 + ...-the-error-message-of-mrconfigid-mro.patch | 75 + ...efine-supported-KVM-features-for-TDX.patch | 80 ++ ...kvm-i386-tdx-Disable-PIC-for-TDX-VMs.patch | 56 + ...kvm-i386-tdx-Disable-SMM-for-TDX-VMs.patch | 61 + ...-Don-t-initialize-pc.rom-for-TDX-VMs.patch | 78 + ...86-tdx-Don-t-mask-off-CPUID_EXT_PDCM.patch | 59 + ...-Don-t-synchronize-guest-tsc-for-TDs.patch | 46 + ...x-Don-t-treat-SYSCALL-as-unavailable.patch | 60 + ...le-user-exit-on-KVM_HC_MAP_GPA_RANGE.patch | 55 + ...nd-exit-when-named-cpu-model-is-requ.patch | 59 + ...Fetch-and-validate-CPUID-of-TD-guest.patch | 233 +++ SOURCES/kvm-i386-tdx-Finalize-TDX-VM.patch | 44 + ...vm-i386-tdx-Fix-build-on-32-bit-host.patch | 121 ++ ...86-tdx-Fix-the-report-of-gpa-in-QAPI.patch | 78 + ...-typo-of-the-comment-of-struct-TdxGu.patch | 50 + ...m-i386-tdx-Force-exposing-CPUID-0x1f.patch | 45 + ..._capabilities-via-KVM_TDX_CAPABILITI.patch | 193 +++ ...dx-Handle-KVM_SYSTEM_EVENT_TDX_FATAL.patch | 146 ++ ...lement-adjust_cpuid_features-for-TDX.patch | 179 +++ ...nt-tdx_kvm_init-to-initialize-TDX-VM.patch | 92 ++ ...6-tdx-Implement-tdx_kvm_type-for-TDX.patch | 76 + ...plement-user-specified-tsc-frequency.patch | 101 ++ ...tialize-TDX-before-creating-TD-vcpus.patch | 283 ++++ ...ce-is_tdx_vm-helper-and-cache-tdx_gu.patch | 104 ++ .../kvm-i386-tdx-Make-invtsc-default-on.patch | 42 + ...-Make-sept_ve_disable-set-by-default.patch | 48 + ...nfigure-MSR_IA32_UCODE_REV-in-kvm_in.patch | 92 ++ ...6-tdx-Parse-TDVF-metadata-for-TDX-VM.patch | 116 ++ ...enumeration-of-GetQuote-in-tdx_handl.patch | 70 + ...move-task-watch-only-when-it-s-valid.patch | 50 + ...the-redundant-qemu_mutex_init-tdx-lo.patch | 50 + ...C-bus-rate-to-match-with-what-TDX-mo.patch | 78 + ...nd-check-kernel_irqchip-mode-for-TDX.patch | 64 + ..._readonly_mem_enabled-to-false-for-T.patch | 54 + ...ue-of-GetTdVmCallInfo-based-on-capab.patch | 60 + .../kvm-i386-tdx-Setup-the-TD-HOB-list.patch | 264 ++++ ...-user-configurable-mrconfigid-mrowne.patch | 226 +++ ...386-tdx-Track-RAM-entries-for-TDX-VM.patch | 223 +++ ...em_ptr-for-each-firmware-entry-of-TD.patch | 151 ++ .../kvm-i386-tdx-Validate-TD-attributes.patch | 106 ++ ...alidate-phys_bits-against-host-value.patch | 88 ++ ...U-features-up-with-attributes-of-TD-.patch | 78 + ...X_REPORT_FATAL_ERROR-with-GuestPanic.patch | 240 ++++ ...86-tdx-handle-TDG.VP.VMCALL-GetQuote.patch | 798 +++++++++++ ...handle-TDG.VP.VMCALL-GetTdVmCallInfo.patch | 111 ++ ...TDVMCALL_SETUP_EVENT_NOTIFY_INTERRUP.patch | 197 +++ ...-tdx-implement-tdx_cpu_instance_init.patch | 50 + .../kvm-i386-tdx-load-TDVF-for-TD-guest.patch | 105 ++ ...troduce-helpers-for-various-topology.patch | 101 ++ ...date-the-comment-of-x86_apicid_from_.patch | 49 + ...piler-add-QEMU_UNINITIALIZED-attribu.patch | 10 +- ...truct-copy-in-qio_dns_resolver_looku.patch | 73 + ...ter-free-in-websocket-handshake-code.patch | 189 +++ ...ock-resource-release-to-close-method.patch | 84 ++ ...ts-Improve-iotest-194-to-mirror-data.patch | 42 + ...ts-common.rc-add-disk_usage-function.patch | 68 + ...-Check-KVM_CAP_MAX_VCPUS-at-vm-level.patch | 45 + ...m-Introduce-kvm_arch_pre_create_vcpu.patch | 187 +++ ...m_filter_msr-and-related-definitions.patch | 90 ++ .../kvm-kvm-remove-unnecessary-ifdef.patch | 52 + ...ux-headers-Update-to-Linux-v6.14-rc3.patch | 495 +++++++ ...ux-headers-Update-to-Linux-v6.15-rc3.patch | 949 +++++++++++++ ...ux-headers-update-from-6.15-kvm-next.patch | 126 ++ ...mory_region_set_ram_discard_manager-.patch | 160 +++ ...helper-to-get-intersection-of-a-Memo.patch | 157 ++ ...-definiton-of-ReplayRamPopulate-and-.patch | 277 ++++ ...add-valgrind-option-en-dis-able-valg.patch | 110 ++ ...F-for-incoming-migration-on-Migratio.patch | 180 +++ ...py-Spatial-locality-page-hint-for-pr.patch | 237 ++++ ...-override-to-declare-target-already-.patch | 295 ++++ ...Drop-redundant-zero_target-parameter.patch | 241 ++++ SOURCES/kvm-mirror-Minor-refactoring.patch | 92 ++ ...-sync-mode-rather-than-bool-to-inter.patch | 139 ++ ...O-when-destination-is-detect-zeroes-.patch | 58 + ...zeroing-destination-if-it-is-already.patch | 180 +++ ...ing-zeroes-when-target-is-already-ze.patch | 355 +++++ ...p-automatic-zero-init-of-large-array.patch | 10 +- ...p-automatic-zero-init-of-large-array.patch | 10 +- ...dd-QAPI-events-to-report-connection-.patch | 16 +- ...ci-Use-PCI-PM-capability-initializer.patch | 153 ++ ...-pcie-virtio-Remove-redundant-pm_cap.patch | 99 ++ ...coordinated-discarding-of-RAM-with-g.patch | 129 ++ ...physmem-replace-assertion-with-error.patch | 72 + ...a-implement-a-guest-get-load-command.patch | 12 +- ...se-order-of-instance_post_init-calls.patch | 79 ++ ...utes-Introduce-RamBlockAttributes-to.patch | 613 ++++++++ ...drv_get_specific_info-implementation.patch | 14 +- ...vm-redhat-Enable-virtio-mem-on-s390x.patch | 36 + ...hat-allow-5-level-paging-for-TDX-VMs.patch | 33 + SOURCES/kvm-redhat-enable-CONFIG_TDX.patch | 33 + ...86-add-CPUID-and-MSR-bits-from-Clear.patch | 99 ++ SOURCES/kvm-reset-Add-RESET_TYPE_WAKEUP.patch | 94 ++ ...ype-for-qemu_devices_reset-and-Machi.patch | 360 +++++ ...-rocker-do-not-pollute-the-namespace.patch | 241 ++++ ...90x-Fix-leak-in-machine_set_loadparm.patch | 60 + ...390x-introduce-s390_get_memory_limit.patch | 144 ++ ...pport-for-guests-that-request-direct.patch | 256 ++++ ...te-QEMU-supports-relaxed-translation.patch | 72 + ...-s390x-pv-prepare-for-memory-devices.patch | 46 + ...s390x-remember-the-maximum-page-size.patch | 107 ++ ...-s390-virtio-hcall-to-s390-hypercall.patch | 113 ++ ...call-introduce-DIAG500-STORAGE_LIMIT.patch | 99 ++ ...390-skeys-prepare-for-memory-devices.patch | 56 + ...rib-kvm-prepare-for-memory-devices-a.patch | 155 ++ ...o-ccw-don-t-crash-on-weird-RAM-sizes.patch | 63 + ...o-ccw-move-setting-the-maximum-guest.patch | 140 ++ ...irtio-ccw-prepare-for-memory-devices.patch | 117 ++ ...o-hcall-prepare-for-more-diag500-hyp.patch | 163 +++ ...o-hcall-remove-hypercall-registratio.patch | 296 ++++ ...-add-support-for-virtio-based-memory.patch | 423 ++++++ SOURCES/kvm-s390x-virtio-mem-support.patch | 459 ++++++ ...error-from-qemu-trace-stap-on-missin.patch | 90 ++ ...arget-i386-Add-PerfMonV2-feature-bit.patch | 105 ++ ...couple-of-feature-bits-in-CPUID_Fn80.patch | 83 ++ ...386-Add-support-for-EPYC-Turin-model.patch | 200 +++ ...ble-fdp-excptn-only-and-zero-fcs-fds.patch | 75 + ...xclude-hv-syndbg-from-hv-passthrough.patch | 102 ++ ...se-IBPB-BRTYPE-and-SBPB-CPUID-bits-t.patch | 70 + ...se-bits-related-to-SRSO-vulnerabilit.patch | 84 ++ ...conditional-CONFIG_SYNDBG-enablement.patch | 108 ++ ...-invtsc-migratable-when-user-sets-ts.patch | 73 + ...t-CPUID-subleaf-info-for-unsupported.patch | 47 + ...ve-AccelCPUClass-cpu_class_init-need.patch | 122 ++ ...te-EPYC-CPU-model-for-Cache-property.patch | 147 ++ ...te-EPYC-Genoa-for-Cache-property-per.patch | 167 +++ ...te-EPYC-Milan-CPU-model-for-Cache-pr.patch | 146 ++ ...te-EPYC-Rome-CPU-model-for-Cache-pro.patch | 147 ++ ...w-reordering-max_x86_cpu_initfn-vs-a.patch | 84 ++ ...e-host_cpu_instance_init-and-host_cp.patch | 103 ++ ...-accel_cpu_instance_init-to-.instanc.patch | 68 + ...rget-i386-move-max_features-to-class.patch | 145 ++ ...-whpx-add-accel-CPU-class-that-sets-.patch | 164 +++ ...-Reduce-system-specific-declarations.patch | 99 ++ ...-fix-locking-for-interrupt-injection.patch | 62 + ...-Convert-CPU-to-Resettable-interface.patch | 284 ++++ ...est-mirror-sparse-for-recent-patches.patch | 545 +++++++ ...util-sockets-fix-mem-leak-on-error-o.patch | 53 + ...splay-update-interval-when-VM-state-.patch | 12 +- ...ate-Linux-headers-to-KVM-tree-master.patch | 62 + ...vm-update-Linux-headers-to-v6.16-rc3.patch | 497 +++++++ ...s-Add-support-for-keep-alive-flag-to.patch | 86 ++ ...s-Introduce-inet-socket-options-cont.patch | 314 ++++ ...s-Refactor-inet_parse-to-use-QemuOpt.patch | 461 ++++++ ...s-Refactor-setting-client-sockopts-i.patch | 83 ++ ...s-Refactor-success-and-failure-paths.patch | 141 ++ SOURCES/kvm-vfio-helpers-Align-mmaps.patch | 19 +- ...actor-vfio_region_mmap-error-handlin.patch | 19 +- .../kvm-vfio-pci-Delete-local-pm_cap.patch | 81 ++ ...-kconfig-memory-devices-are-PCI-only.patch | 87 ++ ...upport-for-suspend-wake-up-with-plug.patch | 79 ++ ...ew-Resettable-framework-instead-of-L.patch | 141 ++ ...-warn-about-THP-sizes-on-a-kernel-wi.patch | 59 + ...g-memory-only-during-system-resets-n.patch | 258 ++++ ...tio-net-disable-USO-for-virt-rhel9.6.patch | 135 ++ SOURCES/qemu-ga.sysconfig | 2 +- SPECS/qemu-kvm.spec | 1260 +++++++++++++++-- 250 files changed, 29951 insertions(+), 384 deletions(-) create mode 100644 SOURCES/kvm-Enable-amd-iommu-device.patch create mode 100644 SOURCES/kvm-amd_iommu-Add-support-for-pass-though-mode.patch create mode 100644 SOURCES/kvm-amd_iommu-Check-APIC-ID-255-for-XTSup.patch create mode 100644 SOURCES/kvm-amd_iommu-Rename-variable-mmio-to-mr_mmio.patch create mode 100644 SOURCES/kvm-amd_iommu-Send-notification-when-invalidate-interrup.patch create mode 100644 SOURCES/kvm-amd_iommu-Use-shared-memory-region-for-Interrupt-Rem.patch create mode 100644 SOURCES/kvm-arm-Use-arm_virt_compat_set-to-apply-the-compat.patch create mode 100644 SOURCES/kvm-arm-kvm-report-registers-we-failed-to-set.patch create mode 100644 SOURCES/kvm-block-Add-new-bdrv_co_is_all_zeroes-function.patch create mode 100644 SOURCES/kvm-block-Expand-block-status-mode-from-bool-to-flags.patch create mode 100644 SOURCES/kvm-block-Let-bdrv_co_is_zero_fast-consolidate-adjacent-.patch create mode 100644 SOURCES/kvm-cpu-Don-t-set-vcpu_dirty-when-guest_state_protected.patch create mode 100644 SOURCES/kvm-cpu-Remove-nr_cores-from-struct-CPUState.patch create mode 100644 SOURCES/kvm-crypto-Define-macros-for-hash-algorithm-digest-lengt.patch create mode 100644 SOURCES/kvm-docs-Add-TDX-documentation.patch create mode 100644 SOURCES/kvm-docs-devel-reset-Document-reset-expectations-for-DMA.patch create mode 100644 SOURCES/kvm-file-posix-gluster-Handle-zero-block-status-hint-bet.patch create mode 100644 SOURCES/kvm-headers-Add-definitions-from-UEFI-spec-for-volumes-r.patch create mode 100644 SOURCES/kvm-hw-arm-smmuv3-Move-reset-to-exit-phase.patch create mode 100644 SOURCES/kvm-hw-i386-amd_iommu-Allow-migration-when-explicitly-cr.patch create mode 100644 SOURCES/kvm-hw-i386-amd_iommu-Assign-pci-id-0x1419-for-the-AMD-I.patch create mode 100644 SOURCES/kvm-hw-i386-amd_iommu-Isolate-AMDVI-PCI-from-amd-iommu-d.patch create mode 100644 SOURCES/kvm-hw-i386-intel-iommu-Migrate-to-3-phase-reset.patch create mode 100644 SOURCES/kvm-hw-pci-Basic-support-for-PCI-power-management.patch create mode 100644 SOURCES/kvm-hw-pci-Rename-has_power-to-enabled.patch create mode 100644 SOURCES/kvm-hw-s390-ccw-device-Convert-to-three-phase-reset.patch create mode 100644 SOURCES/kvm-hw-s390-virtio-ccw-Convert-to-three-phase-reset.patch create mode 100644 SOURCES/kvm-hw-s390x-ccw-device-Fix-memory-leak-in-loadparm-sett.patch create mode 100644 SOURCES/kvm-hw-vfio-common-Add-a-trace-point-in-vfio_reset_handl.patch create mode 100644 SOURCES/kvm-hw-vfio-pci-Re-order-pre-reset.patch create mode 100644 SOURCES/kvm-hw-virtio-Also-include-md-stubs-in-case-CONFIG_VIRTI.patch create mode 100644 SOURCES/kvm-hw-virtio-virtio-iommu-Migrate-to-3-phase-reset.patch create mode 100644 SOURCES/kvm-i386-Introduce-tdx-guest-object.patch create mode 100644 SOURCES/kvm-i386-Remove-unused-parameter-uint32_t-bit-in-feature.patch create mode 100644 SOURCES/kvm-i386-apic-Skip-kvm_apic_put-for-TDX.patch create mode 100644 SOURCES/kvm-i386-cgs-Introduce-x86_confidential_guest_check_feat.patch create mode 100644 SOURCES/kvm-i386-cgs-Rename-mask_cpuid_features-to-adjust_cpuid_.patch create mode 100644 SOURCES/kvm-i386-cpu-Cleanup-host_cpu_max_instance_init.patch create mode 100644 SOURCES/kvm-i386-cpu-Consolidate-the-helper-to-get-Host-s-vendor.patch create mode 100644 SOURCES/kvm-i386-cpu-Drop-cores_per_pkg-in-cpu_x86_cpuid.patch create mode 100644 SOURCES/kvm-i386-cpu-Drop-the-check-of-phys_bits-in-host_cpu_rea.patch create mode 100644 SOURCES/kvm-i386-cpu-Drop-the-variable-smp_cores-and-smp_threads.patch create mode 100644 SOURCES/kvm-i386-cpu-Extract-a-common-fucntion-to-setup-value-of.patch create mode 100644 SOURCES/kvm-i386-cpu-Hoist-check-of-CPUID_EXT3_TOPOEXT-against-t.patch create mode 100644 SOURCES/kvm-i386-cpu-Introduce-enable_cpuid_0x1f-to-force-exposi.patch create mode 100644 SOURCES/kvm-i386-cpu-Move-adjustment-of-CPUID_EXT_PDCM-before-fe.patch create mode 100644 SOURCES/kvm-i386-cpu-Move-x86_ext_save_areas-initialization-to-..patch create mode 100644 SOURCES/kvm-i386-cpu-Rename-enable_cpuid_0x1f-to-force_cpuid_0x1.patch create mode 100644 SOURCES/kvm-i386-cpu-Set-and-track-CPUID_EXT3_CMP_LEG-in-env-fea.patch create mode 100644 SOURCES/kvm-i386-cpu-Set-up-CPUID_HT-in-x86_cpu_expand_features-.patch create mode 100644 SOURCES/kvm-i386-cpu-Track-a-X86CPUTopoInfo-directly-in-CPUX86St.patch create mode 100644 SOURCES/kvm-i386-cpu-introduce-x86_confidential_guest_cpu_instan.patch create mode 100644 SOURCES/kvm-i386-tdvf-Fix-build-on-32-bit-host.patch create mode 100644 SOURCES/kvm-i386-tdvf-Introduce-function-to-parse-TDVF-metadata.patch create mode 100644 SOURCES/kvm-i386-tdx-Add-TDVF-memory-via-KVM_TDX_INIT_MEM_REGION.patch create mode 100644 SOURCES/kvm-i386-tdx-Add-TDX-fixed1-bits-to-supported-CPUIDs.patch create mode 100644 SOURCES/kvm-i386-tdx-Add-XFD-to-supported-bit-of-TDX.patch create mode 100644 SOURCES/kvm-i386-tdx-Add-property-sept-ve-disable-for-tdx-guest-.patch create mode 100644 SOURCES/kvm-i386-tdx-Add-supported-CPUID-bits-related-to-TD-Attr.patch create mode 100644 SOURCES/kvm-i386-tdx-Add-supported-CPUID-bits-relates-to-XFAM.patch create mode 100644 SOURCES/kvm-i386-tdx-Call-KVM_TDX_INIT_VCPU-to-initialize-TDX-vc.patch create mode 100644 SOURCES/kvm-i386-tdx-Clarify-the-error-message-of-mrconfigid-mro.patch create mode 100644 SOURCES/kvm-i386-tdx-Define-supported-KVM-features-for-TDX.patch create mode 100644 SOURCES/kvm-i386-tdx-Disable-PIC-for-TDX-VMs.patch create mode 100644 SOURCES/kvm-i386-tdx-Disable-SMM-for-TDX-VMs.patch create mode 100644 SOURCES/kvm-i386-tdx-Don-t-initialize-pc.rom-for-TDX-VMs.patch create mode 100644 SOURCES/kvm-i386-tdx-Don-t-mask-off-CPUID_EXT_PDCM.patch create mode 100644 SOURCES/kvm-i386-tdx-Don-t-synchronize-guest-tsc-for-TDs.patch create mode 100644 SOURCES/kvm-i386-tdx-Don-t-treat-SYSCALL-as-unavailable.patch create mode 100644 SOURCES/kvm-i386-tdx-Enable-user-exit-on-KVM_HC_MAP_GPA_RANGE.patch create mode 100644 SOURCES/kvm-i386-tdx-Error-and-exit-when-named-cpu-model-is-requ.patch create mode 100644 SOURCES/kvm-i386-tdx-Fetch-and-validate-CPUID-of-TD-guest.patch create mode 100644 SOURCES/kvm-i386-tdx-Finalize-TDX-VM.patch create mode 100644 SOURCES/kvm-i386-tdx-Fix-build-on-32-bit-host.patch create mode 100644 SOURCES/kvm-i386-tdx-Fix-the-report-of-gpa-in-QAPI.patch create mode 100644 SOURCES/kvm-i386-tdx-Fix-the-typo-of-the-comment-of-struct-TdxGu.patch create mode 100644 SOURCES/kvm-i386-tdx-Force-exposing-CPUID-0x1f.patch create mode 100644 SOURCES/kvm-i386-tdx-Get-tdx_capabilities-via-KVM_TDX_CAPABILITI.patch create mode 100644 SOURCES/kvm-i386-tdx-Handle-KVM_SYSTEM_EVENT_TDX_FATAL.patch create mode 100644 SOURCES/kvm-i386-tdx-Implement-adjust_cpuid_features-for-TDX.patch create mode 100644 SOURCES/kvm-i386-tdx-Implement-tdx_kvm_init-to-initialize-TDX-VM.patch create mode 100644 SOURCES/kvm-i386-tdx-Implement-tdx_kvm_type-for-TDX.patch create mode 100644 SOURCES/kvm-i386-tdx-Implement-user-specified-tsc-frequency.patch create mode 100644 SOURCES/kvm-i386-tdx-Initialize-TDX-before-creating-TD-vcpus.patch create mode 100644 SOURCES/kvm-i386-tdx-Introduce-is_tdx_vm-helper-and-cache-tdx_gu.patch create mode 100644 SOURCES/kvm-i386-tdx-Make-invtsc-default-on.patch create mode 100644 SOURCES/kvm-i386-tdx-Make-sept_ve_disable-set-by-default.patch create mode 100644 SOURCES/kvm-i386-tdx-Only-configure-MSR_IA32_UCODE_REV-in-kvm_in.patch create mode 100644 SOURCES/kvm-i386-tdx-Parse-TDVF-metadata-for-TDX-VM.patch create mode 100644 SOURCES/kvm-i386-tdx-Remove-enumeration-of-GetQuote-in-tdx_handl.patch create mode 100644 SOURCES/kvm-i386-tdx-Remove-task-watch-only-when-it-s-valid.patch create mode 100644 SOURCES/kvm-i386-tdx-Remove-the-redundant-qemu_mutex_init-tdx-lo.patch create mode 100644 SOURCES/kvm-i386-tdx-Set-APIC-bus-rate-to-match-with-what-TDX-mo.patch create mode 100644 SOURCES/kvm-i386-tdx-Set-and-check-kernel_irqchip-mode-for-TDX.patch create mode 100644 SOURCES/kvm-i386-tdx-Set-kvm_readonly_mem_enabled-to-false-for-T.patch create mode 100644 SOURCES/kvm-i386-tdx-Set-value-of-GetTdVmCallInfo-based-on-capab.patch create mode 100644 SOURCES/kvm-i386-tdx-Setup-the-TD-HOB-list.patch create mode 100644 SOURCES/kvm-i386-tdx-Support-user-configurable-mrconfigid-mrowne.patch create mode 100644 SOURCES/kvm-i386-tdx-Track-RAM-entries-for-TDX-VM.patch create mode 100644 SOURCES/kvm-i386-tdx-Track-mem_ptr-for-each-firmware-entry-of-TD.patch create mode 100644 SOURCES/kvm-i386-tdx-Validate-TD-attributes.patch create mode 100644 SOURCES/kvm-i386-tdx-Validate-phys_bits-against-host-value.patch create mode 100644 SOURCES/kvm-i386-tdx-Wire-CPU-features-up-with-attributes-of-TD-.patch create mode 100644 SOURCES/kvm-i386-tdx-Wire-TDX_REPORT_FATAL_ERROR-with-GuestPanic.patch create mode 100644 SOURCES/kvm-i386-tdx-handle-TDG.VP.VMCALL-GetQuote.patch create mode 100644 SOURCES/kvm-i386-tdx-handle-TDG.VP.VMCALL-GetTdVmCallInfo.patch create mode 100644 SOURCES/kvm-i386-tdx-handle-TDVMCALL_SETUP_EVENT_NOTIFY_INTERRUP.patch create mode 100644 SOURCES/kvm-i386-tdx-implement-tdx_cpu_instance_init.patch create mode 100644 SOURCES/kvm-i386-tdx-load-TDVF-for-TD-guest.patch create mode 100644 SOURCES/kvm-i386-topology-Introduce-helpers-for-various-topology.patch create mode 100644 SOURCES/kvm-i386-topology-Update-the-comment-of-x86_apicid_from_.patch create mode 100644 SOURCES/kvm-io-Fix-partial-struct-copy-in-qio_dns_resolver_looku.patch create mode 100644 SOURCES/kvm-io-fix-use-after-free-in-websocket-handshake-code.patch create mode 100644 SOURCES/kvm-io-move-websock-resource-release-to-close-method.patch create mode 100644 SOURCES/kvm-iotests-Improve-iotest-194-to-mirror-data.patch create mode 100644 SOURCES/kvm-iotests-common.rc-add-disk_usage-function.patch create mode 100644 SOURCES/kvm-kvm-Check-KVM_CAP_MAX_VCPUS-at-vm-level.patch create mode 100644 SOURCES/kvm-kvm-Introduce-kvm_arch_pre_create_vcpu.patch create mode 100644 SOURCES/kvm-kvm-i386-make-kvm_filter_msr-and-related-definitions.patch create mode 100644 SOURCES/kvm-kvm-remove-unnecessary-ifdef.patch create mode 100644 SOURCES/kvm-linux-headers-Update-to-Linux-v6.14-rc3.patch create mode 100644 SOURCES/kvm-linux-headers-Update-to-Linux-v6.15-rc3.patch create mode 100644 SOURCES/kvm-linux-headers-update-from-6.15-kvm-next.patch create mode 100644 SOURCES/kvm-memory-Change-memory_region_set_ram_discard_manager-.patch create mode 100644 SOURCES/kvm-memory-Export-a-helper-to-get-intersection-of-a-Memo.patch create mode 100644 SOURCES/kvm-memory-Unify-the-definiton-of-ReplayRamPopulate-and-.patch create mode 100644 SOURCES/kvm-meson-configure-add-valgrind-option-en-dis-able-valg.patch create mode 100644 SOURCES/kvm-migration-Fix-UAF-for-incoming-migration-on-Migratio.patch create mode 100644 SOURCES/kvm-migration-postcopy-Spatial-locality-page-hint-for-pr.patch create mode 100644 SOURCES/kvm-mirror-Allow-QMP-override-to-declare-target-already-.patch create mode 100644 SOURCES/kvm-mirror-Drop-redundant-zero_target-parameter.patch create mode 100644 SOURCES/kvm-mirror-Minor-refactoring.patch create mode 100644 SOURCES/kvm-mirror-Pass-full-sync-mode-rather-than-bool-to-inter.patch create mode 100644 SOURCES/kvm-mirror-Reduce-I-O-when-destination-is-detect-zeroes-.patch create mode 100644 SOURCES/kvm-mirror-Skip-pre-zeroing-destination-if-it-is-already.patch create mode 100644 SOURCES/kvm-mirror-Skip-writing-zeroes-when-target-is-already-ze.patch create mode 100644 SOURCES/kvm-pci-Use-PCI-PM-capability-initializer.patch create mode 100644 SOURCES/kvm-pcie-virtio-Remove-redundant-pm_cap.patch create mode 100644 SOURCES/kvm-physmem-Support-coordinated-discarding-of-RAM-with-g.patch create mode 100644 SOURCES/kvm-physmem-replace-assertion-with-error.patch create mode 100644 SOURCES/kvm-qom-reverse-order-of-instance_post_init-calls.patch create mode 100644 SOURCES/kvm-ram-block-attributes-Introduce-RamBlockAttributes-to.patch create mode 100644 SOURCES/kvm-redhat-Enable-virtio-mem-on-s390x.patch create mode 100644 SOURCES/kvm-redhat-allow-5-level-paging-for-TDX-VMs.patch create mode 100644 SOURCES/kvm-redhat-enable-CONFIG_TDX.patch create mode 100644 SOURCES/kvm-redhat-target-i386-add-CPUID-and-MSR-bits-from-Clear.patch create mode 100644 SOURCES/kvm-reset-Add-RESET_TYPE_WAKEUP.patch create mode 100644 SOURCES/kvm-reset-Use-ResetType-for-qemu_devices_reset-and-Machi.patch create mode 100644 SOURCES/kvm-rocker-do-not-pollute-the-namespace.patch create mode 100644 SOURCES/kvm-s390x-Fix-leak-in-machine_set_loadparm.patch create mode 100644 SOURCES/kvm-s390x-introduce-s390_get_memory_limit.patch create mode 100644 SOURCES/kvm-s390x-pci-add-support-for-guests-that-request-direct.patch create mode 100644 SOURCES/kvm-s390x-pci-indicate-QEMU-supports-relaxed-translation.patch create mode 100644 SOURCES/kvm-s390x-pv-prepare-for-memory-devices.patch create mode 100644 SOURCES/kvm-s390x-remember-the-maximum-page-size.patch create mode 100644 SOURCES/kvm-s390x-rename-s390-virtio-hcall-to-s390-hypercall.patch create mode 100644 SOURCES/kvm-s390x-s390-hypercall-introduce-DIAG500-STORAGE_LIMIT.patch create mode 100644 SOURCES/kvm-s390x-s390-skeys-prepare-for-memory-devices.patch create mode 100644 SOURCES/kvm-s390x-s390-stattrib-kvm-prepare-for-memory-devices-a.patch create mode 100644 SOURCES/kvm-s390x-s390-virtio-ccw-don-t-crash-on-weird-RAM-sizes.patch create mode 100644 SOURCES/kvm-s390x-s390-virtio-ccw-move-setting-the-maximum-guest.patch create mode 100644 SOURCES/kvm-s390x-s390-virtio-ccw-prepare-for-memory-devices.patch create mode 100644 SOURCES/kvm-s390x-s390-virtio-hcall-prepare-for-more-diag500-hyp.patch create mode 100644 SOURCES/kvm-s390x-s390-virtio-hcall-remove-hypercall-registratio.patch create mode 100644 SOURCES/kvm-s390x-virtio-ccw-add-support-for-virtio-based-memory.patch create mode 100644 SOURCES/kvm-s390x-virtio-mem-support.patch create mode 100644 SOURCES/kvm-scripts-improve-error-from-qemu-trace-stap-on-missin.patch create mode 100644 SOURCES/kvm-target-i386-Add-PerfMonV2-feature-bit.patch create mode 100644 SOURCES/kvm-target-i386-Add-couple-of-feature-bits-in-CPUID_Fn80.patch create mode 100644 SOURCES/kvm-target-i386-Add-support-for-EPYC-Turin-model.patch create mode 100644 SOURCES/kvm-target-i386-Enable-fdp-excptn-only-and-zero-fcs-fds.patch create mode 100644 SOURCES/kvm-target-i386-Exclude-hv-syndbg-from-hv-passthrough.patch create mode 100644 SOURCES/kvm-target-i386-Expose-IBPB-BRTYPE-and-SBPB-CPUID-bits-t.patch create mode 100644 SOURCES/kvm-target-i386-Expose-bits-related-to-SRSO-vulnerabilit.patch create mode 100644 SOURCES/kvm-target-i386-Fix-conditional-CONFIG_SYNDBG-enablement.patch create mode 100644 SOURCES/kvm-target-i386-Make-invtsc-migratable-when-user-sets-ts.patch create mode 100644 SOURCES/kvm-target-i386-Print-CPUID-subleaf-info-for-unsupported.patch create mode 100644 SOURCES/kvm-target-i386-Remove-AccelCPUClass-cpu_class_init-need.patch create mode 100644 SOURCES/kvm-target-i386-Update-EPYC-CPU-model-for-Cache-property.patch create mode 100644 SOURCES/kvm-target-i386-Update-EPYC-Genoa-for-Cache-property-per.patch create mode 100644 SOURCES/kvm-target-i386-Update-EPYC-Milan-CPU-model-for-Cache-pr.patch create mode 100644 SOURCES/kvm-target-i386-Update-EPYC-Rome-CPU-model-for-Cache-pro.patch create mode 100644 SOURCES/kvm-target-i386-allow-reordering-max_x86_cpu_initfn-vs-a.patch create mode 100644 SOURCES/kvm-target-i386-merge-host_cpu_instance_init-and-host_cp.patch create mode 100644 SOURCES/kvm-target-i386-move-accel_cpu_instance_init-to-.instanc.patch create mode 100644 SOURCES/kvm-target-i386-move-max_features-to-class.patch create mode 100644 SOURCES/kvm-target-i386-nvmm-whpx-add-accel-CPU-class-that-sets-.patch create mode 100644 SOURCES/kvm-target-i386-sev-Reduce-system-specific-declarations.patch create mode 100644 SOURCES/kvm-target-i386-tdx-fix-locking-for-interrupt-injection.patch create mode 100644 SOURCES/kvm-target-s390-Convert-CPU-to-Resettable-interface.patch create mode 100644 SOURCES/kvm-tests-Add-iotest-mirror-sparse-for-recent-patches.patch create mode 100644 SOURCES/kvm-tests-unit-test-util-sockets-fix-mem-leak-on-error-o.patch create mode 100644 SOURCES/kvm-update-Linux-headers-to-KVM-tree-master.patch create mode 100644 SOURCES/kvm-update-Linux-headers-to-v6.16-rc3.patch create mode 100644 SOURCES/kvm-util-qemu-sockets-Add-support-for-keep-alive-flag-to.patch create mode 100644 SOURCES/kvm-util-qemu-sockets-Introduce-inet-socket-options-cont.patch create mode 100644 SOURCES/kvm-util-qemu-sockets-Refactor-inet_parse-to-use-QemuOpt.patch create mode 100644 SOURCES/kvm-util-qemu-sockets-Refactor-setting-client-sockopts-i.patch create mode 100644 SOURCES/kvm-util-qemu-sockets-Refactor-success-and-failure-paths.patch create mode 100644 SOURCES/kvm-vfio-pci-Delete-local-pm_cap.patch create mode 100644 SOURCES/kvm-virtio-kconfig-memory-devices-are-PCI-only.patch create mode 100644 SOURCES/kvm-virtio-mem-Add-support-for-suspend-wake-up-with-plug.patch create mode 100644 SOURCES/kvm-virtio-mem-Use-new-Resettable-framework-instead-of-L.patch create mode 100644 SOURCES/kvm-virtio-mem-don-t-warn-about-THP-sizes-on-a-kernel-wi.patch create mode 100644 SOURCES/kvm-virtio-mem-unplug-memory-only-during-system-resets-n.patch create mode 100644 SOURCES/kvm-virtio-net-disable-USO-for-virt-rhel9.6.patch diff --git a/SOURCES/kvm-Enable-amd-iommu-device.patch b/SOURCES/kvm-Enable-amd-iommu-device.patch new file mode 100644 index 0000000..44846a5 --- /dev/null +++ b/SOURCES/kvm-Enable-amd-iommu-device.patch @@ -0,0 +1,38 @@ +From 0608561efc441f234d9aaf45f1867ffb5c43cffe Mon Sep 17 00:00:00 2001 +From: John Allen +Date: Wed, 11 Jun 2025 15:41:14 -0500 +Subject: [PATCH 26/57] Enable amd-iommu device + +RH-Author: John Allen +RH-MergeRequest: 380: Add ability to manually specify the AMDVI-PCI device +RH-Jira: RHEL-70925 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [3/3] 852500a18275e14bcd94d598ccd0ee33b76578dc (johnalle/qemu-kvm-fork) + +Now that the amdvi-pci device that amd-iommu creates can be specified +manually, amd-iommu device can be enabled. + +JIRA: https://issues.redhat.com/browse/RHEL-70925 + +Upstream: RHEL ONLY + +Signed-off-by: John Allen +--- + configs/devices/x86_64-softmmu/x86_64-rh-devices.mak | 1 + + 1 file changed, 1 insertion(+) + +diff --git a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak +index 3e5f693b62..2b15fdc2db 100644 +--- a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak ++++ b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak +@@ -97,6 +97,7 @@ CONFIG_VIRTIO_MEM=y + CONFIG_VIRTIO_PCI=y + CONFIG_VIRTIO_VGA=y + CONFIG_VIRTIO_IOMMU=y ++CONFIG_AMD_IOMMU=y + CONFIG_VMMOUSE=y + CONFIG_VMPORT=y + CONFIG_VTD=y +-- +2.39.3 + diff --git a/SOURCES/kvm-amd_iommu-Add-support-for-pass-though-mode.patch b/SOURCES/kvm-amd_iommu-Add-support-for-pass-though-mode.patch new file mode 100644 index 0000000..b0038a7 --- /dev/null +++ b/SOURCES/kvm-amd_iommu-Add-support-for-pass-though-mode.patch @@ -0,0 +1,141 @@ +From 4114553452f7187283aefa001bc8342fc65b6b72 Mon Sep 17 00:00:00 2001 +From: John Allen +Date: Wed, 11 Dec 2024 15:06:48 -0600 +Subject: [PATCH 04/57] amd_iommu: Add support for pass though mode + +RH-Author: John Allen +RH-MergeRequest: 303: Interrupt Remap support for emulated amd viommu +RH-Jira: RHEL-66202 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [2/5] 0434fefd554baf27fb9d93026af513c621f8cdb0 (johnalle/qemu-kvm-fork) + +JIRA: https://issues.redhat.com/browse/RHEL-66202 + +commit c1f46999ef506d9854534560a94d02cf3cf9edd1 +Author: Suravee Suthikulpanit +Date: Fri Sep 27 12:29:10 2024 -0500 + + amd_iommu: Add support for pass though mode + + Introduce 'nodma' shared memory region to support PT mode + so that for each device, we only create an alias to shared memory + region when DMA-remapping is disabled. + + Reviewed-by: Alejandro Jimenez + Signed-off-by: Suravee Suthikulpanit + Signed-off-by: Santosh Shukla + Message-Id: <20240927172913.121477-3-santosh.shukla@amd.com> + Reviewed-by: Michael S. Tsirkin + Signed-off-by: Michael S. Tsirkin + +Signed-off-by: John Allen +--- + hw/i386/amd_iommu.c | 49 ++++++++++++++++++++++++++++++++++++--------- + hw/i386/amd_iommu.h | 2 ++ + 2 files changed, 42 insertions(+), 9 deletions(-) + +diff --git a/hw/i386/amd_iommu.c b/hw/i386/amd_iommu.c +index 148b5ee51d..567cb8adc9 100644 +--- a/hw/i386/amd_iommu.c ++++ b/hw/i386/amd_iommu.c +@@ -60,8 +60,9 @@ struct AMDVIAddressSpace { + uint8_t bus_num; /* bus number */ + uint8_t devfn; /* device function */ + AMDVIState *iommu_state; /* AMDVI - one per machine */ +- MemoryRegion root; /* AMDVI Root memory map region */ ++ MemoryRegion root; /* AMDVI Root memory map region */ + IOMMUMemoryRegion iommu; /* Device's address translation region */ ++ MemoryRegion iommu_nodma; /* Alias of shared nodma memory region */ + MemoryRegion iommu_ir; /* Device's interrupt remapping region */ + AddressSpace as; /* device's corresponding address space */ + }; +@@ -1412,6 +1413,7 @@ static AddressSpace *amdvi_host_dma_iommu(PCIBus *bus, void *opaque, int devfn) + AMDVIState *s = opaque; + AMDVIAddressSpace **iommu_as, *amdvi_dev_as; + int bus_num = pci_bus_num(bus); ++ X86IOMMUState *x86_iommu = X86_IOMMU_DEVICE(s); + + iommu_as = s->address_spaces[bus_num]; + +@@ -1436,13 +1438,13 @@ static AddressSpace *amdvi_host_dma_iommu(PCIBus *bus, void *opaque, int devfn) + * Memory region relationships looks like (Address range shows + * only lower 32 bits to make it short in length...): + * +- * |-----------------+-------------------+----------| +- * | Name | Address range | Priority | +- * |-----------------+-------------------+----------+ +- * | amdvi_root | 00000000-ffffffff | 0 | +- * | amdvi_iommu | 00000000-ffffffff | 1 | +- * | amdvi_iommu_ir | fee00000-feefffff | 64 | +- * |-----------------+-------------------+----------| ++ * |--------------------+-------------------+----------| ++ * | Name | Address range | Priority | ++ * |--------------------+-------------------+----------+ ++ * | amdvi-root | 00000000-ffffffff | 0 | ++ * | amdvi-iommu_nodma | 00000000-ffffffff | 0 | ++ * | amdvi-iommu_ir | fee00000-feefffff | 64 | ++ * |--------------------+-------------------+----------| + */ + memory_region_init_iommu(&amdvi_dev_as->iommu, + sizeof(amdvi_dev_as->iommu), +@@ -1461,7 +1463,25 @@ static AddressSpace *amdvi_host_dma_iommu(PCIBus *bus, void *opaque, int devfn) + 64); + memory_region_add_subregion_overlap(&amdvi_dev_as->root, 0, + MEMORY_REGION(&amdvi_dev_as->iommu), +- 1); ++ 0); ++ ++ /* Build the DMA Disabled alias to shared memory */ ++ memory_region_init_alias(&amdvi_dev_as->iommu_nodma, OBJECT(s), ++ "amdvi-sys", &s->mr_sys, 0, ++ memory_region_size(&s->mr_sys)); ++ memory_region_add_subregion_overlap(&amdvi_dev_as->root, 0, ++ &amdvi_dev_as->iommu_nodma, ++ 0); ++ ++ if (!x86_iommu->pt_supported) { ++ memory_region_set_enabled(&amdvi_dev_as->iommu_nodma, false); ++ memory_region_set_enabled(MEMORY_REGION(&amdvi_dev_as->iommu), ++ true); ++ } else { ++ memory_region_set_enabled(MEMORY_REGION(&amdvi_dev_as->iommu), ++ false); ++ memory_region_set_enabled(&amdvi_dev_as->iommu_nodma, true); ++ } + } + return &iommu_as[devfn]->as; + } +@@ -1602,6 +1622,17 @@ static void amdvi_sysbus_realize(DeviceState *dev, Error **errp) + "amdvi-mmio", AMDVI_MMIO_SIZE); + memory_region_add_subregion(get_system_memory(), AMDVI_BASE_ADDR, + &s->mr_mmio); ++ ++ /* Create the share memory regions by all devices */ ++ memory_region_init(&s->mr_sys, OBJECT(s), "amdvi-sys", UINT64_MAX); ++ ++ /* set up the DMA disabled memory region */ ++ memory_region_init_alias(&s->mr_nodma, OBJECT(s), ++ "amdvi-nodma", get_system_memory(), 0, ++ memory_region_size(get_system_memory())); ++ memory_region_add_subregion_overlap(&s->mr_sys, 0, ++ &s->mr_nodma, 0); ++ + pci_setup_iommu(bus, &amdvi_iommu_ops, s); + amdvi_init(s); + } +diff --git a/hw/i386/amd_iommu.h b/hw/i386/amd_iommu.h +index e5c2ae94f2..be417e51c4 100644 +--- a/hw/i386/amd_iommu.h ++++ b/hw/i386/amd_iommu.h +@@ -354,6 +354,8 @@ struct AMDVIState { + uint32_t pprlog_tail; /* ppr log tail */ + + MemoryRegion mr_mmio; /* MMIO region */ ++ MemoryRegion mr_sys; ++ MemoryRegion mr_nodma; + uint8_t mmior[AMDVI_MMIO_SIZE]; /* read/write MMIO */ + uint8_t w1cmask[AMDVI_MMIO_SIZE]; /* read/write 1 clear mask */ + uint8_t romask[AMDVI_MMIO_SIZE]; /* MMIO read/only mask */ +-- +2.39.3 + diff --git a/SOURCES/kvm-amd_iommu-Check-APIC-ID-255-for-XTSup.patch b/SOURCES/kvm-amd_iommu-Check-APIC-ID-255-for-XTSup.patch new file mode 100644 index 0000000..a230203 --- /dev/null +++ b/SOURCES/kvm-amd_iommu-Check-APIC-ID-255-for-XTSup.patch @@ -0,0 +1,66 @@ +From 0397ebacdba6539147d9986255c3f81cbfdabf1e Mon Sep 17 00:00:00 2001 +From: John Allen +Date: Wed, 11 Dec 2024 15:07:03 -0600 +Subject: [PATCH 07/57] amd_iommu: Check APIC ID > 255 for XTSup + +RH-Author: John Allen +RH-MergeRequest: 303: Interrupt Remap support for emulated amd viommu +RH-Jira: RHEL-66202 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [5/5] f39b3e3cdefc2b562f1ad2ef939a37bf404f355a (johnalle/qemu-kvm-fork) + +JIRA: https://issues.redhat.com/browse/RHEL-66202 + +commit b12cb3819baf6d9ee8140d4dd6d36fa829e2c6d9 +Author: Suravee Suthikulpanit +Date: Fri Sep 27 12:29:13 2024 -0500 + + amd_iommu: Check APIC ID > 255 for XTSup + + The XTSup mode enables x2APIC support for AMD IOMMU, which is needed + to support vcpu w/ APIC ID > 255. + + Reviewed-by: Alejandro Jimenez + Signed-off-by: Suravee Suthikulpanit + Signed-off-by: Santosh Shukla + Message-Id: <20240927172913.121477-6-santosh.shukla@amd.com> + Reviewed-by: Michael S. Tsirkin + Signed-off-by: Michael S. Tsirkin + +Signed-off-by: John Allen +--- + hw/i386/amd_iommu.c | 11 +++++++++++ + 1 file changed, 11 insertions(+) + +diff --git a/hw/i386/amd_iommu.c b/hw/i386/amd_iommu.c +index 82d76dfca9..d804656ea8 100644 +--- a/hw/i386/amd_iommu.c ++++ b/hw/i386/amd_iommu.c +@@ -32,6 +32,7 @@ + #include "trace.h" + #include "hw/i386/apic-msidef.h" + #include "hw/qdev-properties.h" ++#include "kvm/kvm_i386.h" + + /* used AMD-Vi MMIO registers */ + const char *amdvi_mmio_low[] = { +@@ -1651,6 +1652,16 @@ static void amdvi_sysbus_realize(DeviceState *dev, Error **errp) + memory_region_add_subregion_overlap(&s->mr_sys, AMDVI_INT_ADDR_FIRST, + &s->mr_ir, 1); + ++ /* AMD IOMMU with x2APIC mode requires xtsup=on */ ++ if (x86ms->apic_id_limit > 255 && !s->xtsup) { ++ error_report("AMD IOMMU with x2APIC confguration requires xtsup=on"); ++ exit(EXIT_FAILURE); ++ } ++ if (s->xtsup && kvm_irqchip_is_split() && !kvm_enable_x2apic()) { ++ error_report("AMD IOMMU xtsup=on requires support on the KVM side"); ++ exit(EXIT_FAILURE); ++ } ++ + pci_setup_iommu(bus, &amdvi_iommu_ops, s); + amdvi_init(s); + } +-- +2.39.3 + diff --git a/SOURCES/kvm-amd_iommu-Rename-variable-mmio-to-mr_mmio.patch b/SOURCES/kvm-amd_iommu-Rename-variable-mmio-to-mr_mmio.patch new file mode 100644 index 0000000..76c9fd6 --- /dev/null +++ b/SOURCES/kvm-amd_iommu-Rename-variable-mmio-to-mr_mmio.patch @@ -0,0 +1,94 @@ +From f733325d3d91576ae9f6e341faabc301542fc6c8 Mon Sep 17 00:00:00 2001 +From: John Allen +Date: Wed, 11 Dec 2024 15:06:44 -0600 +Subject: [PATCH 03/57] amd_iommu: Rename variable mmio to mr_mmio + +RH-Author: John Allen +RH-MergeRequest: 303: Interrupt Remap support for emulated amd viommu +RH-Jira: RHEL-66202 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/5] 1996a48efb7210d4d1e0b929be2d115d672e1a02 (johnalle/qemu-kvm-fork) + +JIRA: https://issues.redhat.com/browse/RHEL-66202 + +commit 2e6f051cfc58e69dcb392cd245d8f01b0c2e963f +Author: Suravee Suthikulpanit +Date: Fri Sep 27 12:29:09 2024 -0500 + + amd_iommu: Rename variable mmio to mr_mmio + + Rename the MMIO memory region variable 'mmio' to 'mr_mmio' + so to correctly name align with struct AMDVIState::variable type. + + No functional change intended. + + Reviewed-by: Alejandro Jimenez + Signed-off-by: Suravee Suthikulpanit + Signed-off-by: Santosh Shukla + Message-Id: <20240927172913.121477-2-santosh.shukla@amd.com> + Reviewed-by: Michael S. Tsirkin + Signed-off-by: Michael S. Tsirkin + +Signed-off-by: John Allen +--- + hw/i386/acpi-build.c | 4 ++-- + hw/i386/amd_iommu.c | 6 +++--- + hw/i386/amd_iommu.h | 2 +- + 3 files changed, 6 insertions(+), 6 deletions(-) + +diff --git a/hw/i386/acpi-build.c b/hw/i386/acpi-build.c +index 5d4bd2b710..032fb1f904 100644 +--- a/hw/i386/acpi-build.c ++++ b/hw/i386/acpi-build.c +@@ -2397,7 +2397,7 @@ build_amd_iommu(GArray *table_data, BIOSLinker *linker, const char *oem_id, + /* Capability offset */ + build_append_int_noprefix(table_data, s->pci.capab_offset, 2); + /* IOMMU base address */ +- build_append_int_noprefix(table_data, s->mmio.addr, 8); ++ build_append_int_noprefix(table_data, s->mr_mmio.addr, 8); + /* PCI Segment Group */ + build_append_int_noprefix(table_data, 0, 2); + /* IOMMU info */ +@@ -2432,7 +2432,7 @@ build_amd_iommu(GArray *table_data, BIOSLinker *linker, const char *oem_id, + /* Capability offset */ + build_append_int_noprefix(table_data, s->pci.capab_offset, 2); + /* IOMMU base address */ +- build_append_int_noprefix(table_data, s->mmio.addr, 8); ++ build_append_int_noprefix(table_data, s->mr_mmio.addr, 8); + /* PCI Segment Group */ + build_append_int_noprefix(table_data, 0, 2); + /* IOMMU info */ +diff --git a/hw/i386/amd_iommu.c b/hw/i386/amd_iommu.c +index 87643d2891..148b5ee51d 100644 +--- a/hw/i386/amd_iommu.c ++++ b/hw/i386/amd_iommu.c +@@ -1598,10 +1598,10 @@ static void amdvi_sysbus_realize(DeviceState *dev, Error **errp) + x86ms->ioapic_as = amdvi_host_dma_iommu(bus, s, AMDVI_IOAPIC_SB_DEVID); + + /* set up MMIO */ +- memory_region_init_io(&s->mmio, OBJECT(s), &mmio_mem_ops, s, "amdvi-mmio", +- AMDVI_MMIO_SIZE); ++ memory_region_init_io(&s->mr_mmio, OBJECT(s), &mmio_mem_ops, s, ++ "amdvi-mmio", AMDVI_MMIO_SIZE); + memory_region_add_subregion(get_system_memory(), AMDVI_BASE_ADDR, +- &s->mmio); ++ &s->mr_mmio); + pci_setup_iommu(bus, &amdvi_iommu_ops, s); + amdvi_init(s); + } +diff --git a/hw/i386/amd_iommu.h b/hw/i386/amd_iommu.h +index 73619fe9ea..e5c2ae94f2 100644 +--- a/hw/i386/amd_iommu.h ++++ b/hw/i386/amd_iommu.h +@@ -353,7 +353,7 @@ struct AMDVIState { + uint32_t pprlog_head; /* ppr log head */ + uint32_t pprlog_tail; /* ppr log tail */ + +- MemoryRegion mmio; /* MMIO region */ ++ MemoryRegion mr_mmio; /* MMIO region */ + uint8_t mmior[AMDVI_MMIO_SIZE]; /* read/write MMIO */ + uint8_t w1cmask[AMDVI_MMIO_SIZE]; /* read/write 1 clear mask */ + uint8_t romask[AMDVI_MMIO_SIZE]; /* MMIO read/only mask */ +-- +2.39.3 + diff --git a/SOURCES/kvm-amd_iommu-Send-notification-when-invalidate-interrup.patch b/SOURCES/kvm-amd_iommu-Send-notification-when-invalidate-interrup.patch new file mode 100644 index 0000000..044f8d1 --- /dev/null +++ b/SOURCES/kvm-amd_iommu-Send-notification-when-invalidate-interrup.patch @@ -0,0 +1,81 @@ +From 17ce6ac0d8edb04ba79bb39d3f695cd0506a9dc2 Mon Sep 17 00:00:00 2001 +From: John Allen +Date: Wed, 11 Dec 2024 15:06:59 -0600 +Subject: [PATCH 06/57] amd_iommu: Send notification when invalidate interrupt + entry cache + +RH-Author: John Allen +RH-MergeRequest: 303: Interrupt Remap support for emulated amd viommu +RH-Jira: RHEL-66202 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [4/5] d57e8fb4e69f3c01d32673bf658aae5067d6b969 (johnalle/qemu-kvm-fork) + +JIRA: https://issues.redhat.com/browse/RHEL-66202 + +commit f84aad4d718b83d2a4d90485992e5421430032e1 +Author: Suravee Suthikulpanit +Date: Fri Sep 27 12:29:12 2024 -0500 + + amd_iommu: Send notification when invalidate interrupt entry cache + + In order to support AMD IOMMU interrupt remapping emulation with PCI + pass-through devices, QEMU needs to notify VFIO when guest IOMMU driver + updates and invalidate the guest interrupt remapping table (IRT), and + communicate information so that the host IOMMU driver can update + the shadowed interrupt remapping table in the host IOMMU. + + Therefore, send notification when guest IOMMU emulates the IRT + invalidation commands. + + Reviewed-by: Alejandro Jimenez + Signed-off-by: Suravee Suthikulpanit + Signed-off-by: Santosh Shukla + Message-Id: <20240927172913.121477-5-santosh.shukla@amd.com> + Reviewed-by: Michael S. Tsirkin + Signed-off-by: Michael S. Tsirkin + +Signed-off-by: John Allen +--- + hw/i386/amd_iommu.c | 12 ++++++++++++ + 1 file changed, 12 insertions(+) + +diff --git a/hw/i386/amd_iommu.c b/hw/i386/amd_iommu.c +index 8fcf5eacb4..82d76dfca9 100644 +--- a/hw/i386/amd_iommu.c ++++ b/hw/i386/amd_iommu.c +@@ -431,6 +431,12 @@ static void amdvi_complete_ppr(AMDVIState *s, uint64_t *cmd) + trace_amdvi_ppr_exec(); + } + ++static void amdvi_intremap_inval_notify_all(AMDVIState *s, bool global, ++ uint32_t index, uint32_t mask) ++{ ++ x86_iommu_iec_notify_all(X86_IOMMU_DEVICE(s), global, index, mask); ++} ++ + static void amdvi_inval_all(AMDVIState *s, uint64_t *cmd) + { + if (extract64(cmd[0], 0, 60) || cmd[1]) { +@@ -438,6 +444,9 @@ static void amdvi_inval_all(AMDVIState *s, uint64_t *cmd) + s->cmdbuf + s->cmdbuf_head); + } + ++ /* Notify global invalidation */ ++ amdvi_intremap_inval_notify_all(s, true, 0, 0); ++ + amdvi_iotlb_reset(s); + trace_amdvi_all_inval(); + } +@@ -486,6 +495,9 @@ static void amdvi_inval_inttable(AMDVIState *s, uint64_t *cmd) + return; + } + ++ /* Notify global invalidation */ ++ amdvi_intremap_inval_notify_all(s, true, 0, 0); ++ + trace_amdvi_intr_inval(); + } + +-- +2.39.3 + diff --git a/SOURCES/kvm-amd_iommu-Use-shared-memory-region-for-Interrupt-Rem.patch b/SOURCES/kvm-amd_iommu-Use-shared-memory-region-for-Interrupt-Rem.patch new file mode 100644 index 0000000..39ad4ef --- /dev/null +++ b/SOURCES/kvm-amd_iommu-Use-shared-memory-region-for-Interrupt-Rem.patch @@ -0,0 +1,105 @@ +From 4859d41adfaae8933e074dcefdc81edd3832c914 Mon Sep 17 00:00:00 2001 +From: John Allen +Date: Wed, 11 Dec 2024 15:06:55 -0600 +Subject: [PATCH 05/57] amd_iommu: Use shared memory region for Interrupt + Remapping + +RH-Author: John Allen +RH-MergeRequest: 303: Interrupt Remap support for emulated amd viommu +RH-Jira: RHEL-66202 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [3/5] 48c0513c80257bfbd12c2cf3bab2503bd95d0b1c (johnalle/qemu-kvm-fork) + +JIRA: https://issues.redhat.com/browse/RHEL-66202 + +commit 9fc9dbac61ddde7d8df37e84c8e02cec249d3222 +Author: Suravee Suthikulpanit +Date: Fri Sep 27 12:29:11 2024 -0500 + + amd_iommu: Use shared memory region for Interrupt Remapping + + Use shared memory region for interrupt remapping which can be + aliased by all devices. + + Reviewed-by: Alejandro Jimenez + Signed-off-by: Suravee Suthikulpanit + Signed-off-by: Santosh Shukla + Message-Id: <20240927172913.121477-4-santosh.shukla@amd.com> + Reviewed-by: Michael S. Tsirkin + Signed-off-by: Michael S. Tsirkin + +Signed-off-by: John Allen +--- + hw/i386/amd_iommu.c | 22 ++++++++++++++-------- + hw/i386/amd_iommu.h | 1 + + 2 files changed, 15 insertions(+), 8 deletions(-) + +diff --git a/hw/i386/amd_iommu.c b/hw/i386/amd_iommu.c +index 567cb8adc9..8fcf5eacb4 100644 +--- a/hw/i386/amd_iommu.c ++++ b/hw/i386/amd_iommu.c +@@ -1443,7 +1443,7 @@ static AddressSpace *amdvi_host_dma_iommu(PCIBus *bus, void *opaque, int devfn) + * |--------------------+-------------------+----------+ + * | amdvi-root | 00000000-ffffffff | 0 | + * | amdvi-iommu_nodma | 00000000-ffffffff | 0 | +- * | amdvi-iommu_ir | fee00000-feefffff | 64 | ++ * | amdvi-iommu_ir | fee00000-feefffff | 1 | + * |--------------------+-------------------+----------| + */ + memory_region_init_iommu(&amdvi_dev_as->iommu, +@@ -1454,13 +1454,6 @@ static AddressSpace *amdvi_host_dma_iommu(PCIBus *bus, void *opaque, int devfn) + memory_region_init(&amdvi_dev_as->root, OBJECT(s), + "amdvi_root", UINT64_MAX); + address_space_init(&amdvi_dev_as->as, &amdvi_dev_as->root, name); +- memory_region_init_io(&amdvi_dev_as->iommu_ir, OBJECT(s), +- &amdvi_ir_ops, s, "amd_iommu_ir", +- AMDVI_INT_ADDR_SIZE); +- memory_region_add_subregion_overlap(&amdvi_dev_as->root, +- AMDVI_INT_ADDR_FIRST, +- &amdvi_dev_as->iommu_ir, +- 64); + memory_region_add_subregion_overlap(&amdvi_dev_as->root, 0, + MEMORY_REGION(&amdvi_dev_as->iommu), + 0); +@@ -1472,6 +1465,13 @@ static AddressSpace *amdvi_host_dma_iommu(PCIBus *bus, void *opaque, int devfn) + memory_region_add_subregion_overlap(&amdvi_dev_as->root, 0, + &amdvi_dev_as->iommu_nodma, + 0); ++ /* Build the Interrupt Remapping alias to shared memory */ ++ memory_region_init_alias(&amdvi_dev_as->iommu_ir, OBJECT(s), ++ "amdvi-ir", &s->mr_ir, 0, ++ memory_region_size(&s->mr_ir)); ++ memory_region_add_subregion_overlap(MEMORY_REGION(&amdvi_dev_as->iommu), ++ AMDVI_INT_ADDR_FIRST, ++ &amdvi_dev_as->iommu_ir, 1); + + if (!x86_iommu->pt_supported) { + memory_region_set_enabled(&amdvi_dev_as->iommu_nodma, false); +@@ -1633,6 +1633,12 @@ static void amdvi_sysbus_realize(DeviceState *dev, Error **errp) + memory_region_add_subregion_overlap(&s->mr_sys, 0, + &s->mr_nodma, 0); + ++ /* set up the Interrupt Remapping memory region */ ++ memory_region_init_io(&s->mr_ir, OBJECT(s), &amdvi_ir_ops, ++ s, "amdvi-ir", AMDVI_INT_ADDR_SIZE); ++ memory_region_add_subregion_overlap(&s->mr_sys, AMDVI_INT_ADDR_FIRST, ++ &s->mr_ir, 1); ++ + pci_setup_iommu(bus, &amdvi_iommu_ops, s); + amdvi_init(s); + } +diff --git a/hw/i386/amd_iommu.h b/hw/i386/amd_iommu.h +index be417e51c4..e0dac4d9a9 100644 +--- a/hw/i386/amd_iommu.h ++++ b/hw/i386/amd_iommu.h +@@ -356,6 +356,7 @@ struct AMDVIState { + MemoryRegion mr_mmio; /* MMIO region */ + MemoryRegion mr_sys; + MemoryRegion mr_nodma; ++ MemoryRegion mr_ir; + uint8_t mmior[AMDVI_MMIO_SIZE]; /* read/write MMIO */ + uint8_t w1cmask[AMDVI_MMIO_SIZE]; /* read/write 1 clear mask */ + uint8_t romask[AMDVI_MMIO_SIZE]; /* MMIO read/only mask */ +-- +2.39.3 + diff --git a/SOURCES/kvm-arm-Use-arm_virt_compat_set-to-apply-the-compat.patch b/SOURCES/kvm-arm-Use-arm_virt_compat_set-to-apply-the-compat.patch new file mode 100644 index 0000000..e293f6a --- /dev/null +++ b/SOURCES/kvm-arm-Use-arm_virt_compat_set-to-apply-the-compat.patch @@ -0,0 +1,53 @@ +From 173beb6698538dcffefab36772e107ffb0b4fbbd Mon Sep 17 00:00:00 2001 +From: Shaoqin Huang +Date: Mon, 28 Apr 2025 04:34:27 -0400 +Subject: [PATCH 2/5] arm: Use arm_virt_compat_set() to apply the compat + +RH-Author: Shaoqin Huang +RH-MergeRequest: 353: virtio-net: disable USO for virt-rhel9.6 +RH-Jira: RHEL-80313 +RH-Acked-by: Thomas Huth +RH-Acked-by: Eric Auger +RH-Commit: [2/2] 6e7a158e65296928040e70622b3cee59e45c1c36 (shahuang/qemu-kvm) + +JIRA: https://issues.redhat.com/browse/RHEL-80313 +Upstream Status: RHEL only + +Since the pauth and uso both should apply for the latest machine type, +move them to the arm_virt_compat_set() which applies the compat to all +machine types automatically. + +Signed-off-by: Shaoqin Huang +--- + hw/arm/virt.c | 8 ++++---- + 1 file changed, 4 insertions(+), 4 deletions(-) + +diff --git a/hw/arm/virt.c b/hw/arm/virt.c +index 896deaa025..2aef94e776 100644 +--- a/hw/arm/virt.c ++++ b/hw/arm/virt.c +@@ -127,6 +127,10 @@ static void arm_virt_compat_set(MachineClass *mc) + arm_virt_compat_len); + compat_props_add(mc->compat_props, arm_rhel_compat, + arm_rhel_compat_len); ++ compat_props_add(mc->compat_props, arm_rhel9_compat, ++ arm_rhel9_compat_len); ++ compat_props_add(mc->compat_props, hw_compat_rhel_9, ++ hw_compat_rhel_9_len); + } + + #define DEFINE_VIRT_MACHINE_IMPL(latest, ...) \ +@@ -3599,10 +3603,6 @@ DEFINE_VIRT_MACHINE(2, 6) + + static void virt_rhel_machine_9_6_0_options(MachineClass *mc) + { +- compat_props_add(mc->compat_props, arm_rhel9_compat, arm_rhel9_compat_len); +- +- /* NB: remember to move this line to the *latest* RHEL 9 machine */ +- compat_props_add(mc->compat_props, hw_compat_rhel_9, hw_compat_rhel_9_len); + } + DEFINE_VIRT_MACHINE_AS_LATEST(9, 6, 0) + +-- +2.48.1 + diff --git a/SOURCES/kvm-arm-kvm-report-registers-we-failed-to-set.patch b/SOURCES/kvm-arm-kvm-report-registers-we-failed-to-set.patch new file mode 100644 index 0000000..d17acca --- /dev/null +++ b/SOURCES/kvm-arm-kvm-report-registers-we-failed-to-set.patch @@ -0,0 +1,154 @@ +From c58596d3eea1b98e5fccb6f00f207ce5bf3f90c8 Mon Sep 17 00:00:00 2001 +From: Cornelia Huck +Date: Thu, 11 Sep 2025 17:41:59 +0200 +Subject: [PATCH] arm/kvm: report registers we failed to set + +RH-Author: Eric Auger +RH-MergeRequest: 493: arm/kvm: report registers we failed to set +RH-Jira: RHEL-120502 +RH-Acked-by: Sebastian Ott +RH-Acked-by: Cornelia Huck +RH-Acked-by: Gavin Shan +RH-Acked-by: Donald Dutile +RH-Commit: [1/1] 20bfe1faca2dd944bd9afcb2e1c198f0a5723259 + +If we fail migration because of a mismatch of some registers between +source and destination, the error message is not very informative: + +qemu-system-aarch64: error while loading state for instance 0x0 ofdevice 'cpu' +qemu-system-aarch64: Failed to put registers after init: Invalid argument + +At least try to give the user a hint which registers had a problem, +even if they cannot really do anything about it right now. + +Sample output: + +Could not set register op0:3 op1:0 crn:0 crm:0 op2:0 to c00fac31 (is 413fd0c1) + +We could be even more helpful once we support writable ID registers, +at which point the user might actually be able to configure something +that is migratable. + +Suggested-by: Eric Auger +Reviewed-by: Sebastian Ott +Signed-off-by: Cornelia Huck +Message-id: 20250911154159.158046-1-cohuck@redhat.com +Signed-off-by: Peter Maydell +(cherry picked from commit 19f6dcfe6b8b2a3523362812fc696ab83050d316) +Signed-off-by: Eric Auger +--- + target/arm/kvm.c | 86 ++++++++++++++++++++++++++++++++++++++++++++++++ + 1 file changed, 86 insertions(+) + +diff --git a/target/arm/kvm.c b/target/arm/kvm.c +index e0469e7831..557cfcf670 100644 +--- a/target/arm/kvm.c ++++ b/target/arm/kvm.c +@@ -923,6 +923,58 @@ bool write_kvmstate_to_list(ARMCPU *cpu) + return ok; + } + ++/* pretty-print a KVM register */ ++#define CP_REG_ARM64_SYSREG_OP(_reg, _op) \ ++ ((uint8_t)((_reg & CP_REG_ARM64_SYSREG_ ## _op ## _MASK) >> \ ++ CP_REG_ARM64_SYSREG_ ## _op ## _SHIFT)) ++ ++static gchar *kvm_print_sve_register_name(uint64_t regidx) ++{ ++ uint16_t sve_reg = regidx & 0x000000000000ffff; ++ ++ if (regidx == KVM_REG_ARM64_SVE_VLS) { ++ return g_strdup_printf("SVE VLS"); ++ } ++ /* zreg, preg, ffr */ ++ switch (sve_reg & 0xfc00) { ++ case 0: ++ return g_strdup_printf("SVE zreg n:%d slice:%d", ++ (sve_reg & 0x03e0) >> 5, sve_reg & 0x001f); ++ case 0x04: ++ return g_strdup_printf("SVE preg n:%d slice:%d", ++ (sve_reg & 0x01e0) >> 5, sve_reg & 0x001f); ++ case 0x06: ++ return g_strdup_printf("SVE ffr slice:%d", sve_reg & 0x001f); ++ default: ++ return g_strdup_printf("SVE ???"); ++ } ++} ++ ++static gchar *kvm_print_register_name(uint64_t regidx) ++{ ++ switch ((regidx & KVM_REG_ARM_COPROC_MASK)) { ++ case KVM_REG_ARM_CORE: ++ return g_strdup_printf("core reg %"PRIx64, regidx); ++ case KVM_REG_ARM_DEMUX: ++ return g_strdup_printf("demuxed reg %"PRIx64, regidx); ++ case KVM_REG_ARM64_SYSREG: ++ return g_strdup_printf("op0:%d op1:%d crn:%d crm:%d op2:%d", ++ CP_REG_ARM64_SYSREG_OP(regidx, OP0), ++ CP_REG_ARM64_SYSREG_OP(regidx, OP1), ++ CP_REG_ARM64_SYSREG_OP(regidx, CRN), ++ CP_REG_ARM64_SYSREG_OP(regidx, CRM), ++ CP_REG_ARM64_SYSREG_OP(regidx, OP2)); ++ case KVM_REG_ARM_FW: ++ return g_strdup_printf("fw reg %d", (int)(regidx & 0xffff)); ++ case KVM_REG_ARM64_SVE: ++ return kvm_print_sve_register_name(regidx); ++ case KVM_REG_ARM_FW_FEAT_BMAP: ++ return g_strdup_printf("fw feat reg %d", (int)(regidx & 0xffff)); ++ default: ++ return g_strdup_printf("%"PRIx64, regidx); ++ } ++} ++ + bool write_list_to_kvmstate(ARMCPU *cpu, int level) + { + CPUState *cs = CPU(cpu); +@@ -950,11 +1002,45 @@ bool write_list_to_kvmstate(ARMCPU *cpu, int level) + g_assert_not_reached(); + } + if (ret) { ++ gchar *reg_str = kvm_print_register_name(regidx); ++ + /* We might fail for "unknown register" and also for + * "you tried to set a register which is constant with + * a different value from what it actually contains". + */ + ok = false; ++ switch (ret) { ++ case -ENOENT: ++ error_report("Could not set register %s: unknown to KVM", ++ reg_str); ++ break; ++ case -EINVAL: ++ if ((regidx & KVM_REG_SIZE_MASK) == KVM_REG_SIZE_U32) { ++ if (!kvm_get_one_reg(cs, regidx, &v32)) { ++ error_report("Could not set register %s to %x (is %x)", ++ reg_str, (uint32_t)cpu->cpreg_values[i], ++ v32); ++ } else { ++ error_report("Could not set register %s to %x", ++ reg_str, (uint32_t)cpu->cpreg_values[i]); ++ } ++ } else /* U64 */ { ++ uint64_t v64; ++ ++ if (!kvm_get_one_reg(cs, regidx, &v64)) { ++ error_report("Could not set register %s to %"PRIx64" (is %"PRIx64")", ++ reg_str, cpu->cpreg_values[i], v64); ++ } else { ++ error_report("Could not set register %s to %"PRIx64, ++ reg_str, cpu->cpreg_values[i]); ++ } ++ } ++ break; ++ default: ++ error_report("Could not set register %s: %s", ++ reg_str, strerror(-ret)); ++ } ++ g_free(reg_str); + } + } + return ok; +-- +2.50.1 + diff --git a/SOURCES/kvm-block-Add-new-bdrv_co_is_all_zeroes-function.patch b/SOURCES/kvm-block-Add-new-bdrv_co_is_all_zeroes-function.patch new file mode 100644 index 0000000..174ab4f --- /dev/null +++ b/SOURCES/kvm-block-Add-new-bdrv_co_is_all_zeroes-function.patch @@ -0,0 +1,145 @@ +From f2cd96a040dd7863484d22a3995a2904605dadde Mon Sep 17 00:00:00 2001 +From: Eric Blake +Date: Fri, 9 May 2025 15:40:21 -0500 +Subject: [PATCH 06/16] block: Add new bdrv_co_is_all_zeroes() function + +RH-Author: Eric Blake +RH-MergeRequest: 365: blockdev-mirror: More efficient handling of sparse mirrors +RH-Jira: RHEL-82906 RHEL-83015 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Jon Maloy +RH-Commit: [4/14] aabcba8323df698a72842f299e9242a5eee3aea6 (ebblake/centos-qemu-kvm) + +There are some optimizations that require knowing if an image starts +out as reading all zeroes, such as making blockdev-mirror faster by +skipping the copying of source zeroes to the destination. The +existing bdrv_co_is_zero_fast() is a good building block for answering +this question, but it tends to give an answer of 0 for a file we just +created via QMP 'blockdev-create' or similar (such as 'qemu-img create +-f raw'). Why? Because file-posix.c insists on allocating a tiny +header to any file rather than leaving it 100% sparse, due to some +filesystems that are unable to answer alignment probes on a hole. But +teaching file-posix.c to read the tiny header doesn't scale - the +problem of a small header is also visible when libvirt sets up an NBD +client to a just-created file on a migration destination host. + +So, we need a wrapper function that handles a bit more complexity in a +common manner for all block devices - when the BDS is mostly a hole, +but has a small non-hole header, it is still worth the time to read +that header and check if it reads as all zeroes before giving up and +returning a pessimistic answer. + +Signed-off-by: Eric Blake +Reviewed-by: Stefan Hajnoczi +Message-ID: <20250509204341.3553601-19-eblake@redhat.com> +(cherry picked from commit 52726096707c5c8b90597c445de897fa64d56e73) +Conflicts: + block/io.c - context with header names +Jira: https://issues.redhat.com/browse/RHEL-82906 +Jira: https://issues.redhat.com/browse/RHEL-83015 +Signed-off-by: Eric Blake +--- + block/io.c | 62 ++++++++++++++++++++++++++++++++++++++++ + include/block/block-io.h | 2 ++ + 2 files changed, 64 insertions(+) + +diff --git a/block/io.c b/block/io.c +index 293c5dd393..1f01337599 100644 +--- a/block/io.c ++++ b/block/io.c +@@ -38,10 +38,14 @@ + #include "qemu/error-report.h" + #include "qemu/main-loop.h" + #include "sysemu/replay.h" ++#include "qemu/units.h" + + /* Maximum bounce buffer for copy-on-read and write zeroes, in bytes */ + #define MAX_BOUNCE_BUFFER (32768 << BDRV_SECTOR_BITS) + ++/* Maximum read size for checking if data reads as zero, in bytes */ ++#define MAX_ZERO_CHECK_BUFFER (128 * KiB) ++ + static void coroutine_fn GRAPH_RDLOCK + bdrv_parent_cb_resize(BlockDriverState *bs); + +@@ -2774,6 +2778,64 @@ int coroutine_fn bdrv_co_is_zero_fast(BlockDriverState *bs, int64_t offset, + return 1; + } + ++/* ++ * Check @bs (and its backing chain) to see if the entire image is known ++ * to read as zeroes. ++ * Return 1 if that is the case, 0 otherwise and -errno on error. ++ * This test is meant to be fast rather than accurate so returning 0 ++ * does not guarantee non-zero data; however, a return of 1 is reliable, ++ * and this function can report 1 in more cases than bdrv_co_is_zero_fast. ++ */ ++int coroutine_fn bdrv_co_is_all_zeroes(BlockDriverState *bs) ++{ ++ int ret; ++ int64_t pnum, bytes; ++ char *buf; ++ QEMUIOVector local_qiov; ++ IO_CODE(); ++ ++ bytes = bdrv_co_getlength(bs); ++ if (bytes < 0) { ++ return bytes; ++ } ++ ++ /* First probe - see if the entire image reads as zero */ ++ ret = bdrv_co_common_block_status_above(bs, NULL, false, BDRV_WANT_ZERO, ++ 0, bytes, &pnum, NULL, NULL, ++ NULL); ++ if (ret < 0) { ++ return ret; ++ } ++ if (ret & BDRV_BLOCK_ZERO) { ++ return bdrv_co_is_zero_fast(bs, pnum, bytes - pnum); ++ } ++ ++ /* ++ * Because of the way 'blockdev-create' works, raw files tend to ++ * be created with a non-sparse region at the front to make ++ * alignment probing easier. If the block starts with only a ++ * small allocated region, it is still worth the effort to see if ++ * the rest of the image is still sparse, coupled with manually ++ * reading the first region to see if it reads zero after all. ++ */ ++ if (pnum > MAX_ZERO_CHECK_BUFFER) { ++ return 0; ++ } ++ ret = bdrv_co_is_zero_fast(bs, pnum, bytes - pnum); ++ if (ret <= 0) { ++ return ret; ++ } ++ /* Only the head of the image is unknown, and it's small. Read it. */ ++ buf = qemu_blockalign(bs, pnum); ++ qemu_iovec_init_buf(&local_qiov, buf, pnum); ++ ret = bdrv_driver_preadv(bs, 0, pnum, &local_qiov, 0, 0); ++ if (ret >= 0) { ++ ret = buffer_is_zero(buf, pnum); ++ } ++ qemu_vfree(buf); ++ return ret; ++} ++ + int coroutine_fn bdrv_co_is_allocated(BlockDriverState *bs, int64_t offset, + int64_t bytes, int64_t *pnum) + { +diff --git a/include/block/block-io.h b/include/block/block-io.h +index b49e0537dd..b99cc98d26 100644 +--- a/include/block/block-io.h ++++ b/include/block/block-io.h +@@ -161,6 +161,8 @@ bdrv_is_allocated_above(BlockDriverState *bs, BlockDriverState *base, + + int coroutine_fn GRAPH_RDLOCK + bdrv_co_is_zero_fast(BlockDriverState *bs, int64_t offset, int64_t bytes); ++int coroutine_fn GRAPH_RDLOCK ++bdrv_co_is_all_zeroes(BlockDriverState *bs); + + int GRAPH_RDLOCK + bdrv_apply_auto_read_only(BlockDriverState *bs, const char *errmsg, +-- +2.48.1 + diff --git a/SOURCES/kvm-block-Expand-block-status-mode-from-bool-to-flags.patch b/SOURCES/kvm-block-Expand-block-status-mode-from-bool-to-flags.patch new file mode 100644 index 0000000..ce9b9cc --- /dev/null +++ b/SOURCES/kvm-block-Expand-block-status-mode-from-bool-to-flags.patch @@ -0,0 +1,689 @@ +From 26f5d221dd16137bed3527ee120cdf085e2c7e23 Mon Sep 17 00:00:00 2001 +From: Eric Blake +Date: Fri, 9 May 2025 15:40:18 -0500 +Subject: [PATCH 03/16] block: Expand block status mode from bool to flags + +RH-Author: Eric Blake +RH-MergeRequest: 365: blockdev-mirror: More efficient handling of sparse mirrors +RH-Jira: RHEL-82906 RHEL-83015 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Jon Maloy +RH-Commit: [1/14] 9de5245def80e9815ed306e4abce9caec56cef6f (ebblake/centos-qemu-kvm) + +This patch is purely mechanical, changing bool want_zero into an +unsigned int for bitwise-or of flags. As of this patch, all +implementations are unchanged (the old want_zero==true is now +mode==BDRV_WANT_PRECISE which is a superset of BDRV_WANT_ZERO); but +the callers in io.c that used to pass want_zero==false are now +prepared for future driver changes that can now distinguish bewteen +BDRV_WANT_ZERO vs. BDRV_WANT_ALLOCATED. The next patch will actually +change the file-posix driver along those lines, now that we have +more-specific hints. + +As for the background why this patch is useful: right now, the +file-posix driver recognizes that if allocation is being queried, the +entire image can be reported as allocated (there is no backing file to +refer to) - but this throws away information on whether the entire +image reads as zero (trivially true if lseek(SEEK_HOLE) at offset 0 +returns -ENXIO, a bit more complicated to prove if the raw file was +created with 'qemu-img create' since we intentionally allocate a small +chunk of all-zero data to help with alignment probing). Later patches +will add a generic algorithm for seeing if an entire file reads as +zeroes. + +Signed-off-by: Eric Blake +Reviewed-by: Stefan Hajnoczi +Message-ID: <20250509204341.3553601-16-eblake@redhat.com> +(cherry picked from commit c33159dec79069514f78faecfe268439226b0f5b) +Jira: https://issues.redhat.com/browse/RHEL-82906 +Jira: https://issues.redhat.com/browse/RHEL-83015 +Signed-off-by: Eric Blake +--- + block/blkdebug.c | 6 ++-- + block/copy-before-write.c | 4 +-- + block/coroutines.h | 4 +-- + block/file-posix.c | 4 +-- + block/gluster.c | 4 +-- + block/io.c | 51 ++++++++++++++++---------------- + block/iscsi.c | 6 ++-- + block/nbd.c | 4 +-- + block/null.c | 6 ++-- + block/parallels.c | 6 ++-- + block/qcow.c | 2 +- + block/qcow2.c | 6 ++-- + block/qed.c | 6 ++-- + block/quorum.c | 4 +-- + block/raw-format.c | 4 +-- + block/rbd.c | 6 ++-- + block/snapshot-access.c | 4 +-- + block/vdi.c | 4 +-- + block/vmdk.c | 2 +- + block/vpc.c | 2 +- + block/vvfat.c | 6 ++-- + include/block/block-common.h | 11 +++++++ + include/block/block_int-common.h | 27 +++++++++-------- + include/block/block_int-io.h | 4 +-- + tests/unit/test-block-iothread.c | 2 +- + 25 files changed, 99 insertions(+), 86 deletions(-) + +diff --git a/block/blkdebug.c b/block/blkdebug.c +index c95c818c38..736ae2b56b 100644 +--- a/block/blkdebug.c ++++ b/block/blkdebug.c +@@ -751,9 +751,9 @@ blkdebug_co_pdiscard(BlockDriverState *bs, int64_t offset, int64_t bytes) + } + + static int coroutine_fn GRAPH_RDLOCK +-blkdebug_co_block_status(BlockDriverState *bs, bool want_zero, int64_t offset, +- int64_t bytes, int64_t *pnum, int64_t *map, +- BlockDriverState **file) ++blkdebug_co_block_status(BlockDriverState *bs, unsigned int mode, ++ int64_t offset, int64_t bytes, int64_t *pnum, ++ int64_t *map, BlockDriverState **file) + { + int err; + +diff --git a/block/copy-before-write.c b/block/copy-before-write.c +index 853e01a1eb..36488cdeca 100644 +--- a/block/copy-before-write.c ++++ b/block/copy-before-write.c +@@ -290,8 +290,8 @@ cbw_co_preadv_snapshot(BlockDriverState *bs, int64_t offset, int64_t bytes, + } + + static int coroutine_fn GRAPH_RDLOCK +-cbw_co_snapshot_block_status(BlockDriverState *bs, +- bool want_zero, int64_t offset, int64_t bytes, ++cbw_co_snapshot_block_status(BlockDriverState *bs, unsigned int mode, ++ int64_t offset, int64_t bytes, + int64_t *pnum, int64_t *map, + BlockDriverState **file) + { +diff --git a/block/coroutines.h b/block/coroutines.h +index f3226682d6..811ef12e43 100644 +--- a/block/coroutines.h ++++ b/block/coroutines.h +@@ -47,7 +47,7 @@ int coroutine_fn GRAPH_RDLOCK + bdrv_co_common_block_status_above(BlockDriverState *bs, + BlockDriverState *base, + bool include_base, +- bool want_zero, ++ unsigned int mode, + int64_t offset, + int64_t bytes, + int64_t *pnum, +@@ -78,7 +78,7 @@ int co_wrapper_mixed_bdrv_rdlock + bdrv_common_block_status_above(BlockDriverState *bs, + BlockDriverState *base, + bool include_base, +- bool want_zero, ++ unsigned int mode, + int64_t offset, + int64_t bytes, + int64_t *pnum, +diff --git a/block/file-posix.c b/block/file-posix.c +index f17a3f4d10..9ca55620ca 100644 +--- a/block/file-posix.c ++++ b/block/file-posix.c +@@ -3277,7 +3277,7 @@ static int find_allocation(BlockDriverState *bs, off_t start, + * well exceed it. + */ + static int coroutine_fn raw_co_block_status(BlockDriverState *bs, +- bool want_zero, ++ unsigned int mode, + int64_t offset, + int64_t bytes, int64_t *pnum, + int64_t *map, +@@ -3293,7 +3293,7 @@ static int coroutine_fn raw_co_block_status(BlockDriverState *bs, + return ret; + } + +- if (!want_zero) { ++ if (mode != BDRV_WANT_PRECISE) { + *pnum = bytes; + *map = offset; + *file = bs; +diff --git a/block/gluster.c b/block/gluster.c +index f8b415f381..ae5c45666b 100644 +--- a/block/gluster.c ++++ b/block/gluster.c +@@ -1466,7 +1466,7 @@ exit: + * (Based on raw_co_block_status() from file-posix.c.) + */ + static int coroutine_fn qemu_gluster_co_block_status(BlockDriverState *bs, +- bool want_zero, ++ unsigned int mode, + int64_t offset, + int64_t bytes, + int64_t *pnum, +@@ -1483,7 +1483,7 @@ static int coroutine_fn qemu_gluster_co_block_status(BlockDriverState *bs, + return ret; + } + +- if (!want_zero) { ++ if (mode != BDRV_WANT_PRECISE) { + *pnum = bytes; + *map = offset; + *file = bs; +diff --git a/block/io.c b/block/io.c +index 3e189837a1..daaafe00d7 100644 +--- a/block/io.c ++++ b/block/io.c +@@ -2360,10 +2360,8 @@ int bdrv_flush_all(void) + * Drivers not implementing the functionality are assumed to not support + * backing files, hence all their sectors are reported as allocated. + * +- * If 'want_zero' is true, the caller is querying for mapping +- * purposes, with a focus on valid BDRV_BLOCK_OFFSET_VALID, _DATA, and +- * _ZERO where possible; otherwise, the result favors larger 'pnum', +- * with a focus on accurate BDRV_BLOCK_ALLOCATED. ++ * 'mode' serves as a hint as to which results are favored; see the ++ * BDRV_WANT_* macros for details. + * + * If 'offset' is beyond the end of the disk image the return value is + * BDRV_BLOCK_EOF and 'pnum' is set to 0. +@@ -2383,7 +2381,7 @@ int bdrv_flush_all(void) + * set to the host mapping and BDS corresponding to the guest offset. + */ + static int coroutine_fn GRAPH_RDLOCK +-bdrv_co_do_block_status(BlockDriverState *bs, bool want_zero, ++bdrv_co_do_block_status(BlockDriverState *bs, unsigned int mode, + int64_t offset, int64_t bytes, + int64_t *pnum, int64_t *map, BlockDriverState **file) + { +@@ -2472,7 +2470,7 @@ bdrv_co_do_block_status(BlockDriverState *bs, bool want_zero, + local_file = bs; + local_map = aligned_offset; + } else { +- ret = bs->drv->bdrv_co_block_status(bs, want_zero, aligned_offset, ++ ret = bs->drv->bdrv_co_block_status(bs, mode, aligned_offset, + aligned_bytes, pnum, &local_map, + &local_file); + +@@ -2484,10 +2482,10 @@ bdrv_co_do_block_status(BlockDriverState *bs, bool want_zero, + * the cache requires an RCU update, so double check here to avoid + * such an update if possible. + * +- * Check want_zero, because we only want to update the cache when we ++ * Check mode, because we only want to update the cache when we + * have accurate information about what is zero and what is data. + */ +- if (want_zero && ++ if (mode == BDRV_WANT_PRECISE && + ret == (BDRV_BLOCK_DATA | BDRV_BLOCK_OFFSET_VALID) && + QLIST_EMPTY(&bs->children)) + { +@@ -2544,7 +2542,7 @@ bdrv_co_do_block_status(BlockDriverState *bs, bool want_zero, + + if (ret & BDRV_BLOCK_RAW) { + assert(ret & BDRV_BLOCK_OFFSET_VALID && local_file); +- ret = bdrv_co_do_block_status(local_file, want_zero, local_map, ++ ret = bdrv_co_do_block_status(local_file, mode, local_map, + *pnum, pnum, &local_map, &local_file); + goto out; + } +@@ -2556,7 +2554,7 @@ bdrv_co_do_block_status(BlockDriverState *bs, bool want_zero, + + if (!cow_bs) { + ret |= BDRV_BLOCK_ZERO; +- } else if (want_zero) { ++ } else if (mode == BDRV_WANT_PRECISE) { + int64_t size2 = bdrv_co_getlength(cow_bs); + + if (size2 >= 0 && offset >= size2) { +@@ -2565,14 +2563,14 @@ bdrv_co_do_block_status(BlockDriverState *bs, bool want_zero, + } + } + +- if (want_zero && ret & BDRV_BLOCK_RECURSE && ++ if (mode == BDRV_WANT_PRECISE && ret & BDRV_BLOCK_RECURSE && + local_file && local_file != bs && + (ret & BDRV_BLOCK_DATA) && !(ret & BDRV_BLOCK_ZERO) && + (ret & BDRV_BLOCK_OFFSET_VALID)) { + int64_t file_pnum; + int ret2; + +- ret2 = bdrv_co_do_block_status(local_file, want_zero, local_map, ++ ret2 = bdrv_co_do_block_status(local_file, mode, local_map, + *pnum, &file_pnum, NULL, NULL); + if (ret2 >= 0) { + /* Ignore errors. This is just providing extra information, it +@@ -2623,7 +2621,7 @@ int coroutine_fn + bdrv_co_common_block_status_above(BlockDriverState *bs, + BlockDriverState *base, + bool include_base, +- bool want_zero, ++ unsigned int mode, + int64_t offset, + int64_t bytes, + int64_t *pnum, +@@ -2650,7 +2648,7 @@ bdrv_co_common_block_status_above(BlockDriverState *bs, + return 0; + } + +- ret = bdrv_co_do_block_status(bs, want_zero, offset, bytes, pnum, ++ ret = bdrv_co_do_block_status(bs, mode, offset, bytes, pnum, + map, file); + ++*depth; + if (ret < 0 || *pnum == 0 || ret & BDRV_BLOCK_ALLOCATED || bs == base) { +@@ -2667,7 +2665,7 @@ bdrv_co_common_block_status_above(BlockDriverState *bs, + for (p = bdrv_filter_or_cow_bs(bs); include_base || p != base; + p = bdrv_filter_or_cow_bs(p)) + { +- ret = bdrv_co_do_block_status(p, want_zero, offset, bytes, pnum, ++ ret = bdrv_co_do_block_status(p, mode, offset, bytes, pnum, + map, file); + ++*depth; + if (ret < 0) { +@@ -2730,7 +2728,8 @@ int coroutine_fn bdrv_co_block_status_above(BlockDriverState *bs, + BlockDriverState **file) + { + IO_CODE(); +- return bdrv_co_common_block_status_above(bs, base, false, true, offset, ++ return bdrv_co_common_block_status_above(bs, base, false, ++ BDRV_WANT_PRECISE, offset, + bytes, pnum, map, file, NULL); + } + +@@ -2761,8 +2760,9 @@ int coroutine_fn bdrv_co_is_zero_fast(BlockDriverState *bs, int64_t offset, + return 1; + } + +- ret = bdrv_co_common_block_status_above(bs, NULL, false, false, offset, +- bytes, &pnum, NULL, NULL, NULL); ++ ret = bdrv_co_common_block_status_above(bs, NULL, false, BDRV_WANT_ZERO, ++ offset, bytes, &pnum, NULL, NULL, ++ NULL); + + if (ret < 0) { + return ret; +@@ -2778,9 +2778,9 @@ int coroutine_fn bdrv_co_is_allocated(BlockDriverState *bs, int64_t offset, + int64_t dummy; + IO_CODE(); + +- ret = bdrv_co_common_block_status_above(bs, bs, true, false, offset, +- bytes, pnum ? pnum : &dummy, NULL, +- NULL, NULL); ++ ret = bdrv_co_common_block_status_above(bs, bs, true, BDRV_WANT_ALLOCATED, ++ offset, bytes, pnum ? pnum : &dummy, ++ NULL, NULL, NULL); + if (ret < 0) { + return ret; + } +@@ -2813,7 +2813,8 @@ int coroutine_fn bdrv_co_is_allocated_above(BlockDriverState *bs, + int ret; + IO_CODE(); + +- ret = bdrv_co_common_block_status_above(bs, base, include_base, false, ++ ret = bdrv_co_common_block_status_above(bs, base, include_base, ++ BDRV_WANT_ALLOCATED, + offset, bytes, pnum, NULL, NULL, + &depth); + if (ret < 0) { +@@ -3710,8 +3711,8 @@ bdrv_co_preadv_snapshot(BdrvChild *child, int64_t offset, int64_t bytes, + } + + int coroutine_fn +-bdrv_co_snapshot_block_status(BlockDriverState *bs, +- bool want_zero, int64_t offset, int64_t bytes, ++bdrv_co_snapshot_block_status(BlockDriverState *bs, unsigned int mode, ++ int64_t offset, int64_t bytes, + int64_t *pnum, int64_t *map, + BlockDriverState **file) + { +@@ -3729,7 +3730,7 @@ bdrv_co_snapshot_block_status(BlockDriverState *bs, + } + + bdrv_inc_in_flight(bs); +- ret = drv->bdrv_co_snapshot_block_status(bs, want_zero, offset, bytes, ++ ret = drv->bdrv_co_snapshot_block_status(bs, mode, offset, bytes, + pnum, map, file); + bdrv_dec_in_flight(bs); + +diff --git a/block/iscsi.c b/block/iscsi.c +index 979bf90cb7..d7caa4b363 100644 +--- a/block/iscsi.c ++++ b/block/iscsi.c +@@ -694,9 +694,9 @@ out_unlock: + + + static int coroutine_fn iscsi_co_block_status(BlockDriverState *bs, +- bool want_zero, int64_t offset, +- int64_t bytes, int64_t *pnum, +- int64_t *map, ++ unsigned int mode, ++ int64_t offset, int64_t bytes, ++ int64_t *pnum, int64_t *map, + BlockDriverState **file) + { + IscsiLun *iscsilun = bs->opaque; +diff --git a/block/nbd.c b/block/nbd.c +index d464315766..a359aa236e 100644 +--- a/block/nbd.c ++++ b/block/nbd.c +@@ -1397,8 +1397,8 @@ nbd_client_co_pdiscard(BlockDriverState *bs, int64_t offset, int64_t bytes) + } + + static int coroutine_fn GRAPH_RDLOCK nbd_client_co_block_status( +- BlockDriverState *bs, bool want_zero, int64_t offset, int64_t bytes, +- int64_t *pnum, int64_t *map, BlockDriverState **file) ++ BlockDriverState *bs, unsigned int mode, int64_t offset, ++ int64_t bytes, int64_t *pnum, int64_t *map, BlockDriverState **file) + { + int ret, request_ret; + NBDExtent64 extent = { 0 }; +diff --git a/block/null.c b/block/null.c +index 4730acc1eb..95021230c8 100644 +--- a/block/null.c ++++ b/block/null.c +@@ -227,9 +227,9 @@ static int null_reopen_prepare(BDRVReopenState *reopen_state, + } + + static int coroutine_fn null_co_block_status(BlockDriverState *bs, +- bool want_zero, int64_t offset, +- int64_t bytes, int64_t *pnum, +- int64_t *map, ++ unsigned int mode, ++ int64_t offset, int64_t bytes, ++ int64_t *pnum, int64_t *map, + BlockDriverState **file) + { + BDRVNullState *s = bs->opaque; +diff --git a/block/parallels.c b/block/parallels.c +index 9205a0864f..22ea7834fd 100644 +--- a/block/parallels.c ++++ b/block/parallels.c +@@ -416,9 +416,9 @@ parallels_co_flush_to_os(BlockDriverState *bs) + } + + static int coroutine_fn GRAPH_RDLOCK +-parallels_co_block_status(BlockDriverState *bs, bool want_zero, int64_t offset, +- int64_t bytes, int64_t *pnum, int64_t *map, +- BlockDriverState **file) ++parallels_co_block_status(BlockDriverState *bs, unsigned int mode, ++ int64_t offset, int64_t bytes, int64_t *pnum, ++ int64_t *map, BlockDriverState **file) + { + BDRVParallelsState *s = bs->opaque; + int count; +diff --git a/block/qcow.c b/block/qcow.c +index c2f89db055..2e18c42d8f 100644 +--- a/block/qcow.c ++++ b/block/qcow.c +@@ -530,7 +530,7 @@ get_cluster_offset(BlockDriverState *bs, uint64_t offset, int allocate, + } + + static int coroutine_fn GRAPH_RDLOCK +-qcow_co_block_status(BlockDriverState *bs, bool want_zero, ++qcow_co_block_status(BlockDriverState *bs, unsigned int mode, + int64_t offset, int64_t bytes, int64_t *pnum, + int64_t *map, BlockDriverState **file) + { +diff --git a/block/qcow2.c b/block/qcow2.c +index a4cffb628c..788da07fee 100644 +--- a/block/qcow2.c ++++ b/block/qcow2.c +@@ -2147,9 +2147,9 @@ static void qcow2_join_options(QDict *options, QDict *old_options) + } + + static int coroutine_fn GRAPH_RDLOCK +-qcow2_co_block_status(BlockDriverState *bs, bool want_zero, int64_t offset, +- int64_t count, int64_t *pnum, int64_t *map, +- BlockDriverState **file) ++qcow2_co_block_status(BlockDriverState *bs, unsigned int mode, ++ int64_t offset, int64_t count, int64_t *pnum, ++ int64_t *map, BlockDriverState **file) + { + BDRVQcow2State *s = bs->opaque; + uint64_t host_offset; +diff --git a/block/qed.c b/block/qed.c +index fa5bc11085..b135e981e5 100644 +--- a/block/qed.c ++++ b/block/qed.c +@@ -832,9 +832,9 @@ fail: + } + + static int coroutine_fn GRAPH_RDLOCK +-bdrv_qed_co_block_status(BlockDriverState *bs, bool want_zero, int64_t pos, +- int64_t bytes, int64_t *pnum, int64_t *map, +- BlockDriverState **file) ++bdrv_qed_co_block_status(BlockDriverState *bs, unsigned int mode, ++ int64_t pos, int64_t bytes, int64_t *pnum, ++ int64_t *map, BlockDriverState **file) + { + BDRVQEDState *s = bs->opaque; + size_t len = MIN(bytes, SIZE_MAX); +diff --git a/block/quorum.c b/block/quorum.c +index db8fe891c4..bb4ed9483e 100644 +--- a/block/quorum.c ++++ b/block/quorum.c +@@ -1226,7 +1226,7 @@ static void quorum_child_perm(BlockDriverState *bs, BdrvChild *c, + * region contains zeroes, and BDRV_BLOCK_DATA otherwise. + */ + static int coroutine_fn GRAPH_RDLOCK +-quorum_co_block_status(BlockDriverState *bs, bool want_zero, ++quorum_co_block_status(BlockDriverState *bs, unsigned int mode, + int64_t offset, int64_t count, + int64_t *pnum, int64_t *map, BlockDriverState **file) + { +@@ -1238,7 +1238,7 @@ quorum_co_block_status(BlockDriverState *bs, bool want_zero, + for (i = 0; i < s->num_children; i++) { + int64_t bytes; + ret = bdrv_co_common_block_status_above(s->children[i]->bs, NULL, false, +- want_zero, offset, count, ++ mode, offset, count, + &bytes, NULL, NULL, NULL); + if (ret < 0) { + quorum_report_bad(QUORUM_OP_TYPE_READ, offset, count, +diff --git a/block/raw-format.c b/block/raw-format.c +index ac7e8495f6..623bca87a6 100644 +--- a/block/raw-format.c ++++ b/block/raw-format.c +@@ -283,8 +283,8 @@ fail: + } + + static int coroutine_fn GRAPH_RDLOCK +-raw_co_block_status(BlockDriverState *bs, bool want_zero, int64_t offset, +- int64_t bytes, int64_t *pnum, int64_t *map, ++raw_co_block_status(BlockDriverState *bs, unsigned int mode, ++ int64_t offset, int64_t bytes, int64_t *pnum, int64_t *map, + BlockDriverState **file) + { + BDRVRawState *s = bs->opaque; +diff --git a/block/rbd.c b/block/rbd.c +index 9c0fd0cb3f..627f8eb05a 100644 +--- a/block/rbd.c ++++ b/block/rbd.c +@@ -1504,9 +1504,9 @@ static int qemu_rbd_diff_iterate_cb(uint64_t offs, size_t len, + } + + static int coroutine_fn qemu_rbd_co_block_status(BlockDriverState *bs, +- bool want_zero, int64_t offset, +- int64_t bytes, int64_t *pnum, +- int64_t *map, ++ unsigned int mode, ++ int64_t offset, int64_t bytes, ++ int64_t *pnum, int64_t *map, + BlockDriverState **file) + { + BDRVRBDState *s = bs->opaque; +diff --git a/block/snapshot-access.c b/block/snapshot-access.c +index 84d0d13f86..972b8f2e68 100644 +--- a/block/snapshot-access.c ++++ b/block/snapshot-access.c +@@ -41,11 +41,11 @@ snapshot_access_co_preadv_part(BlockDriverState *bs, + + static int coroutine_fn GRAPH_RDLOCK + snapshot_access_co_block_status(BlockDriverState *bs, +- bool want_zero, int64_t offset, ++ unsigned int mode, int64_t offset, + int64_t bytes, int64_t *pnum, + int64_t *map, BlockDriverState **file) + { +- return bdrv_co_snapshot_block_status(bs->file->bs, want_zero, offset, ++ return bdrv_co_snapshot_block_status(bs->file->bs, mode, offset, + bytes, pnum, map, file); + } + +diff --git a/block/vdi.c b/block/vdi.c +index 6363da08ce..028fe68488 100644 +--- a/block/vdi.c ++++ b/block/vdi.c +@@ -521,8 +521,8 @@ static int vdi_reopen_prepare(BDRVReopenState *state, + } + + static int coroutine_fn GRAPH_RDLOCK +-vdi_co_block_status(BlockDriverState *bs, bool want_zero, int64_t offset, +- int64_t bytes, int64_t *pnum, int64_t *map, ++vdi_co_block_status(BlockDriverState *bs, unsigned int mode, ++ int64_t offset, int64_t bytes, int64_t *pnum, int64_t *map, + BlockDriverState **file) + { + BDRVVdiState *s = (BDRVVdiState *)bs->opaque; +diff --git a/block/vmdk.c b/block/vmdk.c +index 78f6433607..6f1af82078 100644 +--- a/block/vmdk.c ++++ b/block/vmdk.c +@@ -1777,7 +1777,7 @@ static inline uint64_t vmdk_find_offset_in_cluster(VmdkExtent *extent, + } + + static int coroutine_fn GRAPH_RDLOCK +-vmdk_co_block_status(BlockDriverState *bs, bool want_zero, ++vmdk_co_block_status(BlockDriverState *bs, unsigned int mode, + int64_t offset, int64_t bytes, int64_t *pnum, + int64_t *map, BlockDriverState **file) + { +diff --git a/block/vpc.c b/block/vpc.c +index d95a204612..0dd641b614 100644 +--- a/block/vpc.c ++++ b/block/vpc.c +@@ -721,7 +721,7 @@ fail: + } + + static int coroutine_fn GRAPH_RDLOCK +-vpc_co_block_status(BlockDriverState *bs, bool want_zero, ++vpc_co_block_status(BlockDriverState *bs, unsigned int mode, + int64_t offset, int64_t bytes, + int64_t *pnum, int64_t *map, + BlockDriverState **file) +diff --git a/block/vvfat.c b/block/vvfat.c +index 8ffe8b3b9b..d59231357e 100644 +--- a/block/vvfat.c ++++ b/block/vvfat.c +@@ -3135,9 +3135,9 @@ vvfat_co_pwritev(BlockDriverState *bs, int64_t offset, int64_t bytes, + } + + static int coroutine_fn vvfat_co_block_status(BlockDriverState *bs, +- bool want_zero, int64_t offset, +- int64_t bytes, int64_t *n, +- int64_t *map, ++ unsigned int mode, ++ int64_t offset, int64_t bytes, ++ int64_t *n, int64_t *map, + BlockDriverState **file) + { + *n = bytes; +diff --git a/include/block/block-common.h b/include/block/block-common.h +index 7030669f04..5beee6402b 100644 +--- a/include/block/block-common.h ++++ b/include/block/block-common.h +@@ -333,6 +333,17 @@ typedef enum { + #define BDRV_BLOCK_RECURSE 0x40 + #define BDRV_BLOCK_COMPRESSED 0x80 + ++/* ++ * Block status hints: the bitwise-or of these flags emphasize what ++ * the caller hopes to learn, and some drivers may be able to give ++ * faster answers by doing less work when the hint permits. ++ */ ++#define BDRV_WANT_ZERO BDRV_BLOCK_ZERO ++#define BDRV_WANT_OFFSET_VALID BDRV_BLOCK_OFFSET_VALID ++#define BDRV_WANT_ALLOCATED BDRV_BLOCK_ALLOCATED ++#define BDRV_WANT_PRECISE (BDRV_WANT_ZERO | BDRV_WANT_OFFSET_VALID | \ ++ BDRV_WANT_OFFSET_VALID) ++ + typedef QTAILQ_HEAD(BlockReopenQueue, BlockReopenQueueEntry) BlockReopenQueue; + + typedef struct BDRVReopenState { +diff --git a/include/block/block_int-common.h b/include/block/block_int-common.h +index ebb4e56a50..a9c0daa2a4 100644 +--- a/include/block/block_int-common.h ++++ b/include/block/block_int-common.h +@@ -608,15 +608,16 @@ struct BlockDriver { + * according to the current layer, and should only need to set + * BDRV_BLOCK_DATA, BDRV_BLOCK_ZERO, BDRV_BLOCK_OFFSET_VALID, + * and/or BDRV_BLOCK_RAW; if the current layer defers to a backing +- * layer, the result should be 0 (and not BDRV_BLOCK_ZERO). See +- * block.h for the overall meaning of the bits. As a hint, the +- * flag want_zero is true if the caller cares more about precise +- * mappings (favor accurate _OFFSET_VALID/_ZERO) or false for +- * overall allocation (favor larger *pnum, perhaps by reporting +- * _DATA instead of _ZERO). The block layer guarantees input +- * clamped to bdrv_getlength() and aligned to request_alignment, +- * as well as non-NULL pnum, map, and file; in turn, the driver +- * must return an error or set pnum to an aligned non-zero value. ++ * layer, the result should be 0 (and not BDRV_BLOCK_ZERO). The ++ * caller will synthesize BDRV_BLOCK_ALLOCATED based on the ++ * non-zero results. See block.h for the overall meaning of the ++ * bits. As a hint, the flags in @mode may include a bitwise-or ++ * of BDRV_WANT_ALLOCATED, BDRV_WANT_OFFSET_VALID, or ++ * BDRV_WANT_ZERO based on what the caller is looking for in the ++ * results. The block layer guarantees input clamped to ++ * bdrv_getlength() and aligned to request_alignment, as well as ++ * non-NULL pnum, map, and file; in turn, the driver must return ++ * an error or set pnum to an aligned non-zero value. + * + * Note that @bytes is just a hint on how big of a region the + * caller wants to inspect. It is not a limit on *pnum. +@@ -628,8 +629,8 @@ struct BlockDriver { + * to clamping *pnum for return to its caller. + */ + int coroutine_fn GRAPH_RDLOCK_PTR (*bdrv_co_block_status)( +- BlockDriverState *bs, +- bool want_zero, int64_t offset, int64_t bytes, int64_t *pnum, ++ BlockDriverState *bs, unsigned int mode, ++ int64_t offset, int64_t bytes, int64_t *pnum, + int64_t *map, BlockDriverState **file); + + /* +@@ -653,8 +654,8 @@ struct BlockDriver { + QEMUIOVector *qiov, size_t qiov_offset); + + int coroutine_fn GRAPH_RDLOCK_PTR (*bdrv_co_snapshot_block_status)( +- BlockDriverState *bs, bool want_zero, int64_t offset, int64_t bytes, +- int64_t *pnum, int64_t *map, BlockDriverState **file); ++ BlockDriverState *bs, unsigned int mode, int64_t offset, ++ int64_t bytes, int64_t *pnum, int64_t *map, BlockDriverState **file); + + int coroutine_fn GRAPH_RDLOCK_PTR (*bdrv_co_pdiscard_snapshot)( + BlockDriverState *bs, int64_t offset, int64_t bytes); +diff --git a/include/block/block_int-io.h b/include/block/block_int-io.h +index 4a7cf2b4fd..4f94eb3c5a 100644 +--- a/include/block/block_int-io.h ++++ b/include/block/block_int-io.h +@@ -38,8 +38,8 @@ + int coroutine_fn GRAPH_RDLOCK bdrv_co_preadv_snapshot(BdrvChild *child, + int64_t offset, int64_t bytes, QEMUIOVector *qiov, size_t qiov_offset); + int coroutine_fn GRAPH_RDLOCK bdrv_co_snapshot_block_status( +- BlockDriverState *bs, bool want_zero, int64_t offset, int64_t bytes, +- int64_t *pnum, int64_t *map, BlockDriverState **file); ++ BlockDriverState *bs, unsigned int mode, int64_t offset, ++ int64_t bytes, int64_t *pnum, int64_t *map, BlockDriverState **file); + int coroutine_fn GRAPH_RDLOCK bdrv_co_pdiscard_snapshot(BlockDriverState *bs, + int64_t offset, int64_t bytes); + +diff --git a/tests/unit/test-block-iothread.c b/tests/unit/test-block-iothread.c +index 3766d5de6b..373b72fdd8 100644 +--- a/tests/unit/test-block-iothread.c ++++ b/tests/unit/test-block-iothread.c +@@ -63,7 +63,7 @@ bdrv_test_co_truncate(BlockDriverState *bs, int64_t offset, bool exact, + } + + static int coroutine_fn bdrv_test_co_block_status(BlockDriverState *bs, +- bool want_zero, ++ unsigned int mode, + int64_t offset, int64_t count, + int64_t *pnum, int64_t *map, + BlockDriverState **file) +-- +2.48.1 + diff --git a/SOURCES/kvm-block-Let-bdrv_co_is_zero_fast-consolidate-adjacent-.patch b/SOURCES/kvm-block-Let-bdrv_co_is_zero_fast-consolidate-adjacent-.patch new file mode 100644 index 0000000..8b86f66 --- /dev/null +++ b/SOURCES/kvm-block-Let-bdrv_co_is_zero_fast-consolidate-adjacent-.patch @@ -0,0 +1,90 @@ +From 9f8158e56beae4221e91feb5a98cb4db9076cac4 Mon Sep 17 00:00:00 2001 +From: Eric Blake +Date: Fri, 9 May 2025 15:40:20 -0500 +Subject: [PATCH 05/16] block: Let bdrv_co_is_zero_fast consolidate adjacent + extents + +RH-Author: Eric Blake +RH-MergeRequest: 365: blockdev-mirror: More efficient handling of sparse mirrors +RH-Jira: RHEL-82906 RHEL-83015 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Jon Maloy +RH-Commit: [3/14] 98bf9ff773d9a36f8a8e294e38629e3f20c41334 (ebblake/centos-qemu-kvm) + +Some BDS drivers have a cap on how much block status they can supply +in one query (for example, NBD talking to an older server cannot +inspect more than 4G per query; and qcow2 tends to cap its answers +rather than cross a cluster boundary of an L1 table). Although the +existing callers of bdrv_co_is_zero_fast are not passing in that large +of a 'bytes' parameter, an upcoming caller wants to query the entire +image at once, and will thus benefit from being able to treat adjacent +zero regions in a coalesced manner, rather than claiming the region is +non-zero merely because pnum was truncated and didn't match the +incoming bytes. + +While refactoring this into a loop, note that there is no need to +assign pnum prior to calling bdrv_co_common_block_status_above() (it +is guaranteed to be assigned deeper in the callstack). + +Signed-off-by: Eric Blake +Reviewed-by: Stefan Hajnoczi +Message-ID: <20250509204341.3553601-18-eblake@redhat.com> +(cherry picked from commit 31bf15d97dd1d205a3b264675f9a1b3bd1939068) +Jira: https://issues.redhat.com/browse/RHEL-82906 +Jira: https://issues.redhat.com/browse/RHEL-83015 +Signed-off-by: Eric Blake +--- + block/io.c | 27 +++++++++++++++------------ + 1 file changed, 15 insertions(+), 12 deletions(-) + +diff --git a/block/io.c b/block/io.c +index daaafe00d7..293c5dd393 100644 +--- a/block/io.c ++++ b/block/io.c +@@ -2747,28 +2747,31 @@ int coroutine_fn bdrv_co_block_status(BlockDriverState *bs, int64_t offset, + * by @offset and @bytes is known to read as zeroes. + * Return 1 if that is the case, 0 otherwise and -errno on error. + * This test is meant to be fast rather than accurate so returning 0 +- * does not guarantee non-zero data. ++ * does not guarantee non-zero data; but a return of 1 is reliable. + */ + int coroutine_fn bdrv_co_is_zero_fast(BlockDriverState *bs, int64_t offset, + int64_t bytes) + { + int ret; +- int64_t pnum = bytes; ++ int64_t pnum; + IO_CODE(); + +- if (!bytes) { +- return 1; +- } +- +- ret = bdrv_co_common_block_status_above(bs, NULL, false, BDRV_WANT_ZERO, +- offset, bytes, &pnum, NULL, NULL, +- NULL); ++ while (bytes) { ++ ret = bdrv_co_common_block_status_above(bs, NULL, false, ++ BDRV_WANT_ZERO, offset, bytes, ++ &pnum, NULL, NULL, NULL); + +- if (ret < 0) { +- return ret; ++ if (ret < 0) { ++ return ret; ++ } ++ if (!(ret & BDRV_BLOCK_ZERO)) { ++ return 0; ++ } ++ offset += pnum; ++ bytes -= pnum; + } + +- return (pnum == bytes) && (ret & BDRV_BLOCK_ZERO); ++ return 1; + } + + int coroutine_fn bdrv_co_is_allocated(BlockDriverState *bs, int64_t offset, +-- +2.48.1 + diff --git a/SOURCES/kvm-block-io-skip-head-tail-requests-on-EINVAL.patch b/SOURCES/kvm-block-io-skip-head-tail-requests-on-EINVAL.patch index f82bc66..42e6ecf 100644 --- a/SOURCES/kvm-block-io-skip-head-tail-requests-on-EINVAL.patch +++ b/SOURCES/kvm-block-io-skip-head-tail-requests-on-EINVAL.patch @@ -1,14 +1,14 @@ -From ecdc254dbaa7995a94f67e7dfafb17137d15759e Mon Sep 17 00:00:00 2001 +From e629a362860977161e43ed80bb59d1d05a06b2f2 Mon Sep 17 00:00:00 2001 From: Stefan Hajnoczi Date: Thu, 17 Apr 2025 11:05:28 -0400 -Subject: [PATCH 2/3] block/io: skip head/tail requests on EINVAL +Subject: [PATCH 4/5] block/io: skip head/tail requests on EINVAL RH-Author: Stefan Hajnoczi -RH-MergeRequest: 450: file-posix: probe discard alignment on Linux block devices -RH-Jira: RHEL-87734 +RH-MergeRequest: 355: file-posix: probe discard alignment on Linux block devices +RH-Jira: RHEL-86032 RH-Acked-by: Kevin Wolf RH-Acked-by: Eric Blake -RH-Commit: [2/3] 30b17fc1828c45cf958b8254999ce1ef1f100868 +RH-Commit: [2/3] 0028fb11f18e16e2aba9506eabb2383c406d17b5 (stefanha/centos-stream-qemu-kvm) When guests send misaligned discard requests, the block layer breaks them up into a misaligned head, an aligned main body, and a misaligned diff --git a/SOURCES/kvm-block-skip-automatic-zero-init-of-large-array-in-ioq.patch b/SOURCES/kvm-block-skip-automatic-zero-init-of-large-array-in-ioq.patch index 9582a9f..72c7a02 100644 --- a/SOURCES/kvm-block-skip-automatic-zero-init-of-large-array-in-ioq.patch +++ b/SOURCES/kvm-block-skip-automatic-zero-init-of-large-array-in-ioq.patch @@ -1,17 +1,17 @@ -From 9f8ff1d0ef010b9c0339869f655ee9af6b10dcd5 Mon Sep 17 00:00:00 2001 +From d38bdce712f572e1920e3344132ff6600d657de2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:41 +0100 -Subject: [PATCH 04/31] block: skip automatic zero-init of large array in +Subject: [PATCH 29/57] block: skip automatic zero-init of large array in ioq_submit MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [3/30] 0a24695ab7f3a11ab61b12ae2b95bd45a7babc05 +RH-Commit: [3/30] 301a08b3acdcd95634dec5dab1d96fcfe3abf3be (stefanha/centos-stream-qemu-kvm) The 'ioq_submit' method has a struct array that is 8k in size. Skip the automatic zero-init of this array to eliminate the diff --git a/SOURCES/kvm-chardev-char-fd-skip-automatic-zero-init-of-large-ar.patch b/SOURCES/kvm-chardev-char-fd-skip-automatic-zero-init-of-large-ar.patch index 90c5387..4d56bf1 100644 --- a/SOURCES/kvm-chardev-char-fd-skip-automatic-zero-init-of-large-ar.patch +++ b/SOURCES/kvm-chardev-char-fd-skip-automatic-zero-init-of-large-ar.patch @@ -1,17 +1,17 @@ -From c336909ad95540147d9cfed843874ddd986ad917 Mon Sep 17 00:00:00 2001 +From 1e8798a3adbbfc42167aaba0ee18175deac37193 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:42 +0100 -Subject: [PATCH 05/31] chardev/char-fd: skip automatic zero-init of large +Subject: [PATCH 30/57] chardev/char-fd: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [4/30] e9adf42d47ddd90cc15862e711de0a90d1d25e0a +RH-Commit: [4/30] b16fe5c9af4756e1856cd330df02a1a09d9f33ea (stefanha/centos-stream-qemu-kvm) The 'fd_chr_read' method has a 4k byte array used for copying data between the socket and device. Skip the automatic zero-init diff --git a/SOURCES/kvm-chardev-char-pty-skip-automatic-zero-init-of-large-a.patch b/SOURCES/kvm-chardev-char-pty-skip-automatic-zero-init-of-large-a.patch index 333fdfd..7edacc8 100644 --- a/SOURCES/kvm-chardev-char-pty-skip-automatic-zero-init-of-large-a.patch +++ b/SOURCES/kvm-chardev-char-pty-skip-automatic-zero-init-of-large-a.patch @@ -1,17 +1,17 @@ -From 489ead1e7d721c0f689626e6d5d22241ffdd7bc8 Mon Sep 17 00:00:00 2001 +From 74311b0ee8e211fccff211b975e4ae9236c063dc Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:43 +0100 -Subject: [PATCH 06/31] chardev/char-pty: skip automatic zero-init of large +Subject: [PATCH 31/57] chardev/char-pty: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [5/30] 09be65fa8c25dadbd472321cacc96badf0a2d963 +RH-Commit: [5/30] a3b8458c30f485551093f292c00c20b0e118df77 (stefanha/centos-stream-qemu-kvm) The 'pty_chr_read' method has a 4k byte array used for copying data between the PTY and device. Skip the automatic zero-init diff --git a/SOURCES/kvm-chardev-char-socket-skip-automatic-zero-init-of-larg.patch b/SOURCES/kvm-chardev-char-socket-skip-automatic-zero-init-of-larg.patch index fb89fa5..3b6889b 100644 --- a/SOURCES/kvm-chardev-char-socket-skip-automatic-zero-init-of-larg.patch +++ b/SOURCES/kvm-chardev-char-socket-skip-automatic-zero-init-of-larg.patch @@ -1,17 +1,17 @@ -From 67cf0b18b68071b7b0a036b715f9d406f0bc1ec7 Mon Sep 17 00:00:00 2001 +From d56a8ce56f0de70ab2de266a80e25cf309e72fda Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:44 +0100 -Subject: [PATCH 07/31] chardev/char-socket: skip automatic zero-init of large +Subject: [PATCH 32/57] chardev/char-socket: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [6/30] fa1d406d2f9be389090614d666466fb94f480f84 +RH-Commit: [6/30] 86a2ac03efa1838fb30931c38945ee77de9bbe06 (stefanha/centos-stream-qemu-kvm) The 'tcp_chr_read' method has a 4k byte array used for copying data between the socket and device. Skip the automatic zero-init diff --git a/SOURCES/kvm-cpu-Don-t-set-vcpu_dirty-when-guest_state_protected.patch b/SOURCES/kvm-cpu-Don-t-set-vcpu_dirty-when-guest_state_protected.patch new file mode 100644 index 0000000..b07a22d --- /dev/null +++ b/SOURCES/kvm-cpu-Don-t-set-vcpu_dirty-when-guest_state_protected.patch @@ -0,0 +1,48 @@ +From dd4ab64754a52f1e50273cb8153567b0d2f382de Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 071/115] cpu: Don't set vcpu_dirty when guest_state_protected + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [71/115] c876a59ee5bbfccea9837ac373a2e664354db3ae (bonzini/rhel-qemu-kvm) + +QEMU calls kvm_arch_put_registers() when vcpu_dirty is true in +kvm_vcpu_exec(). However, for confidential guest, like TDX, putting +registers is disallowed due to guest state is protected. + +Only set vcpu_dirty to true with guest state is not protected when +creating the vcpu. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-43-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit b4b7fb5a773e1d2215c2aaa99789eca51914b78f) +Signed-off-by: Paolo Bonzini +--- + accel/kvm/kvm-all.c | 4 +++- + 1 file changed, 3 insertions(+), 1 deletion(-) + +diff --git a/accel/kvm/kvm-all.c b/accel/kvm/kvm-all.c +index c1605bc4fa..43c10c82f6 100644 +--- a/accel/kvm/kvm-all.c ++++ b/accel/kvm/kvm-all.c +@@ -456,7 +456,9 @@ int kvm_create_vcpu(CPUState *cpu) + + cpu->kvm_fd = kvm_fd; + cpu->kvm_state = s; +- cpu->vcpu_dirty = true; ++ if (!s->guest_state_protected) { ++ cpu->vcpu_dirty = true; ++ } + cpu->dirty_pages = 0; + cpu->throttle_us_per_full = 0; + +-- +2.50.1 + diff --git a/SOURCES/kvm-cpu-Remove-nr_cores-from-struct-CPUState.patch b/SOURCES/kvm-cpu-Remove-nr_cores-from-struct-CPUState.patch new file mode 100644 index 0000000..b61e5f3 --- /dev/null +++ b/SOURCES/kvm-cpu-Remove-nr_cores-from-struct-CPUState.patch @@ -0,0 +1,76 @@ +From c57b5e38fd95a68f36a342e19ba7ccb6cbb07948 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 014/115] cpu: Remove nr_cores from struct CPUState + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [14/115] 80f6414c5f8f5e963b0f2251147b8a1ca04f55e4 (bonzini/rhel-qemu-kvm) + +There is no user of it now, remove it. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20241219110125.1266461-9-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 6e090ffe0d188e1f09d4efcd10d82158f92abfbb) +Signed-off-by: Paolo Bonzini +(cherry picked from commit d4c699c310519b99bedf1bdb516cab230d5d846c) +Signed-off-by: Paolo Bonzini +--- + hw/core/cpu-common.c | 1 - + include/hw/core/cpu.h | 2 -- + system/cpus.c | 1 - + 3 files changed, 4 deletions(-) + +diff --git a/hw/core/cpu-common.c b/hw/core/cpu-common.c +index 7982ecd39a..1ac8ab488f 100644 +--- a/hw/core/cpu-common.c ++++ b/hw/core/cpu-common.c +@@ -242,7 +242,6 @@ static void cpu_common_initfn(Object *obj) + cpu->cluster_index = UNASSIGNED_CLUSTER_INDEX; + /* user-mode doesn't have configurable SMP topology */ + /* the default value is changed by qemu_init_vcpu() for system-mode */ +- cpu->nr_cores = 1; + cpu->nr_threads = 1; + cpu->cflags_next_tb = -1; + +diff --git a/include/hw/core/cpu.h b/include/hw/core/cpu.h +index 1c9c775df6..d90e3b3f2c 100644 +--- a/include/hw/core/cpu.h ++++ b/include/hw/core/cpu.h +@@ -402,7 +402,6 @@ struct qemu_work_item; + * Under TCG this value is propagated to @tcg_cflags. + * See TranslationBlock::TCG CF_CLUSTER_MASK. + * @tcg_cflags: Pre-computed cflags for this cpu. +- * @nr_cores: Number of cores within this CPU package. + * @nr_threads: Number of threads within this CPU core. + * @thread: Host thread details, only live once @created is #true + * @sem: WIN32 only semaphore used only for qtest +@@ -461,7 +460,6 @@ struct CPUState { + CPUClass *cc; + /*< public >*/ + +- int nr_cores; + int nr_threads; + + struct QemuThread *thread; +diff --git a/system/cpus.c b/system/cpus.c +index 1c818ff682..909d8128e8 100644 +--- a/system/cpus.c ++++ b/system/cpus.c +@@ -666,7 +666,6 @@ void qemu_init_vcpu(CPUState *cpu) + { + MachineState *ms = MACHINE(qdev_get_machine()); + +- cpu->nr_cores = machine_topo_get_cores_per_socket(ms); + cpu->nr_threads = ms->smp.threads; + cpu->stopped = true; + cpu->random_seed = qemu_guest_random_seed_thread_part1(); +-- +2.50.1 + diff --git a/SOURCES/kvm-crypto-Define-macros-for-hash-algorithm-digest-lengt.patch b/SOURCES/kvm-crypto-Define-macros-for-hash-algorithm-digest-lengt.patch new file mode 100644 index 0000000..5712581 --- /dev/null +++ b/SOURCES/kvm-crypto-Define-macros-for-hash-algorithm-digest-lengt.patch @@ -0,0 +1,74 @@ +From 4df071fec89ab867f8e2d970de48256034e4b286 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:29:04 +0200 +Subject: [PATCH 005/115] crypto: Define macros for hash algorithm digest + lengths +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [5/115] f5f6e0c3cd10baf7a490b67dd6c0542bddba8dfe (bonzini/rhel-qemu-kvm) + +Reviewed-by: Daniel P. Berrangé +Signed-off-by: Dorjoy Chowdhury +Signed-off-by: Daniel P. Berrangé +(cherry picked from commit 5d04de7de54e163b056980be10ee1c281a600276) +Signed-off-by: Paolo Bonzini +--- + crypto/hash.c | 14 +++++++------- + include/crypto/hash.h | 8 ++++++++ + 2 files changed, 15 insertions(+), 7 deletions(-) + +diff --git a/crypto/hash.c b/crypto/hash.c +index b0f8228bdc..8087f5dae6 100644 +--- a/crypto/hash.c ++++ b/crypto/hash.c +@@ -23,13 +23,13 @@ + #include "hashpriv.h" + + static size_t qcrypto_hash_alg_size[QCRYPTO_HASH_ALG__MAX] = { +- [QCRYPTO_HASH_ALG_MD5] = 16, +- [QCRYPTO_HASH_ALG_SHA1] = 20, +- [QCRYPTO_HASH_ALG_SHA224] = 28, +- [QCRYPTO_HASH_ALG_SHA256] = 32, +- [QCRYPTO_HASH_ALG_SHA384] = 48, +- [QCRYPTO_HASH_ALG_SHA512] = 64, +- [QCRYPTO_HASH_ALG_RIPEMD160] = 20, ++ [QCRYPTO_HASH_ALG_MD5] = QCRYPTO_HASH_DIGEST_LEN_MD5, ++ [QCRYPTO_HASH_ALG_SHA1] = QCRYPTO_HASH_DIGEST_LEN_SHA1, ++ [QCRYPTO_HASH_ALG_SHA224] = QCRYPTO_HASH_DIGEST_LEN_SHA224, ++ [QCRYPTO_HASH_ALG_SHA256] = QCRYPTO_HASH_DIGEST_LEN_SHA256, ++ [QCRYPTO_HASH_ALG_SHA384] = QCRYPTO_HASH_DIGEST_LEN_SHA384, ++ [QCRYPTO_HASH_ALG_SHA512] = QCRYPTO_HASH_DIGEST_LEN_SHA512, ++ [QCRYPTO_HASH_ALG_RIPEMD160] = QCRYPTO_HASH_DIGEST_LEN_RIPEMD160, + }; + + size_t qcrypto_hash_digest_len(QCryptoHashAlgorithm alg) +diff --git a/include/crypto/hash.h b/include/crypto/hash.h +index 54d87aa2a1..a113cc3b04 100644 +--- a/include/crypto/hash.h ++++ b/include/crypto/hash.h +@@ -23,6 +23,14 @@ + + #include "qapi/qapi-types-crypto.h" + ++#define QCRYPTO_HASH_DIGEST_LEN_MD5 16 ++#define QCRYPTO_HASH_DIGEST_LEN_SHA1 20 ++#define QCRYPTO_HASH_DIGEST_LEN_SHA224 28 ++#define QCRYPTO_HASH_DIGEST_LEN_SHA256 32 ++#define QCRYPTO_HASH_DIGEST_LEN_SHA384 48 ++#define QCRYPTO_HASH_DIGEST_LEN_SHA512 64 ++#define QCRYPTO_HASH_DIGEST_LEN_RIPEMD160 20 ++ + /* See also "QCryptoHashAlgorithm" defined in qapi/crypto.json */ + + /** +-- +2.50.1 + diff --git a/SOURCES/kvm-docs-Add-TDX-documentation.patch b/SOURCES/kvm-docs-Add-TDX-documentation.patch new file mode 100644 index 0000000..3b0a3ae --- /dev/null +++ b/SOURCES/kvm-docs-Add-TDX-documentation.patch @@ -0,0 +1,222 @@ +From a9c7bbb7a32ba2ea5cd76b87c41f1fbdd789fb3b Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 084/115] docs: Add TDX documentation + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [84/115] 4f9930f4415e7195bf5bbbcc0d79cfc6aaa385e1 (bonzini/rhel-qemu-kvm) + +Add docs/system/i386/tdx.rst for TDX support, and add tdx in +confidential-guest-support.rst + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20250508150002.689633-56-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit dc1424319311f86449c6825ceec2364ee645a363) +Signed-off-by: Paolo Bonzini +--- + docs/system/confidential-guest-support.rst | 1 + + docs/system/i386/tdx.rst | 161 +++++++++++++++++++++ + docs/system/target-i386.rst | 1 + + 3 files changed, 163 insertions(+) + create mode 100644 docs/system/i386/tdx.rst + +diff --git a/docs/system/confidential-guest-support.rst b/docs/system/confidential-guest-support.rst +index 0c490dbda2..66129fbab6 100644 +--- a/docs/system/confidential-guest-support.rst ++++ b/docs/system/confidential-guest-support.rst +@@ -38,6 +38,7 @@ Supported mechanisms + Currently supported confidential guest mechanisms are: + + * AMD Secure Encrypted Virtualization (SEV) (see :doc:`i386/amd-memory-encryption`) ++* Intel Trust Domain Extension (TDX) (see :doc:`i386/tdx`) + * POWER Protected Execution Facility (PEF) (see :ref:`power-papr-protected-execution-facility-pef`) + * s390x Protected Virtualization (PV) (see :doc:`s390x/protvirt`) + +diff --git a/docs/system/i386/tdx.rst b/docs/system/i386/tdx.rst +new file mode 100644 +index 0000000000..8131750b64 +--- /dev/null ++++ b/docs/system/i386/tdx.rst +@@ -0,0 +1,161 @@ ++Intel Trusted Domain eXtension (TDX) ++==================================== ++ ++Intel Trusted Domain eXtensions (TDX) refers to an Intel technology that extends ++Virtual Machine Extensions (VMX) and Multi-Key Total Memory Encryption (MKTME) ++with a new kind of virtual machine guest called a Trust Domain (TD). A TD runs ++in a CPU mode that is designed to protect the confidentiality of its memory ++contents and its CPU state from any other software, including the hosting ++Virtual Machine Monitor (VMM), unless explicitly shared by the TD itself. ++ ++Prerequisites ++------------- ++ ++To run TD, the physical machine needs to have TDX module loaded and initialized ++while KVM hypervisor has TDX support and has TDX enabled. If those requirements ++are met, the ``KVM_CAP_VM_TYPES`` will report the support of ``KVM_X86_TDX_VM``. ++ ++Trust Domain Virtual Firmware (TDVF) ++~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ ++ ++Trust Domain Virtual Firmware (TDVF) is required to provide TD services to boot ++TD Guest OS. TDVF needs to be copied to guest private memory and measured before ++the TD boots. ++ ++KVM vcpu ioctl ``KVM_TDX_INIT_MEM_REGION`` can be used to populate the TDVF ++content into its private memory. ++ ++Since TDX doesn't support readonly memslot, TDVF cannot be mapped as pflash ++device and it actually works as RAM. "-bios" option is chosen to load TDVF. ++ ++OVMF is the opensource firmware that implements the TDVF support. Thus the ++command line to specify and load TDVF is ``-bios OVMF.fd`` ++ ++Feature Configuration ++--------------------- ++ ++Unlike non-TDX VM, the CPU features (enumerated by CPU or MSR) of a TD are not ++under full control of VMM. VMM can only configure part of features of a TD on ++``KVM_TDX_INIT_VM`` command of VM scope ``MEMORY_ENCRYPT_OP`` ioctl. ++ ++The configurable features have three types: ++ ++- Attributes: ++ - PKS (bit 30) controls whether Supervisor Protection Keys is exposed to TD, ++ which determines related CPUID bit and CR4 bit; ++ - PERFMON (bit 63) controls whether PMU is exposed to TD. ++ ++- XSAVE related features (XFAM): ++ XFAM is a 64b mask, which has the same format as XCR0 or IA32_XSS MSR. It ++ determines the set of extended features available for use by the guest TD. ++ ++- CPUID features: ++ Only some bits of some CPUID leaves are directly configurable by VMM. ++ ++What features can be configured is reported via TDX capabilities. ++ ++TDX capabilities ++~~~~~~~~~~~~~~~~ ++ ++The VM scope ``MEMORY_ENCRYPT_OP`` ioctl provides command ``KVM_TDX_CAPABILITIES`` ++to get the TDX capabilities from KVM. It returns a data structure of ++``struct kvm_tdx_capabilities``, which tells the supported configuration of ++attributes, XFAM and CPUIDs. ++ ++TD attributes ++~~~~~~~~~~~~~ ++ ++QEMU supports configuring raw 64-bit TD attributes directly via "attributes" ++property of "tdx-guest" object. Note, it's users' responsibility to provide a ++valid value because some bits may not supported by current QEMU or KVM yet. ++ ++QEMU also supports the configuration of individual attribute bits that are ++supported by it, via properties of "tdx-guest" object. ++E.g., "sept-ve-disable" (bit 28). ++ ++MSR based features ++~~~~~~~~~~~~~~~~~~ ++ ++Current KVM doesn't support MSR based feature (e.g., MSR_IA32_ARCH_CAPABILITIES) ++configuration for TDX, and it's a future work to enable it in QEMU when KVM adds ++support of it. ++ ++Feature check ++~~~~~~~~~~~~~ ++ ++QEMU checks if the final (CPU) features, determined by given cpu model and ++explicit feature adjustment of "+featureA/-featureB", can be supported or not. ++It can produce feature not supported warning like ++ ++ "warning: host doesn't support requested feature: CPUID.07H:EBX.intel-pt [bit 25]" ++ ++It can also produce warning like ++ ++ "warning: TDX forcibly sets the feature: CPUID.80000007H:EDX.invtsc [bit 8]" ++ ++if the fixed-1 feature is requested to be disabled explicitly. This is newly ++added to QEMU for TDX because TDX has fixed-1 features that are forcibly enabled ++by TDX module and VMM cannot disable them. ++ ++Launching a TD (TDX VM) ++----------------------- ++ ++To launch a TD, the necessary command line options are tdx-guest object and ++split kernel-irqchip, as below: ++ ++.. parsed-literal:: ++ ++ |qemu_system_x86| \\ ++ -accel kvm \\ ++ -cpu host \\ ++ -object tdx-guest,id=tdx0 \\ ++ -machine ...,confidential-guest-support=tdx0 \\ ++ -bios OVMF.fd \\ ++ ++Restrictions ++------------ ++ ++ - kernel-irqchip must be split; ++ ++ This is set by default for TDX guest if kernel-irqchip is left on its default ++ 'auto' setting. ++ ++ - No readonly support for private memory; ++ ++ - No SMM support: SMM support requires manipulating the guest register states ++ which is not allowed; ++ ++Debugging ++--------- ++ ++Bit 0 of TD attributes, is DEBUG bit, which decides if the TD runs in off-TD ++debug mode. When in off-TD debug mode, TD's VCPU state and private memory are ++accessible via given SEAMCALLs. This requires KVM to expose APIs to invoke those ++SEAMCALLs and corresonponding QEMU change. ++ ++It's targeted as future work. ++ ++TD attestation ++-------------- ++ ++In TD guest, the attestation process is used to verify the TDX guest ++trustworthiness to other entities before provisioning secrets to the guest. ++ ++TD attestation is initiated first by calling TDG.MR.REPORT inside TD to get the ++REPORT. Then the REPORT data needs to be converted into a remotely verifiable ++Quote by SGX Quoting Enclave (QE). ++ ++It's a future work in QEMU to add support of TD attestation since it lacks ++support in current KVM. ++ ++Live Migration ++-------------- ++ ++Future work. ++ ++References ++---------- ++ ++- `TDX Homepage `__ ++ ++- `SGX QE `__ +diff --git a/docs/system/target-i386.rst b/docs/system/target-i386.rst +index 1b8a1f248a..4d58cdbc4e 100644 +--- a/docs/system/target-i386.rst ++++ b/docs/system/target-i386.rst +@@ -29,6 +29,7 @@ Architectural features + i386/kvm-pv + i386/sgx + i386/amd-memory-encryption ++ i386/tdx + + OS requirements + ~~~~~~~~~~~~~~~ +-- +2.50.1 + diff --git a/SOURCES/kvm-docs-devel-reset-Document-reset-expectations-for-DMA.patch b/SOURCES/kvm-docs-devel-reset-Document-reset-expectations-for-DMA.patch new file mode 100644 index 0000000..fcd1246 --- /dev/null +++ b/SOURCES/kvm-docs-devel-reset-Document-reset-expectations-for-DMA.patch @@ -0,0 +1,53 @@ +From 389c3c6b4215c9be3fd784c73af0e9795e796380 Mon Sep 17 00:00:00 2001 +From: Eric Auger +Date: Tue, 18 Feb 2025 19:25:35 +0100 +Subject: [PATCH 5/9] docs/devel/reset: Document reset expectations for DMA and + IOMMU +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Eric Auger +RH-MergeRequest: 341: Fix vIOMMU reset order +RH-Jira: RHEL-7188 +RH-Acked-by: Peter Xu +RH-Acked-by: Donald Dutile +RH-Acked-by: Cédric Le Goater +RH-Commit: [5/5] be8b9d9e34a2b301430dfa229c6785ab17d3fb16 (eauger1/centos-qemu-kvm) + +To avoid any translation faults, the IOMMUs are expected to be +reset after the devices they protect. Document that we expect +DMA requests to be stopped during the 'enter' or 'hold' phase +while IOMMUs should be reset during the 'exit' phase. + +Signed-off-by: Eric Auger +Reviewed-by: Zhenzhong Duan +Message-Id: <20250218182737.76722-6-eric.auger@redhat.com> +Reviewed-by: Peter Xu +Reviewed-by: Michael S. Tsirkin +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit dd6d545e8f2d9a0e8a8c287ec16469f03ef5c198) +Signed-off-by: Eric Auger +--- + docs/devel/reset.rst | 5 +++++ + 1 file changed, 5 insertions(+) + +diff --git a/docs/devel/reset.rst b/docs/devel/reset.rst +index 9746a4e8a0..24ab630465 100644 +--- a/docs/devel/reset.rst ++++ b/docs/devel/reset.rst +@@ -123,6 +123,11 @@ The *exit* phase is executed only when the last reset operation ends. Therefore + the object does not need to care how many of reset controllers it has and how + many of them have started a reset. + ++DMA capable devices are expected to cancel all outstanding DMA operations ++during either 'enter' or 'hold' phases. IOMMUs are expected to reset during ++the 'exit' phase and this sequencing makes sure no outstanding DMA request ++will fault. ++ + + Handling reset in a resettable object + ------------------------------------- +-- +2.48.1 + diff --git a/SOURCES/kvm-file-posix-Define-DM_MPATH_PROBE_PATHS.patch b/SOURCES/kvm-file-posix-Define-DM_MPATH_PROBE_PATHS.patch index 07a4834..08287d4 100644 --- a/SOURCES/kvm-file-posix-Define-DM_MPATH_PROBE_PATHS.patch +++ b/SOURCES/kvm-file-posix-Define-DM_MPATH_PROBE_PATHS.patch @@ -1,14 +1,14 @@ -From 762f24e92b93c0e8cbb5b0abe135d29fb444737d Mon Sep 17 00:00:00 2001 +From d565fe385b3c45a41fa8e25942220aff38a04fc3 Mon Sep 17 00:00:00 2001 From: Kevin Wolf Date: Tue, 29 Apr 2025 17:05:41 +0200 -Subject: [PATCH 1/2] file-posix: Define DM_MPATH_PROBE_PATHS +Subject: [PATCH 2/3] file-posix: Define DM_MPATH_PROBE_PATHS RH-Author: Kevin Wolf -RH-MergeRequest: 456: file-posix: Fix multipath failover with SCSI passthrough [9.6.z] -RH-Jira: RHEL-95407 +RH-MergeRequest: 372: file-posix: Fix multipath failover with SCSI passthrough [9.7] +RH-Jira: RHEL-95408 RH-Acked-by: Hanna Czenczek RH-Acked-by: Stefan Hajnoczi -RH-Commit: [1/2] 0d9ec74bf3bb999c8baa929e0d25682680fa1731 (kmwolf/rhel-qemu-kvm) +RH-Commit: [1/2] 7615906833a6bb2b4645fa5cd60d78aa9631cb7c (kmwolf/centos-qemu-kvm) While the kernel side isn't merged yet and we're still using old kernel headers, just define DM_MPATH_PROBE_PATHS manually. diff --git a/SOURCES/kvm-file-posix-Fix-crash-on-discard_granularity-0.patch b/SOURCES/kvm-file-posix-Fix-crash-on-discard_granularity-0.patch index 10c28df..8a45dcc 100644 --- a/SOURCES/kvm-file-posix-Fix-crash-on-discard_granularity-0.patch +++ b/SOURCES/kvm-file-posix-Fix-crash-on-discard_granularity-0.patch @@ -1,14 +1,14 @@ -From e7fac8bb7cedcb600bea245fc45089c7ff1e9728 Mon Sep 17 00:00:00 2001 +From 3515c6541f71817727a3a8b18ec5252644b51bc0 Mon Sep 17 00:00:00 2001 From: Kevin Wolf Date: Tue, 29 Apr 2025 17:56:54 +0200 -Subject: [PATCH 3/3] file-posix: Fix crash on discard_granularity == 0 +Subject: [PATCH 5/5] file-posix: Fix crash on discard_granularity == 0 RH-Author: Stefan Hajnoczi -RH-MergeRequest: 450: file-posix: probe discard alignment on Linux block devices -RH-Jira: RHEL-87734 +RH-MergeRequest: 355: file-posix: probe discard alignment on Linux block devices +RH-Jira: RHEL-86032 RH-Acked-by: Kevin Wolf RH-Acked-by: Eric Blake -RH-Commit: [3/3] 89a47a6fceb2222593cff4d8c85376c210dc7cc2 +RH-Commit: [3/3] b8139a4c5b19efff1f15c314447a6abb89db0ae7 (stefanha/centos-stream-qemu-kvm) Block devices that don't support discard have a discard_granularity of 0. Currently, this results in a division by zero when we try to make diff --git a/SOURCES/kvm-file-posix-Probe-paths-and-retry-SG_IO-on-potential-.patch b/SOURCES/kvm-file-posix-Probe-paths-and-retry-SG_IO-on-potential-.patch index 3a6c478..bd716a1 100644 --- a/SOURCES/kvm-file-posix-Probe-paths-and-retry-SG_IO-on-potential-.patch +++ b/SOURCES/kvm-file-posix-Probe-paths-and-retry-SG_IO-on-potential-.patch @@ -1,15 +1,15 @@ -From 50b1a3ec7cfea5a92069e043e8f77a2595480d20 Mon Sep 17 00:00:00 2001 +From 95c651ba1177bd88dbd9b52fe2ec8fedadcdb5c8 Mon Sep 17 00:00:00 2001 From: Kevin Wolf Date: Thu, 22 May 2025 15:08:03 +0200 -Subject: [PATCH 2/2] file-posix: Probe paths and retry SG_IO on potential path +Subject: [PATCH 3/3] file-posix: Probe paths and retry SG_IO on potential path errors RH-Author: Kevin Wolf -RH-MergeRequest: 456: file-posix: Fix multipath failover with SCSI passthrough [9.6.z] -RH-Jira: RHEL-95407 +RH-MergeRequest: 372: file-posix: Fix multipath failover with SCSI passthrough [9.7] +RH-Jira: RHEL-95408 RH-Acked-by: Hanna Czenczek RH-Acked-by: Stefan Hajnoczi -RH-Commit: [2/2] 2f4aed9004889e9df25e30c2273f2524a67c666c (kmwolf/rhel-qemu-kvm) +RH-Commit: [2/2] 4312e9ec609e511afdfb6634e1d2370032d41543 (kmwolf/centos-qemu-kvm) When scsi-block is used on a host multipath device, it runs into the problem that the kernel dm-mpath doesn't know anything about SCSI or diff --git a/SOURCES/kvm-file-posix-gluster-Handle-zero-block-status-hint-bet.patch b/SOURCES/kvm-file-posix-gluster-Handle-zero-block-status-hint-bet.patch new file mode 100644 index 0000000..4b0d130 --- /dev/null +++ b/SOURCES/kvm-file-posix-gluster-Handle-zero-block-status-hint-bet.patch @@ -0,0 +1,64 @@ +From 39e0c370357a414abacd64fb6a172e7b25eb4d82 Mon Sep 17 00:00:00 2001 +From: Eric Blake +Date: Fri, 9 May 2025 15:40:19 -0500 +Subject: [PATCH 04/16] file-posix, gluster: Handle zero block status hint + better + +RH-Author: Eric Blake +RH-MergeRequest: 365: blockdev-mirror: More efficient handling of sparse mirrors +RH-Jira: RHEL-82906 RHEL-83015 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Jon Maloy +RH-Commit: [2/14] 1f7b47ce5f5fb321aee41a16accf5bce3d1bfe95 (ebblake/centos-qemu-kvm) + +Although the previous patch to change 'bool want_zero' into a bitmask +made no semantic change, it is now time to differentiate. When the +caller specifically wants to know what parts of the file read as zero, +we need to use lseek and actually reporting holes, rather than +short-circuiting and advertising full allocation. + +This change will be utilized in later patches to let mirroring +optimize for the case when the destination already reads as zeroes. + +Signed-off-by: Eric Blake +Reviewed-by: Stefan Hajnoczi +Message-ID: <20250509204341.3553601-17-eblake@redhat.com> +(cherry picked from commit a6a0a7fb0e327d17594c971b4a39de14e025b415) +Jira: https://issues.redhat.com/browse/RHEL-82906 +Jira: https://issues.redhat.com/browse/RHEL-83015 +Signed-off-by: Eric Blake +--- + block/file-posix.c | 3 ++- + block/gluster.c | 2 +- + 2 files changed, 3 insertions(+), 2 deletions(-) + +diff --git a/block/file-posix.c b/block/file-posix.c +index 9ca55620ca..ce5da2b4c2 100644 +--- a/block/file-posix.c ++++ b/block/file-posix.c +@@ -3293,7 +3293,8 @@ static int coroutine_fn raw_co_block_status(BlockDriverState *bs, + return ret; + } + +- if (mode != BDRV_WANT_PRECISE) { ++ if (!(mode & BDRV_WANT_ZERO)) { ++ /* There is no backing file - all bytes are allocated in this file. */ + *pnum = bytes; + *map = offset; + *file = bs; +diff --git a/block/gluster.c b/block/gluster.c +index ae5c45666b..175c70164c 100644 +--- a/block/gluster.c ++++ b/block/gluster.c +@@ -1483,7 +1483,7 @@ static int coroutine_fn qemu_gluster_co_block_status(BlockDriverState *bs, + return ret; + } + +- if (mode != BDRV_WANT_PRECISE) { ++ if (!(mode & BDRV_WANT_ZERO)) { + *pnum = bytes; + *map = offset; + *file = bs; +-- +2.48.1 + diff --git a/SOURCES/kvm-file-posix-probe-discard-alignment-on-Linux-block-de.patch b/SOURCES/kvm-file-posix-probe-discard-alignment-on-Linux-block-de.patch index bb8ba25..7d60479 100644 --- a/SOURCES/kvm-file-posix-probe-discard-alignment-on-Linux-block-de.patch +++ b/SOURCES/kvm-file-posix-probe-discard-alignment-on-Linux-block-de.patch @@ -1,15 +1,15 @@ -From 821ddc7a25a4463f3d9f32c7ae2190f56d597c1b Mon Sep 17 00:00:00 2001 +From 29ae77d77cabc3582267cb8a7c4fe10d279a21e6 Mon Sep 17 00:00:00 2001 From: Stefan Hajnoczi Date: Thu, 17 Apr 2025 11:05:27 -0400 -Subject: [PATCH 1/3] file-posix: probe discard alignment on Linux block +Subject: [PATCH 3/5] file-posix: probe discard alignment on Linux block devices RH-Author: Stefan Hajnoczi -RH-MergeRequest: 450: file-posix: probe discard alignment on Linux block devices -RH-Jira: RHEL-87734 +RH-MergeRequest: 355: file-posix: probe discard alignment on Linux block devices +RH-Jira: RHEL-86032 RH-Acked-by: Kevin Wolf RH-Acked-by: Eric Blake -RH-Commit: [1/3] 89037a51ad7f2d9344e506679c1300b7d4f805ab +RH-Commit: [1/3] bb3c17b0da6edeb209874e97d4e2c3b1762a1749 (stefanha/centos-stream-qemu-kvm) Populate the pdiscard_alignment block limit so the block layer is able align discard requests correctly. diff --git a/SOURCES/kvm-headers-Add-definitions-from-UEFI-spec-for-volumes-r.patch b/SOURCES/kvm-headers-Add-definitions-from-UEFI-spec-for-volumes-r.patch new file mode 100644 index 0000000..e983ea3 --- /dev/null +++ b/SOURCES/kvm-headers-Add-definitions-from-UEFI-spec-for-volumes-r.patch @@ -0,0 +1,233 @@ +From 43245dc5a297d6c4097a0191af4f818e416a3f45 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 051/115] headers: Add definitions from UEFI spec for volumes, + resources, etc... + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [51/115] c44f8b832aa268a5b2d3e98a8ce6bfbda7e1c684 (bonzini/rhel-qemu-kvm) + +Add UEFI definitions for literals, enums, structs, GUIDs, etc... that +will be used by TDX to build the UEFI Hand-Off Block (HOB) that is passed +to the Trusted Domain Virtual Firmware (TDVF). + +All values come from the UEFI specification [1], PI spec [2] and TDVF +design guide[3]. + +[1] UEFI Specification v2.1.0 https://uefi.org/sites/default/files/resources/UEFI_Spec_2_10_Aug29.pdf +[2] UEFI PI spec v1.8 https://uefi.org/sites/default/files/resources/UEFI_PI_Spec_1_8_March3.pdf +[3] https://software.intel.com/content/dam/develop/external/us/en/documents/tdx-virtual-firmware-design-guide-rev-1.pdf + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-23-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 88aa6576e4ab40b538f543852128cb17fce37f87) +Signed-off-by: Paolo Bonzini +--- + include/standard-headers/uefi/uefi.h | 187 +++++++++++++++++++++++++++ + 1 file changed, 187 insertions(+) + create mode 100644 include/standard-headers/uefi/uefi.h + +diff --git a/include/standard-headers/uefi/uefi.h b/include/standard-headers/uefi/uefi.h +new file mode 100644 +index 0000000000..5256349ec0 +--- /dev/null ++++ b/include/standard-headers/uefi/uefi.h +@@ -0,0 +1,187 @@ ++/* ++ * Copyright (C) 2025 Intel Corporation ++ * ++ * Author: Isaku Yamahata ++ * ++ * Xiaoyao Li ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#ifndef HW_I386_UEFI_H ++#define HW_I386_UEFI_H ++ ++/***************************************************************************/ ++/* ++ * basic EFI definitions ++ * supplemented with UEFI Specification Version 2.8 (Errata A) ++ * released February 2020 ++ */ ++/* UEFI integer is little endian */ ++ ++typedef struct { ++ uint32_t Data1; ++ uint16_t Data2; ++ uint16_t Data3; ++ uint8_t Data4[8]; ++} EFI_GUID; ++ ++typedef enum { ++ EfiReservedMemoryType, ++ EfiLoaderCode, ++ EfiLoaderData, ++ EfiBootServicesCode, ++ EfiBootServicesData, ++ EfiRuntimeServicesCode, ++ EfiRuntimeServicesData, ++ EfiConventionalMemory, ++ EfiUnusableMemory, ++ EfiACPIReclaimMemory, ++ EfiACPIMemoryNVS, ++ EfiMemoryMappedIO, ++ EfiMemoryMappedIOPortSpace, ++ EfiPalCode, ++ EfiPersistentMemory, ++ EfiUnacceptedMemoryType, ++ EfiMaxMemoryType ++} EFI_MEMORY_TYPE; ++ ++#define EFI_HOB_HANDOFF_TABLE_VERSION 0x0009 ++ ++#define EFI_HOB_TYPE_HANDOFF 0x0001 ++#define EFI_HOB_TYPE_MEMORY_ALLOCATION 0x0002 ++#define EFI_HOB_TYPE_RESOURCE_DESCRIPTOR 0x0003 ++#define EFI_HOB_TYPE_GUID_EXTENSION 0x0004 ++#define EFI_HOB_TYPE_FV 0x0005 ++#define EFI_HOB_TYPE_CPU 0x0006 ++#define EFI_HOB_TYPE_MEMORY_POOL 0x0007 ++#define EFI_HOB_TYPE_FV2 0x0009 ++#define EFI_HOB_TYPE_LOAD_PEIM_UNUSED 0x000A ++#define EFI_HOB_TYPE_UEFI_CAPSULE 0x000B ++#define EFI_HOB_TYPE_FV3 0x000C ++#define EFI_HOB_TYPE_UNUSED 0xFFFE ++#define EFI_HOB_TYPE_END_OF_HOB_LIST 0xFFFF ++ ++typedef struct { ++ uint16_t HobType; ++ uint16_t HobLength; ++ uint32_t Reserved; ++} EFI_HOB_GENERIC_HEADER; ++ ++typedef uint64_t EFI_PHYSICAL_ADDRESS; ++typedef uint32_t EFI_BOOT_MODE; ++ ++typedef struct { ++ EFI_HOB_GENERIC_HEADER Header; ++ uint32_t Version; ++ EFI_BOOT_MODE BootMode; ++ EFI_PHYSICAL_ADDRESS EfiMemoryTop; ++ EFI_PHYSICAL_ADDRESS EfiMemoryBottom; ++ EFI_PHYSICAL_ADDRESS EfiFreeMemoryTop; ++ EFI_PHYSICAL_ADDRESS EfiFreeMemoryBottom; ++ EFI_PHYSICAL_ADDRESS EfiEndOfHobList; ++} EFI_HOB_HANDOFF_INFO_TABLE; ++ ++#define EFI_RESOURCE_SYSTEM_MEMORY 0x00000000 ++#define EFI_RESOURCE_MEMORY_MAPPED_IO 0x00000001 ++#define EFI_RESOURCE_IO 0x00000002 ++#define EFI_RESOURCE_FIRMWARE_DEVICE 0x00000003 ++#define EFI_RESOURCE_MEMORY_MAPPED_IO_PORT 0x00000004 ++#define EFI_RESOURCE_MEMORY_RESERVED 0x00000005 ++#define EFI_RESOURCE_IO_RESERVED 0x00000006 ++#define EFI_RESOURCE_MEMORY_UNACCEPTED 0x00000007 ++#define EFI_RESOURCE_MAX_MEMORY_TYPE 0x00000008 ++ ++#define EFI_RESOURCE_ATTRIBUTE_PRESENT 0x00000001 ++#define EFI_RESOURCE_ATTRIBUTE_INITIALIZED 0x00000002 ++#define EFI_RESOURCE_ATTRIBUTE_TESTED 0x00000004 ++#define EFI_RESOURCE_ATTRIBUTE_SINGLE_BIT_ECC 0x00000008 ++#define EFI_RESOURCE_ATTRIBUTE_MULTIPLE_BIT_ECC 0x00000010 ++#define EFI_RESOURCE_ATTRIBUTE_ECC_RESERVED_1 0x00000020 ++#define EFI_RESOURCE_ATTRIBUTE_ECC_RESERVED_2 0x00000040 ++#define EFI_RESOURCE_ATTRIBUTE_READ_PROTECTED 0x00000080 ++#define EFI_RESOURCE_ATTRIBUTE_WRITE_PROTECTED 0x00000100 ++#define EFI_RESOURCE_ATTRIBUTE_EXECUTION_PROTECTED 0x00000200 ++#define EFI_RESOURCE_ATTRIBUTE_UNCACHEABLE 0x00000400 ++#define EFI_RESOURCE_ATTRIBUTE_WRITE_COMBINEABLE 0x00000800 ++#define EFI_RESOURCE_ATTRIBUTE_WRITE_THROUGH_CACHEABLE 0x00001000 ++#define EFI_RESOURCE_ATTRIBUTE_WRITE_BACK_CACHEABLE 0x00002000 ++#define EFI_RESOURCE_ATTRIBUTE_16_BIT_IO 0x00004000 ++#define EFI_RESOURCE_ATTRIBUTE_32_BIT_IO 0x00008000 ++#define EFI_RESOURCE_ATTRIBUTE_64_BIT_IO 0x00010000 ++#define EFI_RESOURCE_ATTRIBUTE_UNCACHED_EXPORTED 0x00020000 ++#define EFI_RESOURCE_ATTRIBUTE_READ_ONLY_PROTECTED 0x00040000 ++#define EFI_RESOURCE_ATTRIBUTE_READ_ONLY_PROTECTABLE 0x00080000 ++#define EFI_RESOURCE_ATTRIBUTE_READ_PROTECTABLE 0x00100000 ++#define EFI_RESOURCE_ATTRIBUTE_WRITE_PROTECTABLE 0x00200000 ++#define EFI_RESOURCE_ATTRIBUTE_EXECUTION_PROTECTABLE 0x00400000 ++#define EFI_RESOURCE_ATTRIBUTE_PERSISTENT 0x00800000 ++#define EFI_RESOURCE_ATTRIBUTE_PERSISTABLE 0x01000000 ++#define EFI_RESOURCE_ATTRIBUTE_MORE_RELIABLE 0x02000000 ++ ++typedef uint32_t EFI_RESOURCE_TYPE; ++typedef uint32_t EFI_RESOURCE_ATTRIBUTE_TYPE; ++ ++typedef struct { ++ EFI_HOB_GENERIC_HEADER Header; ++ EFI_GUID Owner; ++ EFI_RESOURCE_TYPE ResourceType; ++ EFI_RESOURCE_ATTRIBUTE_TYPE ResourceAttribute; ++ EFI_PHYSICAL_ADDRESS PhysicalStart; ++ uint64_t ResourceLength; ++} EFI_HOB_RESOURCE_DESCRIPTOR; ++ ++typedef struct { ++ EFI_HOB_GENERIC_HEADER Header; ++ EFI_GUID Name; ++ ++ /* guid specific data follows */ ++} EFI_HOB_GUID_TYPE; ++ ++typedef struct { ++ EFI_HOB_GENERIC_HEADER Header; ++ EFI_PHYSICAL_ADDRESS BaseAddress; ++ uint64_t Length; ++} EFI_HOB_FIRMWARE_VOLUME; ++ ++typedef struct { ++ EFI_HOB_GENERIC_HEADER Header; ++ EFI_PHYSICAL_ADDRESS BaseAddress; ++ uint64_t Length; ++ EFI_GUID FvName; ++ EFI_GUID FileName; ++} EFI_HOB_FIRMWARE_VOLUME2; ++ ++typedef struct { ++ EFI_HOB_GENERIC_HEADER Header; ++ EFI_PHYSICAL_ADDRESS BaseAddress; ++ uint64_t Length; ++ uint32_t AuthenticationStatus; ++ bool ExtractedFv; ++ EFI_GUID FvName; ++ EFI_GUID FileName; ++} EFI_HOB_FIRMWARE_VOLUME3; ++ ++typedef struct { ++ EFI_HOB_GENERIC_HEADER Header; ++ uint8_t SizeOfMemorySpace; ++ uint8_t SizeOfIoSpace; ++ uint8_t Reserved[6]; ++} EFI_HOB_CPU; ++ ++typedef struct { ++ EFI_HOB_GENERIC_HEADER Header; ++} EFI_HOB_MEMORY_POOL; ++ ++typedef struct { ++ EFI_HOB_GENERIC_HEADER Header; ++ ++ EFI_PHYSICAL_ADDRESS BaseAddress; ++ uint64_t Length; ++} EFI_HOB_UEFI_CAPSULE; ++ ++#define EFI_HOB_OWNER_ZERO \ ++ ((EFI_GUID){ 0x00000000, 0x0000, 0x0000, \ ++ { 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00 } }) ++ ++#endif +-- +2.50.1 + diff --git a/SOURCES/kvm-hw-arm-smmuv3-Move-reset-to-exit-phase.patch b/SOURCES/kvm-hw-arm-smmuv3-Move-reset-to-exit-phase.patch new file mode 100644 index 0000000..689a6f5 --- /dev/null +++ b/SOURCES/kvm-hw-arm-smmuv3-Move-reset-to-exit-phase.patch @@ -0,0 +1,123 @@ +From a3dfbe30e930c8d794057e45fffd91a9b0e6afd0 Mon Sep 17 00:00:00 2001 +From: Eric Auger +Date: Tue, 18 Feb 2025 19:25:33 +0100 +Subject: [PATCH 3/9] hw/arm/smmuv3: Move reset to exit phase +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Eric Auger +RH-MergeRequest: 341: Fix vIOMMU reset order +RH-Jira: RHEL-7188 +RH-Acked-by: Peter Xu +RH-Acked-by: Donald Dutile +RH-Acked-by: Cédric Le Goater +RH-Commit: [3/5] e291cb45c32e0fab49b200c275553bbe76b97264 (eauger1/centos-qemu-kvm) + +Currently the iommu may be reset before the devices +it protects. For example this happens with virtio-scsi-pci. +when system_reset is issued from qmp monitor: spurious +"virtio: zero sized buffers are not allowed" warnings can +be observed. This happens because outstanding DMA requests +are still happening while the SMMU gets reset. + +This can also happen with VFIO devices. In that case +spurious DMA translation faults can be observed on host. + +Make sure the SMMU is reset in the 'exit' phase after +all DMA capable devices have been reset during the 'enter' +or 'hold' phase. + +Signed-off-by: Eric Auger +Reviewed-by: Zhenzhong Duan + +Message-Id: <20250218182737.76722-4-eric.auger@redhat.com> +Reviewed-by: Peter Xu +Reviewed-by: Michael S. Tsirkin +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit e39e3f8b8dea856f141e9945167d2b18021ef445) +Signed-off-by: Eric Auger +--- + hw/arm/smmu-common.c | 9 +++++++-- + hw/arm/smmuv3.c | 14 ++++++++++---- + hw/arm/trace-events | 1 + + 3 files changed, 18 insertions(+), 6 deletions(-) + +diff --git a/hw/arm/smmu-common.c b/hw/arm/smmu-common.c +index 3f82728758..f4210fcbc1 100644 +--- a/hw/arm/smmu-common.c ++++ b/hw/arm/smmu-common.c +@@ -924,7 +924,12 @@ static void smmu_base_realize(DeviceState *dev, Error **errp) + } + } + +-static void smmu_base_reset_hold(Object *obj, ResetType type) ++/* ++ * Make sure the IOMMU is reset in 'exit' phase after ++ * all outstanding DMA requests have been quiesced during ++ * the 'enter' or 'hold' reset phases ++ */ ++static void smmu_base_reset_exit(Object *obj, ResetType type) + { + SMMUState *s = ARM_SMMU(obj); + +@@ -950,7 +955,7 @@ static void smmu_base_class_init(ObjectClass *klass, void *data) + device_class_set_props(dc, smmu_dev_properties); + device_class_set_parent_realize(dc, smmu_base_realize, + &sbc->parent_realize); +- rc->phases.hold = smmu_base_reset_hold; ++ rc->phases.exit = smmu_base_reset_exit; + } + + static const TypeInfo smmu_base_info = { +diff --git a/hw/arm/smmuv3.c b/hw/arm/smmuv3.c +index 3971976389..2e90570915 100644 +--- a/hw/arm/smmuv3.c ++++ b/hw/arm/smmuv3.c +@@ -1870,13 +1870,19 @@ static void smmu_init_irq(SMMUv3State *s, SysBusDevice *dev) + } + } + +-static void smmu_reset_hold(Object *obj, ResetType type) ++/* ++ * Make sure the IOMMU is reset in 'exit' phase after ++ * all outstanding DMA requests have been quiesced during ++ * the 'enter' or 'hold' reset phases ++ */ ++static void smmu_reset_exit(Object *obj, ResetType type) + { + SMMUv3State *s = ARM_SMMUV3(obj); + SMMUv3Class *c = ARM_SMMUV3_GET_CLASS(s); + +- if (c->parent_phases.hold) { +- c->parent_phases.hold(obj, type); ++ trace_smmu_reset_exit(); ++ if (c->parent_phases.exit) { ++ c->parent_phases.exit(obj, type); + } + + smmuv3_init_regs(s); +@@ -1999,7 +2005,7 @@ static void smmuv3_class_init(ObjectClass *klass, void *data) + SMMUv3Class *c = ARM_SMMUV3_CLASS(klass); + + dc->vmsd = &vmstate_smmuv3; +- resettable_class_set_parent_phases(rc, NULL, smmu_reset_hold, NULL, ++ resettable_class_set_parent_phases(rc, NULL, NULL, smmu_reset_exit, + &c->parent_phases); + device_class_set_parent_realize(dc, smmu_realize, + &c->parent_realize); +diff --git a/hw/arm/trace-events b/hw/arm/trace-events +index be6c8f720b..79ef347e3e 100644 +--- a/hw/arm/trace-events ++++ b/hw/arm/trace-events +@@ -56,6 +56,7 @@ smmuv3_config_cache_inv(uint32_t sid) "Config cache INV for sid=0x%x" + smmuv3_notify_flag_add(const char *iommu) "ADD SMMUNotifier node for iommu mr=%s" + smmuv3_notify_flag_del(const char *iommu) "DEL SMMUNotifier node for iommu mr=%s" + smmuv3_inv_notifiers_iova(const char *name, int asid, int vmid, uint64_t iova, uint8_t tg, uint64_t num_pages, int stage) "iommu mr=%s asid=%d vmid=%d iova=0x%"PRIx64" tg=%d num_pages=0x%"PRIx64" stage=%d" ++smmu_reset_exit(void) "" + + # strongarm.c + strongarm_uart_update_parameters(const char *label, int speed, char parity, int data_bits, int stop_bits) "%s speed=%d parity=%c data=%d stop=%d" +-- +2.48.1 + diff --git a/SOURCES/kvm-hw-audio-ac97-skip-automatic-zero-init-of-large-arra.patch b/SOURCES/kvm-hw-audio-ac97-skip-automatic-zero-init-of-large-arra.patch index a2c5133..ccaf1c4 100644 --- a/SOURCES/kvm-hw-audio-ac97-skip-automatic-zero-init-of-large-arra.patch +++ b/SOURCES/kvm-hw-audio-ac97-skip-automatic-zero-init-of-large-arra.patch @@ -1,16 +1,16 @@ -From d589e33ff7dea4aeb69fe205bea02fb6bd7da618 Mon Sep 17 00:00:00 2001 +From 2018f62f2242d8d4a970d83ebef9b3c2bccf6fda Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:45 +0100 -Subject: [PATCH 08/31] hw/audio/ac97: skip automatic zero-init of large arrays +Subject: [PATCH 33/57] hw/audio/ac97: skip automatic zero-init of large arrays MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [7/30] a2898256b990c1916082a9938740d0fe53da5325 +RH-Commit: [7/30] 4a6b59a9b9122d9f89e99b3e44df19e6d92ed941 (stefanha/centos-stream-qemu-kvm) The 'read_audio' & 'write_audio' methods have a 4k byte array used for copying data between the audio backend and device. Skip the diff --git a/SOURCES/kvm-hw-audio-cs4231a-skip-automatic-zero-init-of-large-a.patch b/SOURCES/kvm-hw-audio-cs4231a-skip-automatic-zero-init-of-large-a.patch index 6ee3f30..95f535f 100644 --- a/SOURCES/kvm-hw-audio-cs4231a-skip-automatic-zero-init-of-large-a.patch +++ b/SOURCES/kvm-hw-audio-cs4231a-skip-automatic-zero-init-of-large-a.patch @@ -1,17 +1,17 @@ -From 63094ee4645705be09f68c547e3f1775ca528951 Mon Sep 17 00:00:00 2001 +From bd32bb22fb324a37b31ed9ac3387524f6f4ea5be Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:46 +0100 -Subject: [PATCH 09/31] hw/audio/cs4231a: skip automatic zero-init of large +Subject: [PATCH 34/57] hw/audio/cs4231a: skip automatic zero-init of large arrays MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [8/30] c6117831ac2ec8e0a700207d5f101eddb67a24a4 +RH-Commit: [8/30] 6c454bcc2927e49896c62718287fb9e4b37b3bb9 (stefanha/centos-stream-qemu-kvm) The 'cs_write_audio' method has a pair of byte arrays, one 4k in size and one 8k, which are used in converting audio samples. Skip the diff --git a/SOURCES/kvm-hw-audio-es1370-skip-automatic-zero-init-of-large-ar.patch b/SOURCES/kvm-hw-audio-es1370-skip-automatic-zero-init-of-large-ar.patch index 91492d2..76a5d89 100644 --- a/SOURCES/kvm-hw-audio-es1370-skip-automatic-zero-init-of-large-ar.patch +++ b/SOURCES/kvm-hw-audio-es1370-skip-automatic-zero-init-of-large-ar.patch @@ -1,17 +1,17 @@ -From 8b487658db35f371e4a527ed18a2ae63b4048f83 Mon Sep 17 00:00:00 2001 +From cb12ddc6ed836091aa7724e2f77ab79cd9089cad Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:47 +0100 -Subject: [PATCH 10/31] hw/audio/es1370: skip automatic zero-init of large +Subject: [PATCH 35/57] hw/audio/es1370: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [9/30] d6be11e78b1af782f12d7a25fc6c297370aafd4a +RH-Commit: [9/30] b992e4247d8d31dc09f9dc7671e7a532558174ec (stefanha/centos-stream-qemu-kvm) The 'es1370_transfer_audio' method has a 4k byte array used for copying data between the audio backend and device. Skip the automatic diff --git a/SOURCES/kvm-hw-audio-gus-skip-automatic-zero-init-of-large-array.patch b/SOURCES/kvm-hw-audio-gus-skip-automatic-zero-init-of-large-array.patch index 2408198..2ce4fa8 100644 --- a/SOURCES/kvm-hw-audio-gus-skip-automatic-zero-init-of-large-array.patch +++ b/SOURCES/kvm-hw-audio-gus-skip-automatic-zero-init-of-large-array.patch @@ -1,16 +1,16 @@ -From 6aac6e3888bd249856dc5bc91a40d9b4eb60f732 Mon Sep 17 00:00:00 2001 +From 9ad7091d82fd0577488f27ab54bb7851fe957020 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:48 +0100 -Subject: [PATCH 11/31] hw/audio/gus: skip automatic zero-init of large array +Subject: [PATCH 36/57] hw/audio/gus: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [10/30] 1af0f37dbbd3f3703988dfd6548e2ba018f05ee7 +RH-Commit: [10/30] 366953d0417ac31e3060fdc327fe8dade3375bf0 (stefanha/centos-stream-qemu-kvm) The 'GUS_read_DMA' method has a 4k byte array used for copying data between the audio backend and device. Skip the automatic diff --git a/SOURCES/kvm-hw-audio-marvell_88w8618-skip-automatic-zero-init-of.patch b/SOURCES/kvm-hw-audio-marvell_88w8618-skip-automatic-zero-init-of.patch index f6e0f3f..3608901 100644 --- a/SOURCES/kvm-hw-audio-marvell_88w8618-skip-automatic-zero-init-of.patch +++ b/SOURCES/kvm-hw-audio-marvell_88w8618-skip-automatic-zero-init-of.patch @@ -1,17 +1,17 @@ -From ad0ae4edded2db9e7fea4c82cb22d47798b34528 Mon Sep 17 00:00:00 2001 +From 5cf61823cbe80b1ace2f5bdb9cc1971956425b98 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:49 +0100 -Subject: [PATCH 12/31] hw/audio/marvell_88w8618: skip automatic zero-init of +Subject: [PATCH 37/57] hw/audio/marvell_88w8618: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [11/30] 22754626ac42807e6306becab38fbaf564a84660 +RH-Commit: [11/30] e09cdb76430552081168873dadfef1b5c8f74327 (stefanha/centos-stream-qemu-kvm) The 'mv88w8618_audio_callback' method has a 4k byte array used for copying data between the audio backend and device. Skip the automatic diff --git a/SOURCES/kvm-hw-audio-sb16-skip-automatic-zero-init-of-large-arra.patch b/SOURCES/kvm-hw-audio-sb16-skip-automatic-zero-init-of-large-arra.patch index 8a61b43..7e531d6 100644 --- a/SOURCES/kvm-hw-audio-sb16-skip-automatic-zero-init-of-large-arra.patch +++ b/SOURCES/kvm-hw-audio-sb16-skip-automatic-zero-init-of-large-arra.patch @@ -1,16 +1,16 @@ -From 06915675e69c2ff0d1a686dcea200429f54fee74 Mon Sep 17 00:00:00 2001 +From 0b4d59d75edd49ef99f0a82fbcbe360c5b48e4f8 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:50 +0100 -Subject: [PATCH 13/31] hw/audio/sb16: skip automatic zero-init of large array +Subject: [PATCH 38/57] hw/audio/sb16: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [12/30] 93e9b488f1ec99b425cc149c3aba8f824d000c73 +RH-Commit: [12/30] 6475d67546bf04745636b317e965bcd89b6fb2d2 (stefanha/centos-stream-qemu-kvm) The 'write_audio' method has a 4k byte array used for copying data between the audio backend and device. Skip the automatic zero-init diff --git a/SOURCES/kvm-hw-audio-via-ac97-skip-automatic-zero-init-of-large-.patch b/SOURCES/kvm-hw-audio-via-ac97-skip-automatic-zero-init-of-large-.patch index dcf7352..c52f0c1 100644 --- a/SOURCES/kvm-hw-audio-via-ac97-skip-automatic-zero-init-of-large-.patch +++ b/SOURCES/kvm-hw-audio-via-ac97-skip-automatic-zero-init-of-large-.patch @@ -1,17 +1,17 @@ -From 180cb8f07e5a7a1c4bfe01709b27cbf8d080f1a8 Mon Sep 17 00:00:00 2001 +From 35332282ef8bd06f59206266006eff222ffe6bec Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:51 +0100 -Subject: [PATCH 14/31] hw/audio/via-ac97: skip automatic zero-init of large +Subject: [PATCH 39/57] hw/audio/via-ac97: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [13/30] d27c1e2fbd89df34bde1248d1b053eca40625838 +RH-Commit: [13/30] 6391a04b29fcbb8bcdbce2c6b786758fc34f0d71 (stefanha/centos-stream-qemu-kvm) The 'out_cb' method has a 4k byte array used for copying data between the audio backend and device. Skip the automatic zero-init diff --git a/SOURCES/kvm-hw-char-sclpconsole-lm-skip-automatic-zero-init-of-l.patch b/SOURCES/kvm-hw-char-sclpconsole-lm-skip-automatic-zero-init-of-l.patch index 9dab844..98d11f0 100644 --- a/SOURCES/kvm-hw-char-sclpconsole-lm-skip-automatic-zero-init-of-l.patch +++ b/SOURCES/kvm-hw-char-sclpconsole-lm-skip-automatic-zero-init-of-l.patch @@ -1,17 +1,17 @@ -From 03b3f8f230e30399e952c2c5eafde3efa7c015b3 Mon Sep 17 00:00:00 2001 +From b0c16a93460c2dfe834a9f439d25dc833dfb7427 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:52 +0100 -Subject: [PATCH 15/31] hw/char/sclpconsole-lm: skip automatic zero-init of +Subject: [PATCH 40/57] hw/char/sclpconsole-lm: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [14/30] fca3dbfc00277d1c35ecdb65a56d26c95c8cb6bb +RH-Commit: [14/30] 1491e0147a799ec523fa67fd49649722a07299e7 (stefanha/centos-stream-qemu-kvm) The 'process_mdb' method has a 4k byte array used for copying data between the guest and the chardev backend. Skip the automatic zero-init diff --git a/SOURCES/kvm-hw-display-vmware_vga-skip-automatic-zero-init-of-la.patch b/SOURCES/kvm-hw-display-vmware_vga-skip-automatic-zero-init-of-la.patch index e28e59a..607fd50 100644 --- a/SOURCES/kvm-hw-display-vmware_vga-skip-automatic-zero-init-of-la.patch +++ b/SOURCES/kvm-hw-display-vmware_vga-skip-automatic-zero-init-of-la.patch @@ -1,17 +1,17 @@ -From 4c88a2da13473491d30328f24239fe305ff037ef Mon Sep 17 00:00:00 2001 +From 7b5624efccf55184278c6f4924efc2141df460f0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:54 +0100 -Subject: [PATCH 17/31] hw/display/vmware_vga: skip automatic zero-init of +Subject: [PATCH 42/57] hw/display/vmware_vga: skip automatic zero-init of large struct MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [16/30] 7b1073ec54071782a9ad32e17293b141bee639c4 +RH-Commit: [16/30] 4aaf459d4356bf28164be742889b9a78d3656703 (stefanha/centos-stream-qemu-kvm) The 'vmsvga_fifo_run' method has a struct which is a little over 20k in size, used for holding image data for cursor changes. Skip the diff --git a/SOURCES/kvm-hw-dma-xlnx_csu_dma-skip-automatic-zero-init-of-larg.patch b/SOURCES/kvm-hw-dma-xlnx_csu_dma-skip-automatic-zero-init-of-larg.patch index 9383da2..d38d141 100644 --- a/SOURCES/kvm-hw-dma-xlnx_csu_dma-skip-automatic-zero-init-of-larg.patch +++ b/SOURCES/kvm-hw-dma-xlnx_csu_dma-skip-automatic-zero-init-of-larg.patch @@ -1,17 +1,17 @@ -From 7208c85b822d45410d21ec42acdc872550a57262 Mon Sep 17 00:00:00 2001 +From cd3500c9e248dbefb36273046e6eee44ee0d5cbe Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:53 +0100 -Subject: [PATCH 16/31] hw/dma/xlnx_csu_dma: skip automatic zero-init of large +Subject: [PATCH 41/57] hw/dma/xlnx_csu_dma: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [15/30] 8ffc738b76ad72de3d0925c5e44408b0f712fcef +RH-Commit: [15/30] 063c88269c7d3bf07ae05aaf2d3d154e2016db81 (stefanha/centos-stream-qemu-kvm) The 'xlnx_csu_dma_src_notify' method has a 4k byte array used for copying DMA data. Skip the automatic zero-init of this array to diff --git a/SOURCES/kvm-hw-hyperv-syndbg-skip-automatic-zero-init-of-large-a.patch b/SOURCES/kvm-hw-hyperv-syndbg-skip-automatic-zero-init-of-large-a.patch index 64425b4..26d816e 100644 --- a/SOURCES/kvm-hw-hyperv-syndbg-skip-automatic-zero-init-of-large-a.patch +++ b/SOURCES/kvm-hw-hyperv-syndbg-skip-automatic-zero-init-of-large-a.patch @@ -1,17 +1,17 @@ -From c7b6fe3f924396dd49bdf13485696a536aa34fb0 Mon Sep 17 00:00:00 2001 +From a4673aab85958c60867b12c65cc3483d734bb6e0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:55 +0100 -Subject: [PATCH 18/31] hw/hyperv/syndbg: skip automatic zero-init of large +Subject: [PATCH 43/57] hw/hyperv/syndbg: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [17/30] 4202f998aff3fd784508ab6e52658cecd197924b +RH-Commit: [17/30] 5f71779c431128601baf46115fe65178532a3836 (stefanha/centos-stream-qemu-kvm) The 'handle_recv_msg' method has a 4k byte array used for copying data between the network socket and guest memory. Skip the automatic diff --git a/SOURCES/kvm-hw-i386-Fix-machine-type-compatibility.patch b/SOURCES/kvm-hw-i386-Fix-machine-type-compatibility.patch index 0ab0cda..430ba65 100644 --- a/SOURCES/kvm-hw-i386-Fix-machine-type-compatibility.patch +++ b/SOURCES/kvm-hw-i386-Fix-machine-type-compatibility.patch @@ -1,14 +1,14 @@ -From 9bd4a89d3e1410b3a5994ab2b33f4332a4246955 Mon Sep 17 00:00:00 2001 +From 2bb5dff02fb393530a12f4f00219cd2f90cd442a Mon Sep 17 00:00:00 2001 From: Sebastian Ott Date: Thu, 15 May 2025 18:45:51 +0200 -Subject: [PATCH] hw/i386: Fix machine type compatibility +Subject: [PATCH 3/5] hw/i386: Fix machine type compatibility RH-Author: Sebastian Ott -RH-MergeRequest: 452: hw/i386: Fix machine type compatibility -RH-Jira: RHEL-92077 +RH-MergeRequest: 364: hw/i386: Fix machine type compatibility +RH-Jira: RHEL-91307 RH-Acked-by: Cornelia Huck RH-Acked-by: Jon Maloy -RH-Commit: [1/1] d594f142e8ce616b6fd1accd6950ab5cebc34984 +RH-Commit: [1/1] 44ddbcb3af119c65e99018d7ed90887f3948907e (seott1/cos-qemu-kvm) Upstream Status: RHEL only @@ -24,7 +24,7 @@ Signed-off-by: Sebastian Ott 4 files changed, 15 insertions(+) diff --git a/hw/i386/pc.c b/hw/i386/pc.c -index fa0e42d072..d8f1b2d899 100644 +index fa9f16cbaf..5237538640 100644 --- a/hw/i386/pc.c +++ b/hw/i386/pc.c @@ -298,6 +298,14 @@ GlobalProperty pc_rhel_compat[] = { @@ -43,7 +43,7 @@ index fa0e42d072..d8f1b2d899 100644 /* pc_rhel_9_5_compat from pc_compat_pc_9_0 (backported from 9.1) */ { TYPE_X86_CPU, "guest-phys-bits", "0" }, diff --git a/hw/i386/pc_piix.c b/hw/i386/pc_piix.c -index 656abb5d39..80d366bf17 100644 +index 10764bf596..0687317db5 100644 --- a/hw/i386/pc_piix.c +++ b/hw/i386/pc_piix.c @@ -885,6 +885,8 @@ static void pc_i440fx_rhel_machine_7_6_0_options(MachineClass *m) @@ -56,10 +56,10 @@ index 656abb5d39..80d366bf17 100644 pc_rhel_9_5_compat_len); compat_props_add(m->compat_props, hw_compat_rhel_9_5, diff --git a/hw/i386/pc_q35.c b/hw/i386/pc_q35.c -index 578f63524f..e3653b44cd 100644 +index 5bf08be0fb..871c760aea 100644 --- a/hw/i386/pc_q35.c +++ b/hw/i386/pc_q35.c -@@ -701,6 +701,8 @@ static void pc_q35_rhel_machine_9_4_0_options(MachineClass *m) +@@ -704,6 +704,8 @@ static void pc_q35_rhel_machine_9_4_0_options(MachineClass *m) compat_props_add(m->compat_props, hw_compat_rhel_9_6, hw_compat_rhel_9_6_len); diff --git a/SOURCES/kvm-hw-i386-amd_iommu-Allow-migration-when-explicitly-cr.patch b/SOURCES/kvm-hw-i386-amd_iommu-Allow-migration-when-explicitly-cr.patch new file mode 100644 index 0000000..d68826a --- /dev/null +++ b/SOURCES/kvm-hw-i386-amd_iommu-Allow-migration-when-explicitly-cr.patch @@ -0,0 +1,117 @@ +From f1ff9d3b379697a2d4627e9529067195841d86a8 Mon Sep 17 00:00:00 2001 +From: Suravee Suthikulpanit +Date: Sun, 4 May 2025 17:04:05 +0000 +Subject: [PATCH 25/57] hw/i386/amd_iommu: Allow migration when explicitly + create the AMDVI-PCI device + +RH-Author: John Allen +RH-MergeRequest: 380: Add ability to manually specify the AMDVI-PCI device +RH-Jira: RHEL-70925 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [2/3] a42b88116e608a79b6fae13ebe3709874f2a853f (johnalle/qemu-kvm-fork) + +Add migration support for AMD IOMMU model by saving necessary AMDVIState +parameters for MMIO registers, device table, command buffer, and event +buffers. + +Also change devtab_len type from size_t to uint64_t to avoid 32-bit build +issue. + +Signed-off-by: Suravee Suthikulpanit +Message-Id: <20250504170405.12623-3-suravee.suthikulpanit@amd.com> +Reviewed-by: Michael S. Tsirkin +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit 28931c2e1591deb4bfaaf744fdc8813e96c230f1) + +JIRA: https://issues.redhat.com/browse/RHEL-70925 + +Signed-off-by: John Allen +--- + hw/i386/amd_iommu.c | 48 +++++++++++++++++++++++++++++++++++++++++++++ + hw/i386/amd_iommu.h | 2 +- + 2 files changed, 49 insertions(+), 1 deletion(-) + +diff --git a/hw/i386/amd_iommu.c b/hw/i386/amd_iommu.c +index 6a5e76cfef..a34e0c5f59 100644 +--- a/hw/i386/amd_iommu.c ++++ b/hw/i386/amd_iommu.c +@@ -1611,8 +1611,55 @@ static void amdvi_sysbus_reset(DeviceState *dev) + amdvi_init(s); + } + ++static const VMStateDescription vmstate_amdvi_sysbus_migratable = { ++ .name = "amd-iommu", ++ .version_id = 1, ++ .minimum_version_id = 1, ++ .priority = MIG_PRI_IOMMU, ++ .fields = (VMStateField[]) { ++ /* Updated in amdvi_handle_control_write() */ ++ VMSTATE_BOOL(enabled, AMDVIState), ++ VMSTATE_BOOL(ga_enabled, AMDVIState), ++ VMSTATE_BOOL(ats_enabled, AMDVIState), ++ VMSTATE_BOOL(cmdbuf_enabled, AMDVIState), ++ VMSTATE_BOOL(completion_wait_intr, AMDVIState), ++ VMSTATE_BOOL(evtlog_enabled, AMDVIState), ++ VMSTATE_BOOL(evtlog_intr, AMDVIState), ++ /* Updated in amdvi_handle_devtab_write() */ ++ VMSTATE_UINT64(devtab, AMDVIState), ++ VMSTATE_UINT64(devtab_len, AMDVIState), ++ /* Updated in amdvi_handle_cmdbase_write() */ ++ VMSTATE_UINT64(cmdbuf, AMDVIState), ++ VMSTATE_UINT64(cmdbuf_len, AMDVIState), ++ /* Updated in amdvi_handle_cmdhead_write() */ ++ VMSTATE_UINT32(cmdbuf_head, AMDVIState), ++ /* Updated in amdvi_handle_cmdtail_write() */ ++ VMSTATE_UINT32(cmdbuf_tail, AMDVIState), ++ /* Updated in amdvi_handle_evtbase_write() */ ++ VMSTATE_UINT64(evtlog, AMDVIState), ++ VMSTATE_UINT32(evtlog_len, AMDVIState), ++ /* Updated in amdvi_handle_evthead_write() */ ++ VMSTATE_UINT32(evtlog_head, AMDVIState), ++ /* Updated in amdvi_handle_evttail_write() */ ++ VMSTATE_UINT32(evtlog_tail, AMDVIState), ++ /* Updated in amdvi_handle_pprbase_write() */ ++ VMSTATE_UINT64(ppr_log, AMDVIState), ++ VMSTATE_UINT32(pprlog_len, AMDVIState), ++ /* Updated in amdvi_handle_pprhead_write() */ ++ VMSTATE_UINT32(pprlog_head, AMDVIState), ++ /* Updated in amdvi_handle_tailhead_write() */ ++ VMSTATE_UINT32(pprlog_tail, AMDVIState), ++ /* MMIO registers */ ++ VMSTATE_UINT8_ARRAY(mmior, AMDVIState, AMDVI_MMIO_SIZE), ++ VMSTATE_UINT8_ARRAY(romask, AMDVIState, AMDVI_MMIO_SIZE), ++ VMSTATE_UINT8_ARRAY(w1cmask, AMDVIState, AMDVI_MMIO_SIZE), ++ VMSTATE_END_OF_LIST() ++ } ++}; ++ + static void amdvi_sysbus_realize(DeviceState *dev, Error **errp) + { ++ DeviceClass *dc = (DeviceClass *) object_get_class(OBJECT(dev)); + AMDVIState *s = AMD_IOMMU_DEVICE(dev); + MachineState *ms = MACHINE(qdev_get_machine()); + PCMachineState *pcms = PC_MACHINE(ms); +@@ -1634,6 +1681,7 @@ static void amdvi_sysbus_realize(DeviceState *dev, Error **errp) + } + + s->pci = AMD_IOMMU_PCI(pdev); ++ dc->vmsd = &vmstate_amdvi_sysbus_migratable; + } else { + s->pci = AMD_IOMMU_PCI(object_new(TYPE_AMD_IOMMU_PCI)); + /* This device should take care of IOMMU PCI properties */ +diff --git a/hw/i386/amd_iommu.h b/hw/i386/amd_iommu.h +index ece71ff0b6..741dd9a910 100644 +--- a/hw/i386/amd_iommu.h ++++ b/hw/i386/amd_iommu.h +@@ -329,7 +329,7 @@ struct AMDVIState { + bool excl_enabled; + + hwaddr devtab; /* base address device table */ +- size_t devtab_len; /* device table length */ ++ uint64_t devtab_len; /* device table length */ + + hwaddr cmdbuf; /* command buffer base address */ + uint64_t cmdbuf_len; /* command buffer length */ +-- +2.39.3 + diff --git a/SOURCES/kvm-hw-i386-amd_iommu-Assign-pci-id-0x1419-for-the-AMD-I.patch b/SOURCES/kvm-hw-i386-amd_iommu-Assign-pci-id-0x1419-for-the-AMD-I.patch new file mode 100644 index 0000000..4542745 --- /dev/null +++ b/SOURCES/kvm-hw-i386-amd_iommu-Assign-pci-id-0x1419-for-the-AMD-I.patch @@ -0,0 +1,57 @@ +From e611119b8b4e0712ab103628051d69ea84538719 Mon Sep 17 00:00:00 2001 +From: Suravee Suthikulpanit +Date: Tue, 25 Mar 2025 02:11:40 +0000 +Subject: [PATCH 23/57] hw/i386/amd_iommu: Assign pci-id 0x1419 for the AMD + IOMMU device +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: John Allen +RH-MergeRequest: 379: hw/i386/amd_iommu: Assign pci-id 0x1419 for the AMD IOMMU device +RH-Jira: RHEL-70926 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/1] 69d847f64543caf328da3e7663e7d2ebe53cd448 (johnalle/qemu-kvm-fork) + +Currently, the QEMU-emulated AMD IOMMU device use PCI vendor id 0x1022 +(AMD) with device id zero (undefined). Eventhough this does not cause any +functional issue for AMD IOMMU driver since it normally uses information +in the ACPI IVRS table to probe and initialize the device per +recommendation in the AMD IOMMU specification, the device id zero causes +the Windows Device Manager utility to show the device as an unknown device. + +Since Windows only recognizes AMD IOMMU device with device id 0x1419 as +listed in the machine.inf file, modify the QEMU AMD IOMMU model to use +the id 0x1419 to avoid the issue. This advertise the IOMMU as the AMD +IOMMU device for Family 15h (Models 10h-1fh). + +Signed-off-by: Suravee Suthikulpanit +Message-Id: <20250325021140.5676-1-suravee.suthikulpanit@amd.com> +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Yan Vugenfirer +Reviewed-by: Michael S. Tsirkin +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit 719255486df2fcbe1b8599786b37f4bb80272f1a) + +JIRA: https://issues.redhat.com/browse/RHEL-70926 + +Signed-off-by: John Allen +--- + hw/i386/amd_iommu.c | 1 + + 1 file changed, 1 insertion(+) + +diff --git a/hw/i386/amd_iommu.c b/hw/i386/amd_iommu.c +index d804656ea8..59e1a01b7c 100644 +--- a/hw/i386/amd_iommu.c ++++ b/hw/i386/amd_iommu.c +@@ -1714,6 +1714,7 @@ static void amdvi_pci_class_init(ObjectClass *klass, void *data) + PCIDeviceClass *k = PCI_DEVICE_CLASS(klass); + + k->vendor_id = PCI_VENDOR_ID_AMD; ++ k->device_id = 0x1419; + k->class_id = 0x0806; + k->realize = amdvi_pci_realize; + +-- +2.39.3 + diff --git a/SOURCES/kvm-hw-i386-amd_iommu-Isolate-AMDVI-PCI-from-amd-iommu-d.patch b/SOURCES/kvm-hw-i386-amd_iommu-Isolate-AMDVI-PCI-from-amd-iommu-d.patch new file mode 100644 index 0000000..6da1d5d --- /dev/null +++ b/SOURCES/kvm-hw-i386-amd_iommu-Isolate-AMDVI-PCI-from-amd-iommu-d.patch @@ -0,0 +1,267 @@ +From 5a697d0f66360acca8216f49c06dc9702231d470 Mon Sep 17 00:00:00 2001 +From: Suravee Suthikulpanit +Date: Sun, 4 May 2025 17:04:04 +0000 +Subject: [PATCH 24/57] hw/i386/amd_iommu: Isolate AMDVI-PCI from amd-iommu + device to allow full control over the PCI device creation +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: John Allen +RH-MergeRequest: 380: Add ability to manually specify the AMDVI-PCI device +RH-Jira: RHEL-70925 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/3] 58254a72ba2d810b57c610462494f76691126521 (johnalle/qemu-kvm-fork) + +Current amd-iommu model internally creates an AMDVI-PCI device. Here is +a snippet from info qtree: + + bus: main-system-bus + type System + dev: amd-iommu, id "" + xtsup = false + pci-id = "" + intremap = "on" + device-iotlb = false + pt = true + ... + dev: q35-pcihost, id "" + MCFG = -1 (0xffffffffffffffff) + pci-hole64-size = 34359738368 (32 GiB) + below-4g-mem-size = 134217728 (128 MiB) + above-4g-mem-size = 0 (0 B) + smm-ranges = true + x-pci-hole64-fix = true + x-config-reg-migration-enabled = true + bypass-iommu = false + bus: pcie.0 + type PCIE + dev: AMDVI-PCI, id "" + addr = 01.0 + romfile = "" + romsize = 4294967295 (0xffffffff) + rombar = -1 (0xffffffffffffffff) + multifunction = false + x-pcie-lnksta-dllla = true + x-pcie-extcap-init = true + failover_pair_id = "" + acpi-index = 0 (0x0) + x-pcie-err-unc-mask = true + x-pcie-ari-nextfn-1 = false + x-max-bounce-buffer-size = 4096 (4 KiB) + x-pcie-ext-tag = true + busnr = 0 (0x0) + class Class 0806, addr 00:01.0, pci id 1022:0000 (sub 1af4:1100) + ... + +This prohibits users from specifying the PCI topology for the amd-iommu device, +which becomes a problem when trying to support VM migration since it does not +guarantee the same enumeration of AMD IOMMU device. + +Therefore, allow the 'AMDVI-PCI' device to optionally be pre-created and +associated with a 'amd-iommu' device via a new 'pci-id' parameter on the +latter. + +For example: + -device AMDVI-PCI,id=iommupci0,bus=pcie.0,addr=0x05 \ + -device amd-iommu,intremap=on,pt=on,xtsup=on,pci-id=iommupci0 \ + +For backward-compatibility, internally create the AMDVI-PCI device if not +specified on the CLI. + +Co-developed-by: Daniel P. Berrangé +Reviewed-by: Daniel P. Berrangé +Signed-off-by: Suravee Suthikulpanit +Message-Id: <20250504170405.12623-2-suravee.suthikulpanit@amd.com> +Reviewed-by: Michael S. Tsirkin +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit f864a3235ea1d1d714b3cde2d9a810ea6344a7b5) + +JIRA: https://issues.redhat.com/browse/RHEL-70925 + +Signed-off-by: John Allen +--- + hw/i386/acpi-build.c | 8 +++---- + hw/i386/amd_iommu.c | 53 ++++++++++++++++++++++++++------------------ + hw/i386/amd_iommu.h | 3 ++- + 3 files changed, 38 insertions(+), 26 deletions(-) + +diff --git a/hw/i386/acpi-build.c b/hw/i386/acpi-build.c +index 032fb1f904..236261f8aa 100644 +--- a/hw/i386/acpi-build.c ++++ b/hw/i386/acpi-build.c +@@ -2392,10 +2392,10 @@ build_amd_iommu(GArray *table_data, BIOSLinker *linker, const char *oem_id, + build_append_int_noprefix(table_data, ivhd_blob->len + 24, 2); + /* DeviceID */ + build_append_int_noprefix(table_data, +- object_property_get_int(OBJECT(&s->pci), "addr", ++ object_property_get_int(OBJECT(s->pci), "addr", + &error_abort), 2); + /* Capability offset */ +- build_append_int_noprefix(table_data, s->pci.capab_offset, 2); ++ build_append_int_noprefix(table_data, s->pci->capab_offset, 2); + /* IOMMU base address */ + build_append_int_noprefix(table_data, s->mr_mmio.addr, 8); + /* PCI Segment Group */ +@@ -2427,10 +2427,10 @@ build_amd_iommu(GArray *table_data, BIOSLinker *linker, const char *oem_id, + build_append_int_noprefix(table_data, ivhd_blob->len + 40, 2); + /* DeviceID */ + build_append_int_noprefix(table_data, +- object_property_get_int(OBJECT(&s->pci), "addr", ++ object_property_get_int(OBJECT(s->pci), "addr", + &error_abort), 2); + /* Capability offset */ +- build_append_int_noprefix(table_data, s->pci.capab_offset, 2); ++ build_append_int_noprefix(table_data, s->pci->capab_offset, 2); + /* IOMMU base address */ + build_append_int_noprefix(table_data, s->mr_mmio.addr, 8); + /* PCI Segment Group */ +diff --git a/hw/i386/amd_iommu.c b/hw/i386/amd_iommu.c +index 59e1a01b7c..6a5e76cfef 100644 +--- a/hw/i386/amd_iommu.c ++++ b/hw/i386/amd_iommu.c +@@ -167,11 +167,11 @@ static void amdvi_generate_msi_interrupt(AMDVIState *s) + { + MSIMessage msg = {}; + MemTxAttrs attrs = { +- .requester_id = pci_requester_id(&s->pci.dev) ++ .requester_id = pci_requester_id(&s->pci->dev) + }; + +- if (msi_enabled(&s->pci.dev)) { +- msg = msi_get_message(&s->pci.dev, 0); ++ if (msi_enabled(&s->pci->dev)) { ++ msg = msi_get_message(&s->pci->dev, 0); + address_space_stl_le(&address_space_memory, msg.address, msg.data, + attrs, NULL); + } +@@ -239,7 +239,7 @@ static void amdvi_page_fault(AMDVIState *s, uint16_t devid, + info |= AMDVI_EVENT_IOPF_I | AMDVI_EVENT_IOPF; + amdvi_encode_event(evt, devid, addr, info); + amdvi_log_event(s, evt); +- pci_word_test_and_set_mask(s->pci.dev.config + PCI_STATUS, ++ pci_word_test_and_set_mask(s->pci->dev.config + PCI_STATUS, + PCI_STATUS_SIG_TARGET_ABORT); + } + /* +@@ -256,7 +256,7 @@ static void amdvi_log_devtab_error(AMDVIState *s, uint16_t devid, + + amdvi_encode_event(evt, devid, devtab, info); + amdvi_log_event(s, evt); +- pci_word_test_and_set_mask(s->pci.dev.config + PCI_STATUS, ++ pci_word_test_and_set_mask(s->pci->dev.config + PCI_STATUS, + PCI_STATUS_SIG_TARGET_ABORT); + } + /* log an event trying to access command buffer +@@ -269,7 +269,7 @@ static void amdvi_log_command_error(AMDVIState *s, hwaddr addr) + + amdvi_encode_event(evt, 0, addr, info); + amdvi_log_event(s, evt); +- pci_word_test_and_set_mask(s->pci.dev.config + PCI_STATUS, ++ pci_word_test_and_set_mask(s->pci->dev.config + PCI_STATUS, + PCI_STATUS_SIG_TARGET_ABORT); + } + /* log an illegal command event +@@ -310,7 +310,7 @@ static void amdvi_log_pagetab_error(AMDVIState *s, uint16_t devid, + info |= AMDVI_EVENT_PAGE_TAB_HW_ERROR; + amdvi_encode_event(evt, devid, addr, info); + amdvi_log_event(s, evt); +- pci_word_test_and_set_mask(s->pci.dev.config + PCI_STATUS, ++ pci_word_test_and_set_mask(s->pci->dev.config + PCI_STATUS, + PCI_STATUS_SIG_TARGET_ABORT); + } + +@@ -1607,7 +1607,7 @@ static void amdvi_sysbus_reset(DeviceState *dev) + { + AMDVIState *s = AMD_IOMMU_DEVICE(dev); + +- msi_reset(&s->pci.dev); ++ msi_reset(&s->pci->dev); + amdvi_init(s); + } + +@@ -1619,14 +1619,32 @@ static void amdvi_sysbus_realize(DeviceState *dev, Error **errp) + X86MachineState *x86ms = X86_MACHINE(ms); + PCIBus *bus = pcms->pcibus; + +- s->iotlb = g_hash_table_new_full(amdvi_uint64_hash, +- amdvi_uint64_equal, g_free, g_free); ++ if (s->pci_id) { ++ PCIDevice *pdev = NULL; ++ int ret = pci_qdev_find_device(s->pci_id, &pdev); + +- /* This device should take care of IOMMU PCI properties */ +- if (!qdev_realize(DEVICE(&s->pci), &bus->qbus, errp)) { +- return; ++ if (ret) { ++ error_report("Cannot find PCI device '%s'", s->pci_id); ++ return; ++ } ++ ++ if (!object_dynamic_cast(OBJECT(pdev), TYPE_AMD_IOMMU_PCI)) { ++ error_report("Device '%s' must be an AMDVI-PCI device type", s->pci_id); ++ return; ++ } ++ ++ s->pci = AMD_IOMMU_PCI(pdev); ++ } else { ++ s->pci = AMD_IOMMU_PCI(object_new(TYPE_AMD_IOMMU_PCI)); ++ /* This device should take care of IOMMU PCI properties */ ++ if (!qdev_realize(DEVICE(s->pci), &bus->qbus, errp)) { ++ return; ++ } + } + ++ s->iotlb = g_hash_table_new_full(amdvi_uint64_hash, ++ amdvi_uint64_equal, g_free, g_free); ++ + /* Pseudo address space under root PCI bus. */ + x86ms->ioapic_as = amdvi_host_dma_iommu(bus, s, AMDVI_IOAPIC_SB_DEVID); + +@@ -1668,6 +1686,7 @@ static void amdvi_sysbus_realize(DeviceState *dev, Error **errp) + + static Property amdvi_properties[] = { + DEFINE_PROP_BOOL("xtsup", AMDVIState, xtsup, false), ++ DEFINE_PROP_STRING("pci-id", AMDVIState, pci_id), + DEFINE_PROP_END_OF_LIST(), + }; + +@@ -1676,13 +1695,6 @@ static const VMStateDescription vmstate_amdvi_sysbus = { + .unmigratable = 1 + }; + +-static void amdvi_sysbus_instance_init(Object *klass) +-{ +- AMDVIState *s = AMD_IOMMU_DEVICE(klass); +- +- object_initialize(&s->pci, sizeof(s->pci), TYPE_AMD_IOMMU_PCI); +-} +- + static void amdvi_sysbus_class_init(ObjectClass *klass, void *data) + { + DeviceClass *dc = DEVICE_CLASS(klass); +@@ -1704,7 +1716,6 @@ static const TypeInfo amdvi_sysbus = { + .name = TYPE_AMD_IOMMU_DEVICE, + .parent = TYPE_X86_IOMMU_DEVICE, + .instance_size = sizeof(AMDVIState), +- .instance_init = amdvi_sysbus_instance_init, + .class_init = amdvi_sysbus_class_init + }; + +diff --git a/hw/i386/amd_iommu.h b/hw/i386/amd_iommu.h +index e0dac4d9a9..ece71ff0b6 100644 +--- a/hw/i386/amd_iommu.h ++++ b/hw/i386/amd_iommu.h +@@ -315,7 +315,8 @@ struct AMDVIPCIState { + + struct AMDVIState { + X86IOMMUState iommu; /* IOMMU bus device */ +- AMDVIPCIState pci; /* IOMMU PCI device */ ++ AMDVIPCIState *pci; /* IOMMU PCI device */ ++ char *pci_id; /* ID of AMDVI-PCI device, if user created */ + + uint32_t version; + +-- +2.39.3 + diff --git a/SOURCES/kvm-hw-i386-intel-iommu-Migrate-to-3-phase-reset.patch b/SOURCES/kvm-hw-i386-intel-iommu-Migrate-to-3-phase-reset.patch new file mode 100644 index 0000000..827c43c --- /dev/null +++ b/SOURCES/kvm-hw-i386-intel-iommu-Migrate-to-3-phase-reset.patch @@ -0,0 +1,96 @@ +From 67b281dc1ccdae05da6c6052c264ecd94723c0b2 Mon Sep 17 00:00:00 2001 +From: Eric Auger +Date: Tue, 18 Feb 2025 19:25:32 +0100 +Subject: [PATCH 2/9] hw/i386/intel-iommu: Migrate to 3-phase reset +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Eric Auger +RH-MergeRequest: 341: Fix vIOMMU reset order +RH-Jira: RHEL-7188 +RH-Acked-by: Peter Xu +RH-Acked-by: Donald Dutile +RH-Acked-by: Cédric Le Goater +RH-Commit: [2/5] 5b9b60b2b796529db10b846881e82e7df4626ec1 (eauger1/centos-qemu-kvm) + +Currently the IOMMU may be reset before the devices +it protects. For example this happens with virtio devices +but also with VFIO devices. In this latter case this +produces spurious translation faults on host. + +Let's use 3-phase reset mechanism and reset the IOMMU on +exit phase after all DMA capable devices have been reset +on 'enter' or 'hold' phase. + +Signed-off-by: Eric Auger +Acked-by: Michael S. Tsirkin +Acked-by: Jason Wang +Zhenzhong Duan + +Message-Id: <20250218182737.76722-3-eric.auger@redhat.com> +Reviewed-by: Peter Xu +Reviewed-by: Michael S. Tsirkin +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit 2aaf48bcf27d8b3da5b30af6c1ced464d3df30f7) +Signed-off-by: Eric Auger + +Conflicts: Code change + hw/i386/intel_iommu.c +We miss e3d0814368d0 ("hw: Use device_class_set_legacy_reset() instead +of opencoding") meaning that instead of removing +device_class_set_legacy_reset(dc, vtd_reset) we remove +dc->reset = vtd_reset; +--- + hw/i386/intel_iommu.c | 12 +++++++++--- + hw/i386/trace-events | 1 + + 2 files changed, 10 insertions(+), 3 deletions(-) + +diff --git a/hw/i386/intel_iommu.c b/hw/i386/intel_iommu.c +index 16d2885fcc..4acefcf5c8 100644 +--- a/hw/i386/intel_iommu.c ++++ b/hw/i386/intel_iommu.c +@@ -4212,10 +4212,11 @@ static void vtd_init(IntelIOMMUState *s) + /* Should not reset address_spaces when reset because devices will still use + * the address space they got at first (won't ask the bus again). + */ +-static void vtd_reset(DeviceState *dev) ++static void vtd_reset_exit(Object *obj, ResetType type) + { +- IntelIOMMUState *s = INTEL_IOMMU_DEVICE(dev); ++ IntelIOMMUState *s = INTEL_IOMMU_DEVICE(obj); + ++ trace_vtd_reset_exit(); + vtd_init(s); + vtd_address_space_refresh_all(s); + } +@@ -4367,8 +4368,13 @@ static void vtd_class_init(ObjectClass *klass, void *data) + { + DeviceClass *dc = DEVICE_CLASS(klass); + X86IOMMUClass *x86_class = X86_IOMMU_DEVICE_CLASS(klass); ++ ResettableClass *rc = RESETTABLE_CLASS(klass); + +- dc->reset = vtd_reset; ++ /* ++ * Use 'exit' reset phase to make sure all DMA requests ++ * have been quiesced during 'enter' or 'hold' phase ++ */ ++ rc->phases.exit = vtd_reset_exit; + dc->vmsd = &vtd_vmstate; + device_class_set_props(dc, vtd_properties); + dc->hotpluggable = false; +diff --git a/hw/i386/trace-events b/hw/i386/trace-events +index 53c02d7ac8..ac9e1a10aa 100644 +--- a/hw/i386/trace-events ++++ b/hw/i386/trace-events +@@ -68,6 +68,7 @@ vtd_frr_new(int index, uint64_t hi, uint64_t lo) "index %d high 0x%"PRIx64" low + vtd_warn_invalid_qi_tail(uint16_t tail) "tail 0x%"PRIx16 + vtd_warn_ir_vector(uint16_t sid, int index, int vec, int target) "sid 0x%"PRIx16" index %d vec %d (should be: %d)" + vtd_warn_ir_trigger(uint16_t sid, int index, int trig, int target) "sid 0x%"PRIx16" index %d trigger %d (should be: %d)" ++vtd_reset_exit(void) "" + + # amd_iommu.c + amdvi_evntlog_fail(uint64_t addr, uint32_t head) "error: fail to write at addr 0x%"PRIx64" + offset 0x%"PRIx32 +-- +2.48.1 + diff --git a/SOURCES/kvm-hw-misc-aspeed_hace-skip-automatic-zero-init-of-larg.patch b/SOURCES/kvm-hw-misc-aspeed_hace-skip-automatic-zero-init-of-larg.patch index 6e3e1f0..a1553f8 100644 --- a/SOURCES/kvm-hw-misc-aspeed_hace-skip-automatic-zero-init-of-larg.patch +++ b/SOURCES/kvm-hw-misc-aspeed_hace-skip-automatic-zero-init-of-larg.patch @@ -1,17 +1,17 @@ -From ee75cefe77904ffc659bdb2df32feef7e01a914e Mon Sep 17 00:00:00 2001 +From 0bfbd2c49c01ee77d3b5a21bf9fe675916cbf0ed Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:56 +0100 -Subject: [PATCH 19/31] hw/misc/aspeed_hace: skip automatic zero-init of large +Subject: [PATCH 44/57] hw/misc/aspeed_hace: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [18/30] fb1e9ede27fca69cf9074c2590c221dce3633a68 +RH-Commit: [18/30] ec8510be6b23b26b3eecd6767e1deb0c0c50dd58 (stefanha/centos-stream-qemu-kvm) The 'do_hash_operation' method has a 256 element iovec array used for holding pointers to data that is to be hashed. Skip the automatic diff --git a/SOURCES/kvm-hw-net-rtl8139-skip-automatic-zero-init-of-large-arr.patch b/SOURCES/kvm-hw-net-rtl8139-skip-automatic-zero-init-of-large-arr.patch index 6e6cc3d..8161972 100644 --- a/SOURCES/kvm-hw-net-rtl8139-skip-automatic-zero-init-of-large-arr.patch +++ b/SOURCES/kvm-hw-net-rtl8139-skip-automatic-zero-init-of-large-arr.patch @@ -1,16 +1,16 @@ -From 09fe29d40b3d1e30e0e921e186a797c4da5ac583 Mon Sep 17 00:00:00 2001 +From cc173deaaa4d9dc6ad9188e0b03f46b7e64f26b2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:57 +0100 -Subject: [PATCH 20/31] hw/net/rtl8139: skip automatic zero-init of large array +Subject: [PATCH 45/57] hw/net/rtl8139: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [19/30] 3c072f59b9a283b40327117585d9f01d32ecc081 +RH-Commit: [19/30] 344c720aef2feb35f84fd4b21f2b1b31e5572286 (stefanha/centos-stream-qemu-kvm) The 'rtl8139_transmit_one' method has a 8k byte array used for copying data between guest and host. Skip the automatic zero-init diff --git a/SOURCES/kvm-hw-net-tulip-skip-automatic-zero-init-of-large-array.patch b/SOURCES/kvm-hw-net-tulip-skip-automatic-zero-init-of-large-array.patch index e5d8430..06ea05e 100644 --- a/SOURCES/kvm-hw-net-tulip-skip-automatic-zero-init-of-large-array.patch +++ b/SOURCES/kvm-hw-net-tulip-skip-automatic-zero-init-of-large-array.patch @@ -1,16 +1,16 @@ -From f9a1a355dbd2d59bbd80d33e713579d93fe3932f Mon Sep 17 00:00:00 2001 +From 400b5c8ae7f06a450ef91230343d7ce489142a38 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:58 +0100 -Subject: [PATCH 21/31] hw/net/tulip: skip automatic zero-init of large array +Subject: [PATCH 46/57] hw/net/tulip: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [20/30] ff60b673f4e06a25fd5efbc51f5536da5d9c99f5 +RH-Commit: [20/30] b3d29de8495c0ff40c26974673adefe4eb27a417 (stefanha/centos-stream-qemu-kvm) The 'tulip_setup_frame' method has a 4k byte array used for copynig DMA data from the device. Skip the automatic zero-init of this array diff --git a/SOURCES/kvm-hw-net-virtio-net-skip-automatic-zero-init-of-large-.patch b/SOURCES/kvm-hw-net-virtio-net-skip-automatic-zero-init-of-large-.patch index 47c8b0b..4fbe7a4 100644 --- a/SOURCES/kvm-hw-net-virtio-net-skip-automatic-zero-init-of-large-.patch +++ b/SOURCES/kvm-hw-net-virtio-net-skip-automatic-zero-init-of-large-.patch @@ -1,17 +1,17 @@ -From 9ecc539204dd6ab7a1124089f9e557248e321282 Mon Sep 17 00:00:00 2001 +From 0925796a4537e20e033a675ebc8899e4580235f3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:36:59 +0100 -Subject: [PATCH 22/31] hw/net/virtio-net: skip automatic zero-init of large +Subject: [PATCH 47/57] hw/net/virtio-net: skip automatic zero-init of large arrays MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [21/30] f093a50cc162bd376fd74ea47c6274d7e718ba69 +RH-Commit: [21/30] 0450189a4c4c779b5a1850e9ea8278a5129c5f7f (stefanha/centos-stream-qemu-kvm) The 'virtio_net_receive_rcu' method has three arrays with VIRTQUEUE_MAX_SIZE elements, which are apprixmately 32k in diff --git a/SOURCES/kvm-hw-net-xgamc-skip-automatic-zero-init-of-large-array.patch b/SOURCES/kvm-hw-net-xgamc-skip-automatic-zero-init-of-large-array.patch index fc1d17f..027ab99 100644 --- a/SOURCES/kvm-hw-net-xgamc-skip-automatic-zero-init-of-large-array.patch +++ b/SOURCES/kvm-hw-net-xgamc-skip-automatic-zero-init-of-large-array.patch @@ -1,16 +1,16 @@ -From 3b39fa3e031d5b8a89c05302f3d73e7d4748bf58 Mon Sep 17 00:00:00 2001 +From 34116b3a243f005938a30e9b38c6f47a62752c3e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:37:00 +0100 -Subject: [PATCH 23/31] hw/net/xgamc: skip automatic zero-init of large array +Subject: [PATCH 48/57] hw/net/xgamc: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [22/30] d83a91284970ced6f60964ded15d04405845e8bb +RH-Commit: [22/30] 63536d627705775c4bf72a511de3d68ec30ac7de (stefanha/centos-stream-qemu-kvm) The 'xgmac_enet_send' method has a 8k byte array used for copying data between guest and host. Skip the automatic zero-init of this diff --git a/SOURCES/kvm-hw-nvme-ctrl-skip-automatic-zero-init-of-large-array.patch b/SOURCES/kvm-hw-nvme-ctrl-skip-automatic-zero-init-of-large-array.patch index 4dd62b7..6a84a1c 100644 --- a/SOURCES/kvm-hw-nvme-ctrl-skip-automatic-zero-init-of-large-array.patch +++ b/SOURCES/kvm-hw-nvme-ctrl-skip-automatic-zero-init-of-large-array.patch @@ -1,16 +1,16 @@ -From 794d838efddc7e96f6e40c1c4bb2b1baf3c95cfb Mon Sep 17 00:00:00 2001 +From 3e0134b45828bf9a623a26ac41d5fbb3a8d2917b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:37:01 +0100 -Subject: [PATCH 24/31] hw/nvme/ctrl: skip automatic zero-init of large arrays +Subject: [PATCH 49/57] hw/nvme/ctrl: skip automatic zero-init of large arrays MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [23/30] 97877d2e280daf654f5893461c0bc9e6f6caa77d +RH-Commit: [23/30] 57ce4361ffb307be4ea4d3edf9e0dac269d16908 (stefanha/centos-stream-qemu-kvm) The 'nvme_map_sgl' method has a 256 element array used for copying data from the device. Skip the automatic zero-init of this array @@ -37,7 +37,7 @@ Signed-off-by: Stefan Hajnoczi 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/hw/nvme/ctrl.c b/hw/nvme/ctrl.c -index 9f277b81d8..f000e2246f 100644 +index d451ee0d00..75d7f20801 100644 --- a/hw/nvme/ctrl.c +++ b/hw/nvme/ctrl.c @@ -1047,7 +1047,8 @@ static uint16_t nvme_map_sgl(NvmeCtrl *n, NvmeSg *sg, NvmeSglDescriptor sgl, diff --git a/SOURCES/kvm-hw-pci-Basic-support-for-PCI-power-management.patch b/SOURCES/kvm-hw-pci-Basic-support-for-PCI-power-management.patch new file mode 100644 index 0000000..6287a46 --- /dev/null +++ b/SOURCES/kvm-hw-pci-Basic-support-for-PCI-power-management.patch @@ -0,0 +1,242 @@ +From 98b0cd83c09d35a3da0ae142c09038174355e87e Mon Sep 17 00:00:00 2001 +From: Alex Williamson +Date: Tue, 25 Feb 2025 14:52:25 -0700 +Subject: [PATCH 2/7] hw/pci: Basic support for PCI power management +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Eric Auger +RH-MergeRequest: 348: PCI: Implement basic PCI PM capability backing +RH-Jira: RHEL-7301 +RH-Acked-by: Cédric Le Goater +RH-Acked-by: Alex Williamson +RH-Acked-by: Jon Maloy +RH-Commit: [2/6] 5faff6382c124711887704fff4f857e8f85e7be5 (eauger1/centos-qemu-kvm) + +Conflicts: contextual conflict in include/hw/pci/pci.h +we don't have 449dca6ac93a ("pcie: enable Extended tag field support") +downstream so we don't have x-pcie-ext-tag definition. + +The memory and IO BARs for devices are only accessible in the D0 power +state. In other power states the PCI spec defines that the device +responds to TLPs and messages with an Unsupported Request response. + +To approximate this behavior, consider the BARs as unmapped when the +device is not in the D0 power state. This makes the BARs inaccessible +and has the additional bonus for vfio-pci that we don't attempt to DMA +map BARs for devices in a non-D0 power state. + +To support this, an interface is added for devices to register the PM +capability, which allows central tracking to enforce valid transitions +and unmap BARs in non-D0 states. + +NB. We currently have device models (eepro100 and pcie_pci_bridge) +that register a PM capability but do not set wmask to enable writes to +the power state field. In order to maintain migration compatibility, +this new helper does not manage the wmask to enable guest writes to +initiate a power state change. The contents and write access of the +PM capability are still managed by the caller. + +Cc: Michael S. Tsirkin +Cc: Marcel Apfelbaum +Signed-off-by: Alex Williamson +Reviewed-by: Eric Auger +Reviewed-by: Michael S. Tsirkin +Link: https://lore.kernel.org/qemu-devel/20250225215237.3314011-2-alex.williamson@redhat.com +Signed-off-by: Cédric Le Goater +(cherry picked from commit 9461afd2008b0820fc45a6a7bc675df1b6791e4f) +Signed-off-by: Eric Auger +--- + hw/pci/pci.c | 93 ++++++++++++++++++++++++++++++++++++- + hw/pci/trace-events | 2 + + include/hw/pci/pci.h | 3 ++ + include/hw/pci/pci_device.h | 3 ++ + 4 files changed, 99 insertions(+), 2 deletions(-) + +diff --git a/hw/pci/pci.c b/hw/pci/pci.c +index 83c9d5b9ea..d774ae47d2 100644 +--- a/hw/pci/pci.c ++++ b/hw/pci/pci.c +@@ -365,6 +365,84 @@ static void pci_msi_trigger(PCIDevice *dev, MSIMessage msg) + attrs, NULL); + } + ++/* ++ * Register and track a PM capability. If wmask is also enabled for the power ++ * state field of the pmcsr register, guest writes may change the device PM ++ * state. BAR access is only enabled while the device is in the D0 state. ++ * Return the capability offset or negative error code. ++ */ ++int pci_pm_init(PCIDevice *d, uint8_t offset, Error **errp) ++{ ++ int cap = pci_add_capability(d, PCI_CAP_ID_PM, offset, PCI_PM_SIZEOF, errp); ++ ++ if (cap < 0) { ++ return cap; ++ } ++ ++ d->pm_cap = cap; ++ d->cap_present |= QEMU_PCI_CAP_PM; ++ ++ return cap; ++} ++ ++static uint8_t pci_pm_state(PCIDevice *d) ++{ ++ uint16_t pmcsr; ++ ++ if (!(d->cap_present & QEMU_PCI_CAP_PM)) { ++ return 0; ++ } ++ ++ pmcsr = pci_get_word(d->config + d->pm_cap + PCI_PM_CTRL); ++ ++ return pmcsr & PCI_PM_CTRL_STATE_MASK; ++} ++ ++/* ++ * Update the PM capability state based on the new value stored in config ++ * space respective to the old, pre-write state provided. If the new value ++ * is rejected (unsupported or invalid transition) restore the old value. ++ * Return the resulting PM state. ++ */ ++static uint8_t pci_pm_update(PCIDevice *d, uint32_t addr, int l, uint8_t old) ++{ ++ uint16_t pmc; ++ uint8_t new; ++ ++ if (!(d->cap_present & QEMU_PCI_CAP_PM) || ++ !range_covers_byte(addr, l, d->pm_cap + PCI_PM_CTRL)) { ++ return old; ++ } ++ ++ new = pci_pm_state(d); ++ if (new == old) { ++ return old; ++ } ++ ++ pmc = pci_get_word(d->config + d->pm_cap + PCI_PM_PMC); ++ ++ /* ++ * Transitions to D1 & D2 are only allowed if supported. Devices may ++ * only transition to higher D-states or to D0. ++ */ ++ if ((!(pmc & PCI_PM_CAP_D1) && new == 1) || ++ (!(pmc & PCI_PM_CAP_D2) && new == 2) || ++ (old && new && new < old)) { ++ pci_word_test_and_clear_mask(d->config + d->pm_cap + PCI_PM_CTRL, ++ PCI_PM_CTRL_STATE_MASK); ++ pci_word_test_and_set_mask(d->config + d->pm_cap + PCI_PM_CTRL, ++ old); ++ trace_pci_pm_bad_transition(d->name, pci_dev_bus_num(d), ++ PCI_SLOT(d->devfn), PCI_FUNC(d->devfn), ++ old, new); ++ return old; ++ } ++ ++ trace_pci_pm_transition(d->name, pci_dev_bus_num(d), PCI_SLOT(d->devfn), ++ PCI_FUNC(d->devfn), old, new); ++ return new; ++} ++ + static void pci_reset_regions(PCIDevice *dev) + { + int r; +@@ -404,6 +482,11 @@ static void pci_do_device_reset(PCIDevice *dev) + pci_get_word(dev->wmask + PCI_INTERRUPT_LINE) | + pci_get_word(dev->w1cmask + PCI_INTERRUPT_LINE)); + dev->config[PCI_CACHE_LINE_SIZE] = 0x0; ++ /* Default PM state is D0 */ ++ if (dev->cap_present & QEMU_PCI_CAP_PM) { ++ pci_word_test_and_clear_mask(dev->config + dev->pm_cap + PCI_PM_CTRL, ++ PCI_PM_CTRL_STATE_MASK); ++ } + pci_reset_regions(dev); + pci_update_mappings(dev); + +@@ -1525,7 +1608,7 @@ static void pci_update_mappings(PCIDevice *d) + continue; + + new_addr = pci_bar_address(d, i, r->type, r->size); +- if (!d->enabled) { ++ if (!d->enabled || pci_pm_state(d)) { + new_addr = PCI_BAR_UNMAPPED; + } + +@@ -1591,6 +1674,7 @@ uint32_t pci_default_read_config(PCIDevice *d, + + void pci_default_write_config(PCIDevice *d, uint32_t addr, uint32_t val_in, int l) + { ++ uint8_t new_pm_state, old_pm_state = pci_pm_state(d); + int i, was_irq_disabled = pci_irq_disabled(d); + uint32_t val = val_in; + +@@ -1603,11 +1687,16 @@ void pci_default_write_config(PCIDevice *d, uint32_t addr, uint32_t val_in, int + d->config[addr + i] = (d->config[addr + i] & ~wmask) | (val & wmask); + d->config[addr + i] &= ~(val & w1cmask); /* W1C: Write 1 to Clear */ + } ++ ++ new_pm_state = pci_pm_update(d, addr, l, old_pm_state); ++ + if (ranges_overlap(addr, l, PCI_BASE_ADDRESS_0, 24) || + ranges_overlap(addr, l, PCI_ROM_ADDRESS, 4) || + ranges_overlap(addr, l, PCI_ROM_ADDRESS1, 4) || +- range_covers_byte(addr, l, PCI_COMMAND)) ++ range_covers_byte(addr, l, PCI_COMMAND) || ++ !!new_pm_state != !!old_pm_state) { + pci_update_mappings(d); ++ } + + if (ranges_overlap(addr, l, PCI_COMMAND, 2)) { + pci_update_irq_disabled(d, was_irq_disabled); +diff --git a/hw/pci/trace-events b/hw/pci/trace-events +index 19643aa8c6..c82a87ffdd 100644 +--- a/hw/pci/trace-events ++++ b/hw/pci/trace-events +@@ -1,6 +1,8 @@ + # See docs/devel/tracing.rst for syntax documentation. + + # pci.c ++pci_pm_bad_transition(const char *dev, uint32_t bus, uint32_t slot, uint32_t func, uint8_t old, uint8_t new) "%s %02x:%02x.%x REJECTED PM transition D%d->D%d" ++pci_pm_transition(const char *dev, uint32_t bus, uint32_t slot, uint32_t func, uint8_t old, uint8_t new) "%s %02x:%02x.%x PM transition D%d->D%d" + pci_update_mappings_del(const char *dev, uint32_t bus, uint32_t slot, uint32_t func, int bar, uint64_t addr, uint64_t size) "%s %02x:%02x.%x %d,0x%"PRIx64"+0x%"PRIx64 + pci_update_mappings_add(const char *dev, uint32_t bus, uint32_t slot, uint32_t func, int bar, uint64_t addr, uint64_t size) "%s %02x:%02x.%x %d,0x%"PRIx64"+0x%"PRIx64 + pci_route_irq(int dev_irq, const char *dev_path, int parent_irq, const char *parent_path) "IRQ %d @%s -> IRQ %d @%s" +diff --git a/include/hw/pci/pci.h b/include/hw/pci/pci.h +index 45365ae085..afeb5a2263 100644 +--- a/include/hw/pci/pci.h ++++ b/include/hw/pci/pci.h +@@ -213,6 +213,8 @@ enum { + QEMU_PCIE_ERR_UNC_MASK = (1 << QEMU_PCIE_ERR_UNC_MASK_BITNR), + #define QEMU_PCIE_ARI_NEXTFN_1_BITNR 12 + QEMU_PCIE_ARI_NEXTFN_1 = (1 << QEMU_PCIE_ARI_NEXTFN_1_BITNR), ++#define QEMU_PCI_CAP_PM_BITNR 14 ++ QEMU_PCI_CAP_PM = (1 << QEMU_PCI_CAP_PM_BITNR), + }; + + typedef struct PCIINTxRoute { +@@ -680,5 +682,6 @@ static inline void pci_irq_pulse(PCIDevice *pci_dev) + MSIMessage pci_get_msi_message(PCIDevice *dev, int vector); + void pci_set_enabled(PCIDevice *pci_dev, bool state); + void pci_set_power(PCIDevice *pci_dev, bool state); ++int pci_pm_init(PCIDevice *pci_dev, uint8_t offset, Error **errp); + + #endif +diff --git a/include/hw/pci/pci_device.h b/include/hw/pci/pci_device.h +index f38fb31119..325d7bcaf7 100644 +--- a/include/hw/pci/pci_device.h ++++ b/include/hw/pci/pci_device.h +@@ -105,6 +105,9 @@ struct PCIDevice { + /* Capability bits */ + uint32_t cap_present; + ++ /* Offset of PM capability in config space */ ++ uint8_t pm_cap; ++ + /* Offset of MSI-X capability in config space */ + uint8_t msix_cap; + +-- +2.48.1 + diff --git a/SOURCES/kvm-hw-pci-Rename-has_power-to-enabled.patch b/SOURCES/kvm-hw-pci-Rename-has_power-to-enabled.patch new file mode 100644 index 0000000..4041ddb --- /dev/null +++ b/SOURCES/kvm-hw-pci-Rename-has_power-to-enabled.patch @@ -0,0 +1,130 @@ +From 8711bb1a54d4f5734d44545cd8e7262bc358f51d Mon Sep 17 00:00:00 2001 +From: Akihiko Odaki +Date: Thu, 9 Jan 2025 15:29:46 +0900 +Subject: [PATCH 1/7] hw/pci: Rename has_power to enabled +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Eric Auger +RH-MergeRequest: 348: PCI: Implement basic PCI PM capability backing +RH-Jira: RHEL-7301 +RH-Acked-by: Cédric Le Goater +RH-Acked-by: Alex Williamson +RH-Acked-by: Jon Maloy +RH-Commit: [1/6] ac8a7427a1203e33aa323933818a7114c0eb4520 (eauger1/centos-qemu-kvm) + +The renamed state will not only represent powering state of PFs, but +also represent SR-IOV VF enablement in the future. + +Signed-off-by: Akihiko Odaki +Reviewed-by: Philippe Mathieu-Daudé +Message-ID: <20250109-reuse-v19-1-f541e82ca5f7@daynix.com> +Signed-off-by: Philippe Mathieu-Daudé +(cherry picked from commit c407eef162f765dd83d45e048585731be41a66fc) +Signed-off-by: Eric Auger +--- + hw/pci/pci.c | 17 +++++++++++------ + hw/pci/pci_host.c | 4 ++-- + include/hw/pci/pci.h | 1 + + include/hw/pci/pci_device.h | 2 +- + 4 files changed, 15 insertions(+), 9 deletions(-) + +diff --git a/hw/pci/pci.c b/hw/pci/pci.c +index fab86d0567..83c9d5b9ea 100644 +--- a/hw/pci/pci.c ++++ b/hw/pci/pci.c +@@ -1525,7 +1525,7 @@ static void pci_update_mappings(PCIDevice *d) + continue; + + new_addr = pci_bar_address(d, i, r->type, r->size); +- if (!d->has_power) { ++ if (!d->enabled) { + new_addr = PCI_BAR_UNMAPPED; + } + +@@ -1613,7 +1613,7 @@ void pci_default_write_config(PCIDevice *d, uint32_t addr, uint32_t val_in, int + pci_update_irq_disabled(d, was_irq_disabled); + memory_region_set_enabled(&d->bus_master_enable_region, + (pci_get_word(d->config + PCI_COMMAND) +- & PCI_COMMAND_MASTER) && d->has_power); ++ & PCI_COMMAND_MASTER) && d->enabled); + } + + msi_write_config(d, addr, val_in, l); +@@ -2886,16 +2886,21 @@ MSIMessage pci_get_msi_message(PCIDevice *dev, int vector) + + void pci_set_power(PCIDevice *d, bool state) + { +- if (d->has_power == state) { ++ pci_set_enabled(d, state); ++} ++ ++void pci_set_enabled(PCIDevice *d, bool state) ++{ ++ if (d->enabled == state) { + return; + } + +- d->has_power = state; ++ d->enabled = state; + pci_update_mappings(d); + memory_region_set_enabled(&d->bus_master_enable_region, + (pci_get_word(d->config + PCI_COMMAND) +- & PCI_COMMAND_MASTER) && d->has_power); +- if (!d->has_power) { ++ & PCI_COMMAND_MASTER) && d->enabled); ++ if (!d->enabled) { + pci_device_reset(d); + } + } +diff --git a/hw/pci/pci_host.c b/hw/pci/pci_host.c +index dfe6fe6184..0d82727cc9 100644 +--- a/hw/pci/pci_host.c ++++ b/hw/pci/pci_host.c +@@ -86,7 +86,7 @@ void pci_host_config_write_common(PCIDevice *pci_dev, uint32_t addr, + * allowing direct removal of unexposed functions. + */ + if ((pci_dev->qdev.hotplugged && !pci_get_function_0(pci_dev)) || +- !pci_dev->has_power || is_pci_dev_ejected(pci_dev)) { ++ !pci_dev->enabled || is_pci_dev_ejected(pci_dev)) { + return; + } + +@@ -111,7 +111,7 @@ uint32_t pci_host_config_read_common(PCIDevice *pci_dev, uint32_t addr, + * allowing direct removal of unexposed functions. + */ + if ((pci_dev->qdev.hotplugged && !pci_get_function_0(pci_dev)) || +- !pci_dev->has_power || is_pci_dev_ejected(pci_dev)) { ++ !pci_dev->enabled || is_pci_dev_ejected(pci_dev)) { + return ~0x0; + } + +diff --git a/include/hw/pci/pci.h b/include/hw/pci/pci.h +index eb26cac810..45365ae085 100644 +--- a/include/hw/pci/pci.h ++++ b/include/hw/pci/pci.h +@@ -678,6 +678,7 @@ static inline void pci_irq_pulse(PCIDevice *pci_dev) + } + + MSIMessage pci_get_msi_message(PCIDevice *dev, int vector); ++void pci_set_enabled(PCIDevice *pci_dev, bool state); + void pci_set_power(PCIDevice *pci_dev, bool state); + + #endif +diff --git a/include/hw/pci/pci_device.h b/include/hw/pci/pci_device.h +index 15694f2489..f38fb31119 100644 +--- a/include/hw/pci/pci_device.h ++++ b/include/hw/pci/pci_device.h +@@ -57,7 +57,7 @@ typedef struct PCIReqIDCache PCIReqIDCache; + struct PCIDevice { + DeviceState qdev; + bool partially_hotplugged; +- bool has_power; ++ bool enabled; + + /* PCI config space */ + uint8_t *config; +-- +2.48.1 + diff --git a/SOURCES/kvm-hw-ppc-spapr_tpm_proxy-skip-automatic-zero-init-of-l.patch b/SOURCES/kvm-hw-ppc-spapr_tpm_proxy-skip-automatic-zero-init-of-l.patch index 0113427..4b12664 100644 --- a/SOURCES/kvm-hw-ppc-spapr_tpm_proxy-skip-automatic-zero-init-of-l.patch +++ b/SOURCES/kvm-hw-ppc-spapr_tpm_proxy-skip-automatic-zero-init-of-l.patch @@ -1,17 +1,17 @@ -From 087151816f810052c013d496e32be1011e5c01ef Mon Sep 17 00:00:00 2001 +From 4c3fe6e7b88c58713c0c499d4bf0658a055ee52e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:37:03 +0100 -Subject: [PATCH 25/31] hw/ppc/spapr_tpm_proxy: skip automatic zero-init of +Subject: [PATCH 50/57] hw/ppc/spapr_tpm_proxy: skip automatic zero-init of large arrays MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [24/30] a2360ae956c03481af7aceb34d26c7d8ba33a1d7 +RH-Commit: [24/30] 8d963380c64a33a27adc99738b42b52864229111 (stefanha/centos-stream-qemu-kvm) The 'tpm_execute' method has a pair of 4k arrays used for copying data between guest and host. Skip the automatic zero-init of these diff --git a/SOURCES/kvm-hw-s390-ccw-device-Convert-to-three-phase-reset.patch b/SOURCES/kvm-hw-s390-ccw-device-Convert-to-three-phase-reset.patch new file mode 100644 index 0000000..d5b71e3 --- /dev/null +++ b/SOURCES/kvm-hw-s390-ccw-device-Convert-to-three-phase-reset.patch @@ -0,0 +1,63 @@ +From 5126609c0714c66a0ec41328017e7e8388c78bf4 Mon Sep 17 00:00:00 2001 +From: Peter Maydell +Date: Fri, 13 Sep 2024 15:31:43 +0100 +Subject: [PATCH 02/26] hw/s390/ccw-device: Convert to three-phase reset +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [2/26] 58f6fc2e65a101e069feac399859464d31e43045 (thuth/qemu-kvm-cs) + +Convert the TYPE_CCW_DEVICE to three-phase reset. This is a +device class which is subclassed, so it needs to be three-phase +before we can convert the subclass. + +Signed-off-by: Peter Maydell +Reviewed-by: Nina Schoetterl-Glausch +Reviewed-by: Philippe Mathieu-Daudé +Acked-by: Thomas Huth +Message-id: 20240830145812.1967042-2-peter.maydell@linaro.org +(cherry picked from commit 6a0e10b76b68e2f412746a1d5ed7d6efee804864) +Signed-off-by: Thomas Huth +--- + hw/s390x/ccw-device.c | 7 ++++--- + 1 file changed, 4 insertions(+), 3 deletions(-) + +diff --git a/hw/s390x/ccw-device.c b/hw/s390x/ccw-device.c +index d7bb364579..30f2fb486f 100644 +--- a/hw/s390x/ccw-device.c ++++ b/hw/s390x/ccw-device.c +@@ -88,9 +88,9 @@ static Property ccw_device_properties[] = { + DEFINE_PROP_END_OF_LIST(), + }; + +-static void ccw_device_reset(DeviceState *d) ++static void ccw_device_reset_hold(Object *obj, ResetType type) + { +- CcwDevice *ccw_dev = CCW_DEVICE(d); ++ CcwDevice *ccw_dev = CCW_DEVICE(obj); + + css_reset_sch(ccw_dev->sch); + } +@@ -99,11 +99,12 @@ static void ccw_device_class_init(ObjectClass *klass, void *data) + { + DeviceClass *dc = DEVICE_CLASS(klass); + CCWDeviceClass *k = CCW_DEVICE_CLASS(klass); ++ ResettableClass *rc = RESETTABLE_CLASS(klass); + + k->realize = ccw_device_realize; + k->refill_ids = ccw_device_refill_ids; + device_class_set_props(dc, ccw_device_properties); +- dc->reset = ccw_device_reset; ++ rc->phases.hold = ccw_device_reset_hold; + dc->bus_type = TYPE_VIRTUAL_CSS_BUS; + } + +-- +2.48.1 + diff --git a/SOURCES/kvm-hw-s390-virtio-ccw-Convert-to-three-phase-reset.patch b/SOURCES/kvm-hw-s390-virtio-ccw-Convert-to-three-phase-reset.patch new file mode 100644 index 0000000..15ea0b0 --- /dev/null +++ b/SOURCES/kvm-hw-s390-virtio-ccw-Convert-to-three-phase-reset.patch @@ -0,0 +1,92 @@ +From 7cbf9be09907407a64d739a2d0862af2ad08eaf5 Mon Sep 17 00:00:00 2001 +From: Peter Maydell +Date: Fri, 13 Sep 2024 15:31:43 +0100 +Subject: [PATCH 03/26] hw/s390/virtio-ccw: Convert to three-phase reset +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [3/26] e06ee194fa289a387433b905eb0999a048681a92 (thuth/qemu-kvm-cs) + +Convert the virtio-ccw code to three-phase reset. This allows us to +remove a call to device_class_set_parent_reset(), replacing it with +the three-phase equivalent resettable_class_set_parent_phases(). +Removing all the device_class_set_parent_reset() uses will allow us +to remove some of the glue code that interworks between three-phase +and legacy reset. + +This is a simple conversion, with no behavioural changes. + +Signed-off-by: Peter Maydell +Reviewed-by: Philippe Mathieu-Daudé +Reviewed-by: Nina Schoetterl-Glausch +Acked-by: Thomas Huth +Reviewed-by: Richard Henderson +Message-id: 20240830145812.1967042-3-peter.maydell@linaro.org +(cherry picked from commit 6affa00d6ebebf24485667fe146470b0d6feb90d) +Signed-off-by: Thomas Huth +--- + hw/s390x/virtio-ccw.c | 13 ++++++++----- + hw/s390x/virtio-ccw.h | 2 +- + 2 files changed, 9 insertions(+), 6 deletions(-) + +diff --git a/hw/s390x/virtio-ccw.c b/hw/s390x/virtio-ccw.c +index b4676909dd..96747318d2 100644 +--- a/hw/s390x/virtio-ccw.c ++++ b/hw/s390x/virtio-ccw.c +@@ -913,14 +913,15 @@ static void virtio_ccw_notify(DeviceState *d, uint16_t vector) + } + } + +-static void virtio_ccw_reset(DeviceState *d) ++static void virtio_ccw_reset_hold(Object *obj, ResetType type) + { +- VirtioCcwDevice *dev = VIRTIO_CCW_DEVICE(d); ++ VirtioCcwDevice *dev = VIRTIO_CCW_DEVICE(obj); + VirtIOCCWDeviceClass *vdc = VIRTIO_CCW_DEVICE_GET_CLASS(dev); + + virtio_ccw_reset_virtio(dev); +- if (vdc->parent_reset) { +- vdc->parent_reset(d); ++ ++ if (vdc->parent_phases.hold) { ++ vdc->parent_phases.hold(obj, type); + } + } + +@@ -1233,11 +1234,13 @@ static void virtio_ccw_device_class_init(ObjectClass *klass, void *data) + DeviceClass *dc = DEVICE_CLASS(klass); + CCWDeviceClass *k = CCW_DEVICE_CLASS(dc); + VirtIOCCWDeviceClass *vdc = VIRTIO_CCW_DEVICE_CLASS(klass); ++ ResettableClass *rc = RESETTABLE_CLASS(klass); + + k->unplug = virtio_ccw_busdev_unplug; + dc->realize = virtio_ccw_busdev_realize; + dc->unrealize = virtio_ccw_busdev_unrealize; +- device_class_set_parent_reset(dc, virtio_ccw_reset, &vdc->parent_reset); ++ resettable_class_set_parent_phases(rc, NULL, virtio_ccw_reset_hold, NULL, ++ &vdc->parent_phases); + } + + static const TypeInfo virtio_ccw_device_info = { +diff --git a/hw/s390x/virtio-ccw.h b/hw/s390x/virtio-ccw.h +index fac186c8f6..c7a830a194 100644 +--- a/hw/s390x/virtio-ccw.h ++++ b/hw/s390x/virtio-ccw.h +@@ -57,7 +57,7 @@ struct VirtIOCCWDeviceClass { + CCWDeviceClass parent_class; + void (*realize)(VirtioCcwDevice *dev, Error **errp); + void (*unrealize)(VirtioCcwDevice *dev); +- void (*parent_reset)(DeviceState *dev); ++ ResettablePhases parent_phases; + }; + + /* Performance improves when virtqueue kick processing is decoupled from the +-- +2.48.1 + diff --git a/SOURCES/kvm-hw-s390x-ccw-device-Fix-memory-leak-in-loadparm-sett.patch b/SOURCES/kvm-hw-s390x-ccw-device-Fix-memory-leak-in-loadparm-sett.patch new file mode 100644 index 0000000..5bf4b1a --- /dev/null +++ b/SOURCES/kvm-hw-s390x-ccw-device-Fix-memory-leak-in-loadparm-sett.patch @@ -0,0 +1,47 @@ +From b25bbfcad4a3df94555f6b5f238910314a5d17ea Mon Sep 17 00:00:00 2001 +From: Kevin Wolf +Date: Wed, 25 Jun 2025 10:27:51 +0200 +Subject: [PATCH 02/57] hw/s390x/ccw-device: Fix memory leak in loadparm setter +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Thomas Huth +RH-MergeRequest: 387: s390x: Fix memory leaks related to loadparm [rhel-9] +RH-Jira: RHEL-98554 +RH-Acked-by: Cédric Le Goater +RH-Acked-by: Kevin Wolf +RH-Commit: [2/2] d85cf8b3c93ede47b51c4aa1336dc54f58b8cc3f (thuth/qemu-kvm-cs) + +Commit bdf12f2a fixed the setter for the "loadparm" machine property, +which gets a string from a visitor, passes it to s390_ipl_fmt_loadparm() +and then forgot to free it. It left another instance of the same problem +unfixed in the "loadparm" device property. Fix it. + +Signed-off-by: Kevin Wolf +Message-ID: <20250625082751.24896-1-kwolf@redhat.com> +Reviewed-by: Eric Farman +Reviewed-by: Halil Pasic +Tested-by: Thomas Huth +Signed-off-by: Thomas Huth +(cherry picked from commit 78e3781541209b3dcd6f4bb66adf3a3e504b88a4) +--- + hw/s390x/ccw-device.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/hw/s390x/ccw-device.c b/hw/s390x/ccw-device.c +index 30f2fb486f..63e937401e 100644 +--- a/hw/s390x/ccw-device.c ++++ b/hw/s390x/ccw-device.c +@@ -57,7 +57,7 @@ static void ccw_device_set_loadparm(Object *obj, Visitor *v, + Error **errp) + { + CcwDevice *dev = CCW_DEVICE(obj); +- char *val; ++ g_autofree char *val = NULL; + int index; + + index = object_property_get_int(obj, "bootindex", NULL); +-- +2.39.3 + diff --git a/SOURCES/kvm-hw-scsi-lsi53c895a-skip-automatic-zero-init-of-large.patch b/SOURCES/kvm-hw-scsi-lsi53c895a-skip-automatic-zero-init-of-large.patch index 9eee3af..77a1f92 100644 --- a/SOURCES/kvm-hw-scsi-lsi53c895a-skip-automatic-zero-init-of-large.patch +++ b/SOURCES/kvm-hw-scsi-lsi53c895a-skip-automatic-zero-init-of-large.patch @@ -1,17 +1,17 @@ -From 3179e7e183295b91e314c32db8697d8cf0947367 Mon Sep 17 00:00:00 2001 +From 45884bfad1f14585407a04eff9230a75bc5095fa Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:37:05 +0100 -Subject: [PATCH 27/31] hw/scsi/lsi53c895a: skip automatic zero-init of large +Subject: [PATCH 52/57] hw/scsi/lsi53c895a: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [26/30] 92ec0b18f26956767d2f3d0284d712696d60852e +RH-Commit: [26/30] 235884d43fcb3e49b320e36faa631a3656d07de6 (stefanha/centos-stream-qemu-kvm) The 'lsi_memcpy' method has a 4k byte array used for copying data to/from the device. Skip the automatic zero-init of this array to diff --git a/SOURCES/kvm-hw-scsi-megasas-skip-automatic-zero-init-of-large-ar.patch b/SOURCES/kvm-hw-scsi-megasas-skip-automatic-zero-init-of-large-ar.patch index 8260337..140160c 100644 --- a/SOURCES/kvm-hw-scsi-megasas-skip-automatic-zero-init-of-large-ar.patch +++ b/SOURCES/kvm-hw-scsi-megasas-skip-automatic-zero-init-of-large-ar.patch @@ -1,17 +1,17 @@ -From 30a938409f664179061b039796354628c714229e Mon Sep 17 00:00:00 2001 +From 9f76103e90ce8406bc5bbda72a7314b82e56652e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:37:06 +0100 -Subject: [PATCH 28/31] hw/scsi/megasas: skip automatic zero-init of large +Subject: [PATCH 53/57] hw/scsi/megasas: skip automatic zero-init of large arrays MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [27/30] 70c3c002601bc54ff81d000092e7be2bdb1eb82a +RH-Commit: [27/30] b3a3f466fd03c64c665c52e26079b03def376f48 (stefanha/centos-stream-qemu-kvm) The 'megasas_dcmd_pd_get_list' and 'megasas_dcmd_get_properties' methods have 4k structs used for copying data from the device. diff --git a/SOURCES/kvm-hw-ufs-lu-skip-automatic-zero-init-of-large-array.patch b/SOURCES/kvm-hw-ufs-lu-skip-automatic-zero-init-of-large-array.patch index e9742a4..175b89b 100644 --- a/SOURCES/kvm-hw-ufs-lu-skip-automatic-zero-init-of-large-array.patch +++ b/SOURCES/kvm-hw-ufs-lu-skip-automatic-zero-init-of-large-array.patch @@ -1,16 +1,16 @@ -From 36046b3119eb2338c811f43b2956c3aa787a2e3c Mon Sep 17 00:00:00 2001 +From 3a0ae5a2f873fc7062262efc24a5403233988f5f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:37:07 +0100 -Subject: [PATCH 29/31] hw/ufs/lu: skip automatic zero-init of large array +Subject: [PATCH 54/57] hw/ufs/lu: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [28/30] 4d68ab4596b8fb97106d07e5af11dc3fdcb25e96 +RH-Commit: [28/30] 62e7c83d15143387f6d6b366c8ec46b312d05577 (stefanha/centos-stream-qemu-kvm) The 'ufs_emulate_scsi_cmd' method has a 4k byte array used for copying data from the device. Skip the automatic zero-init of diff --git a/SOURCES/kvm-hw-usb-hcd-ohci-skip-automatic-zero-init-of-large-ar.patch b/SOURCES/kvm-hw-usb-hcd-ohci-skip-automatic-zero-init-of-large-ar.patch index 6381045..b5daa5b 100644 --- a/SOURCES/kvm-hw-usb-hcd-ohci-skip-automatic-zero-init-of-large-ar.patch +++ b/SOURCES/kvm-hw-usb-hcd-ohci-skip-automatic-zero-init-of-large-ar.patch @@ -1,17 +1,17 @@ -From 31886b02875b9d2d61710c14d6cdc0ab20d6dfc5 Mon Sep 17 00:00:00 2001 +From 6d4761010ea4dc218a1623513f410fc2d1cfc832 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:37:04 +0100 -Subject: [PATCH 26/31] hw/usb/hcd-ohci: skip automatic zero-init of large +Subject: [PATCH 51/57] hw/usb/hcd-ohci: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [25/30] a69a9e0467a1b695b7fa07cdf30733b709867bd2 +RH-Commit: [25/30] 721dd97d384fb755c4a6a00cfc3d867e43f25b0b (stefanha/centos-stream-qemu-kvm) The 'ohci_service_iso_td' method has a 8k byte array used for copying data between guest and host. Skip the automatic zero-init of this diff --git a/SOURCES/kvm-hw-vfio-common-Add-a-trace-point-in-vfio_reset_handl.patch b/SOURCES/kvm-hw-vfio-common-Add-a-trace-point-in-vfio_reset_handl.patch new file mode 100644 index 0000000..0c06398 --- /dev/null +++ b/SOURCES/kvm-hw-vfio-common-Add-a-trace-point-in-vfio_reset_handl.patch @@ -0,0 +1,61 @@ +From 04f11749dd21b4df1ea2818785d650dd6eee2cbe Mon Sep 17 00:00:00 2001 +From: Eric Auger +Date: Tue, 18 Feb 2025 19:25:34 +0100 +Subject: [PATCH 4/9] hw/vfio/common: Add a trace point in vfio_reset_handler +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Eric Auger +RH-MergeRequest: 341: Fix vIOMMU reset order +RH-Jira: RHEL-7188 +RH-Acked-by: Peter Xu +RH-Acked-by: Donald Dutile +RH-Acked-by: Cédric Le Goater +RH-Commit: [4/5] 46878ffdc96997d1f6d09bde3fce350564e499fd (eauger1/centos-qemu-kvm) + +To ease the debug of reset sequence, let's add a trace point +in vfio_reset_handler() + +Signed-off-by: Eric Auger +Reviewed-by: Cédric Le Goater +Acked-by: Michael S. Tsirkin +Reviewed-by: Zhenzhong Duan +Message-Id: <20250218182737.76722-5-eric.auger@redhat.com> +Reviewed-by: Peter Xu +Reviewed-by: Michael S. Tsirkin +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit d410e709526d1cd4aa9085c6e254a622594a02a5) +Signed-off-by: Eric Auger +--- + hw/vfio/common.c | 1 + + hw/vfio/trace-events | 1 + + 2 files changed, 2 insertions(+) + +diff --git a/hw/vfio/common.c b/hw/vfio/common.c +index 36d0cf6585..6982f88fc8 100644 +--- a/hw/vfio/common.c ++++ b/hw/vfio/common.c +@@ -1395,6 +1395,7 @@ void vfio_reset_handler(void *opaque) + { + VFIODevice *vbasedev; + ++ trace_vfio_reset_handler(); + QLIST_FOREACH(vbasedev, &vfio_device_list, global_next) { + if (vbasedev->dev->realized) { + vbasedev->ops->vfio_compute_needs_reset(vbasedev); +diff --git a/hw/vfio/trace-events b/hw/vfio/trace-events +index 3756ff660e..9523a9ccb0 100644 +--- a/hw/vfio/trace-events ++++ b/hw/vfio/trace-events +@@ -120,6 +120,7 @@ vfio_get_dev_region(const char *name, int index, uint32_t type, uint32_t subtype + vfio_legacy_dma_unmap_overflow_workaround(void) "" + vfio_get_dirty_bitmap(uint64_t iova, uint64_t size, uint64_t bitmap_size, uint64_t start, uint64_t dirty_pages) "iova=0x%"PRIx64" size= 0x%"PRIx64" bitmap_size=0x%"PRIx64" start=0x%"PRIx64" dirty_pages=%"PRIu64 + vfio_iommu_map_dirty_notify(uint64_t iova_start, uint64_t iova_end) "iommu dirty @ 0x%"PRIx64" - 0x%"PRIx64 ++vfio_reset_handler(void) "" + + # platform.c + vfio_platform_realize(char *name, char *compat) "vfio device %s, compat = %s" +-- +2.48.1 + diff --git a/SOURCES/kvm-hw-vfio-pci-Re-order-pre-reset.patch b/SOURCES/kvm-hw-vfio-pci-Re-order-pre-reset.patch new file mode 100644 index 0000000..7318f84 --- /dev/null +++ b/SOURCES/kvm-hw-vfio-pci-Re-order-pre-reset.patch @@ -0,0 +1,74 @@ +From d6a961077e753b9ad5a670a1529634fe20322ce2 Mon Sep 17 00:00:00 2001 +From: Alex Williamson +Date: Tue, 25 Feb 2025 14:52:29 -0700 +Subject: [PATCH 6/7] hw/vfio/pci: Re-order pre-reset +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Eric Auger +RH-MergeRequest: 348: PCI: Implement basic PCI PM capability backing +RH-Jira: RHEL-7301 +RH-Acked-by: Cédric Le Goater +RH-Acked-by: Alex Williamson +RH-Acked-by: Jon Maloy +RH-Commit: [6/6] c6c386ecbabda93f8a79da926ece95c2195fbc36 (eauger1/centos-qemu-kvm) + +We want the device in the D0 power state going into reset, but the +config write can enable the BARs in the address space, which are +then removed from the address space once we clear the memory enable +bit in the command register. Re-order to clear the command bit +first, so the power state change doesn't enable the BARs. + +Cc: Cédric Le Goater +Reviewed-by: Zhenzhong Duan +Reviewed-by: Eric Auger +Signed-off-by: Alex Williamson +Reviewed-by: Michael S. Tsirkin +Link: https://lore.kernel.org/qemu-devel/20250225215237.3314011-6-alex.williamson@redhat.com +Signed-off-by: Cédric Le Goater +(cherry picked from commit 518a69a598916749338de3852d41d961d4503115) +Signed-off-by: Eric Auger +--- + hw/vfio/pci.c | 18 +++++++++--------- + 1 file changed, 9 insertions(+), 9 deletions(-) + +diff --git a/hw/vfio/pci.c b/hw/vfio/pci.c +index 595b5c9b25..ffe72fd1d0 100644 +--- a/hw/vfio/pci.c ++++ b/hw/vfio/pci.c +@@ -2414,6 +2414,15 @@ void vfio_pci_pre_reset(VFIOPCIDevice *vdev) + + vfio_disable_interrupts(vdev); + ++ /* ++ * Stop any ongoing DMA by disconnecting I/O, MMIO, and bus master. ++ * Also put INTx Disable in known state. ++ */ ++ cmd = vfio_pci_read_config(pdev, PCI_COMMAND, 2); ++ cmd &= ~(PCI_COMMAND_IO | PCI_COMMAND_MEMORY | PCI_COMMAND_MASTER | ++ PCI_COMMAND_INTX_DISABLE); ++ vfio_pci_write_config(pdev, PCI_COMMAND, cmd, 2); ++ + /* Make sure the device is in D0 */ + if (pdev->pm_cap) { + uint16_t pmcsr; +@@ -2433,15 +2442,6 @@ void vfio_pci_pre_reset(VFIOPCIDevice *vdev) + } + } + } +- +- /* +- * Stop any ongoing DMA by disconnecting I/O, MMIO, and bus master. +- * Also put INTx Disable in known state. +- */ +- cmd = vfio_pci_read_config(pdev, PCI_COMMAND, 2); +- cmd &= ~(PCI_COMMAND_IO | PCI_COMMAND_MEMORY | PCI_COMMAND_MASTER | +- PCI_COMMAND_INTX_DISABLE); +- vfio_pci_write_config(pdev, PCI_COMMAND, cmd, 2); + } + + void vfio_pci_post_reset(VFIOPCIDevice *vdev) +-- +2.48.1 + diff --git a/SOURCES/kvm-hw-virtio-Also-include-md-stubs-in-case-CONFIG_VIRTI.patch b/SOURCES/kvm-hw-virtio-Also-include-md-stubs-in-case-CONFIG_VIRTI.patch new file mode 100644 index 0000000..c062e65 --- /dev/null +++ b/SOURCES/kvm-hw-virtio-Also-include-md-stubs-in-case-CONFIG_VIRTI.patch @@ -0,0 +1,59 @@ +From afa3a488f3ca52a5455987e4cd643882c4b15d8a Mon Sep 17 00:00:00 2001 +From: Thomas Huth +Date: Thu, 13 Mar 2025 07:35:22 +0100 +Subject: [PATCH 24/26] hw/virtio: Also include md stubs in case + CONFIG_VIRTIO_PCI is not set +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [24/26] ae6307b26d01d2a317f7e5d1d3b3a16b6d5f56de (thuth/qemu-kvm-cs) + +For the s390x target, it's possible to build the QEMU binary without +CONFIG_VIRTIO_PCI and only have the virtio-mem device via the ccw +transport. In that case, QEMU currently fails to link correctly: + + /usr/bin/ld: libqemu-s390x-softmmu.a.p/hw_s390x_s390-virtio-ccw.c.o: in function `s390_machine_device_pre_plug': + ../hw/s390x/s390-virtio-ccw.c:579:(.text+0x1e96): undefined reference to `virtio_md_pci_pre_plug' + /usr/bin/ld: libqemu-s390x-softmmu.a.p/hw_s390x_s390-virtio-ccw.c.o: in function `s390_machine_device_plug': + ../hw/s390x/s390-virtio-ccw.c:608:(.text+0x21a4): undefined reference to `virtio_md_pci_plug' + /usr/bin/ld: libqemu-s390x-softmmu.a.p/hw_s390x_s390-virtio-ccw.c.o: in function `s390_machine_device_unplug_request': + ../hw/s390x/s390-virtio-ccw.c:622:(.text+0x2334): undefined reference to `virtio_md_pci_unplug_request' + /usr/bin/ld: libqemu-s390x-softmmu.a.p/hw_s390x_s390-virtio-ccw.c.o: in function `s390_machine_device_unplug': + ../hw/s390x/s390-virtio-ccw.c:633:(.text+0x2436): undefined reference to `virtio_md_pci_unplug' + clang: error: linker command failed with exit code 1 (use -v to see invocation) + +We also need to include the stubs when CONFIG_VIRTIO_PCI is missing. + +Fixes: aa910c20ec5 ("s390x: virtio-mem support") +Message-ID: <20250313063522.1348288-1-thuth@redhat.com> +Reviewed-by: Philippe Mathieu-Daudé +Signed-off-by: Thomas Huth +(cherry picked from commit c1a6bff276ca52ffde472532d92bb5bb122dab3f) +Signed-off-by: Thomas Huth +--- + hw/virtio/meson.build | 3 ++- + 1 file changed, 2 insertions(+), 1 deletion(-) + +diff --git a/hw/virtio/meson.build b/hw/virtio/meson.build +index c38bdd6fa4..e2f9c75625 100644 +--- a/hw/virtio/meson.build ++++ b/hw/virtio/meson.build +@@ -89,7 +89,8 @@ specific_virtio_ss.add_all(when: 'CONFIG_VIRTIO_PCI', if_true: virtio_pci_ss) + system_ss.add_all(when: 'CONFIG_VIRTIO', if_true: system_virtio_ss) + system_ss.add(when: 'CONFIG_VIRTIO', if_false: files('vhost-stub.c')) + system_ss.add(when: 'CONFIG_VIRTIO', if_false: files('virtio-stub.c')) +-system_ss.add(when: 'CONFIG_VIRTIO_MD', if_false: files('virtio-md-stubs.c')) ++system_ss.add(when: ['CONFIG_VIRTIO_MD', 'CONFIG_VIRTIO_PCI'], ++ if_false: files('virtio-md-stubs.c')) + + system_ss.add(files('virtio-hmp-cmds.c')) + +-- +2.48.1 + diff --git a/SOURCES/kvm-hw-virtio-virtio-avoid-cost-of-ftrivial-auto-var-ini.patch b/SOURCES/kvm-hw-virtio-virtio-avoid-cost-of-ftrivial-auto-var-ini.patch index 631abc7..e006e88 100644 --- a/SOURCES/kvm-hw-virtio-virtio-avoid-cost-of-ftrivial-auto-var-ini.patch +++ b/SOURCES/kvm-hw-virtio-virtio-avoid-cost-of-ftrivial-auto-var-ini.patch @@ -1,17 +1,17 @@ -From ea242d728ed1716602b4cdd01d3fabb5ab260781 Mon Sep 17 00:00:00 2001 +From 4727c044a09fb8c4fb6d667f26eb55bb6de7554d Mon Sep 17 00:00:00 2001 From: Stefan Hajnoczi Date: Tue, 10 Jun 2025 13:36:40 +0100 -Subject: [PATCH 03/31] hw/virtio/virtio: avoid cost of -ftrivial-auto-var-init +Subject: [PATCH 28/57] hw/virtio/virtio: avoid cost of -ftrivial-auto-var-init in hot path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [2/30] a90fb4e14c182ace7a28c8335858895b4257f37b +RH-Commit: [2/30] 1c2cc6292deaaac068f4514439703c22c9ccb300 (stefanha/centos-stream-qemu-kvm) Since commit 7ff9ff039380 ("meson: mitigate against use of uninitialize stack for exploits") the -ftrivial-auto-var-init=zero compiler option is diff --git a/SOURCES/kvm-hw-virtio-virtio-iommu-Migrate-to-3-phase-reset.patch b/SOURCES/kvm-hw-virtio-virtio-iommu-Migrate-to-3-phase-reset.patch new file mode 100644 index 0000000..3922e9b --- /dev/null +++ b/SOURCES/kvm-hw-virtio-virtio-iommu-Migrate-to-3-phase-reset.patch @@ -0,0 +1,96 @@ +From 9ca5d7ac4f0ff5f10bf424df8104fe5abe01e431 Mon Sep 17 00:00:00 2001 +From: Eric Auger +Date: Tue, 18 Feb 2025 19:25:31 +0100 +Subject: [PATCH 1/9] hw/virtio/virtio-iommu: Migrate to 3-phase reset +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Eric Auger +RH-MergeRequest: 341: Fix vIOMMU reset order +RH-Jira: RHEL-7188 +RH-Acked-by: Peter Xu +RH-Acked-by: Donald Dutile +RH-Acked-by: Cédric Le Goater +RH-Commit: [1/5] 32bf47497d5d4817a448d07ffa7a844aee82ae3c (eauger1/centos-qemu-kvm) + +Currently the iommu may be reset before the devices +it protects. For example this happens with virtio-net. + +Let's use 3-phase reset mechanism and reset the IOMMU on +exit phase after all DMA capable devices have been +reset during the 'enter' or 'hold' phase. + +Signed-off-by: Eric Auger +Acked-by: Michael S. Tsirkin +Reviewed-by: Zhenzhong Duan +Acked-by: Jason Wang + +Message-Id: <20250218182737.76722-2-eric.auger@redhat.com> +Reviewed-by: Peter Xu +Reviewed-by: Michael S. Tsirkin +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit d261b84d354a41a38336af813f92f636d3fb3f78) +Signed-off-by: Eric Auger +--- + hw/virtio/trace-events | 2 +- + hw/virtio/virtio-iommu.c | 14 ++++++++++---- + 2 files changed, 11 insertions(+), 5 deletions(-) + +diff --git a/hw/virtio/trace-events b/hw/virtio/trace-events +index 04e36ae047..76f0d458b2 100644 +--- a/hw/virtio/trace-events ++++ b/hw/virtio/trace-events +@@ -108,7 +108,7 @@ virtio_pci_notify_write(uint64_t addr, uint64_t val, unsigned int size) "0x%" PR + virtio_pci_notify_write_pio(uint64_t addr, uint64_t val, unsigned int size) "0x%" PRIx64" = 0x%" PRIx64 " (%d)" + + # hw/virtio/virtio-iommu.c +-virtio_iommu_device_reset(void) "reset!" ++virtio_iommu_device_reset_exit(void) "reset!" + virtio_iommu_system_reset(void) "system reset!" + virtio_iommu_get_features(uint64_t features) "device supports features=0x%"PRIx64 + virtio_iommu_device_status(uint8_t status) "driver status = %d" +diff --git a/hw/virtio/virtio-iommu.c b/hw/virtio/virtio-iommu.c +index 59ef4fb217..496200ebc5 100644 +--- a/hw/virtio/virtio-iommu.c ++++ b/hw/virtio/virtio-iommu.c +@@ -1504,11 +1504,11 @@ static void virtio_iommu_device_unrealize(DeviceState *dev) + virtio_cleanup(vdev); + } + +-static void virtio_iommu_device_reset(VirtIODevice *vdev) ++static void virtio_iommu_device_reset_exit(Object *obj, ResetType type) + { +- VirtIOIOMMU *s = VIRTIO_IOMMU(vdev); ++ VirtIOIOMMU *s = VIRTIO_IOMMU(obj); + +- trace_virtio_iommu_device_reset(); ++ trace_virtio_iommu_device_reset_exit(); + + if (s->domains) { + g_tree_destroy(s->domains); +@@ -1669,6 +1669,7 @@ static void virtio_iommu_class_init(ObjectClass *klass, void *data) + { + DeviceClass *dc = DEVICE_CLASS(klass); + VirtioDeviceClass *vdc = VIRTIO_DEVICE_CLASS(klass); ++ ResettableClass *rc = RESETTABLE_CLASS(klass); + + device_class_set_props(dc, virtio_iommu_properties); + dc->vmsd = &vmstate_virtio_iommu; +@@ -1676,7 +1677,12 @@ static void virtio_iommu_class_init(ObjectClass *klass, void *data) + set_bit(DEVICE_CATEGORY_MISC, dc->categories); + vdc->realize = virtio_iommu_device_realize; + vdc->unrealize = virtio_iommu_device_unrealize; +- vdc->reset = virtio_iommu_device_reset; ++ ++ /* ++ * Use 'exit' reset phase to make sure all DMA requests ++ * have been quiesced during 'enter' or 'hold' phase ++ */ ++ rc->phases.exit = virtio_iommu_device_reset_exit; + vdc->get_config = virtio_iommu_get_config; + vdc->set_config = virtio_iommu_set_config; + vdc->get_features = virtio_iommu_get_features; +-- +2.48.1 + diff --git a/SOURCES/kvm-i386-Introduce-tdx-guest-object.patch b/SOURCES/kvm-i386-Introduce-tdx-guest-object.patch new file mode 100644 index 0000000..4dccdee --- /dev/null +++ b/SOURCES/kvm-i386-Introduce-tdx-guest-object.patch @@ -0,0 +1,214 @@ +From dc14d1444d4ad525663848160cd7687ef291c85e Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:45 +0200 +Subject: [PATCH 031/115] i386: Introduce tdx-guest object +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [31/115] f279a3b477fbbe84bb5d6c3a8eb588916b41128e (bonzini/rhel-qemu-kvm) + +Introduce tdx-guest object which inherits X86_CONFIDENTIAL_GUEST, +and will be used to create TDX VMs (TDs) by + + qemu -machine ...,confidential-guest-support=tdx0 \ + -object tdx-guest,id=tdx0 + +It has one QAPI member 'attributes' defined, which allows user to set +TD's attributes directly. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Acked-by: Markus Armbruster +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-3-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 756e12e791771034ac105a5d2c9887bbbb6b7c73) +Signed-off-by: Paolo Bonzini + +Conflict: class_init's second argument is not const +--- + configs/devices/i386-softmmu/default.mak | 1 + + hw/i386/Kconfig | 5 +++ + qapi/qom.json | 15 +++++++++ + target/i386/kvm/meson.build | 2 ++ + target/i386/kvm/tdx.c | 43 ++++++++++++++++++++++++ + target/i386/kvm/tdx.h | 21 ++++++++++++ + 6 files changed, 87 insertions(+) + create mode 100644 target/i386/kvm/tdx.c + create mode 100644 target/i386/kvm/tdx.h + +diff --git a/configs/devices/i386-softmmu/default.mak b/configs/devices/i386-softmmu/default.mak +index 448e3e3b1b..34c21224eb 100644 +--- a/configs/devices/i386-softmmu/default.mak ++++ b/configs/devices/i386-softmmu/default.mak +@@ -18,6 +18,7 @@ + #CONFIG_QXL=n + #CONFIG_SEV=n + #CONFIG_SGA=n ++#CONFIG_TDX=n + #CONFIG_TEST_DEVICES=n + #CONFIG_TPM_CRB=n + #CONFIG_TPM_TIS_ISA=n +diff --git a/hw/i386/Kconfig b/hw/i386/Kconfig +index f4a33b6c08..edd61cd2aa 100644 +--- a/hw/i386/Kconfig ++++ b/hw/i386/Kconfig +@@ -10,6 +10,10 @@ config SGX + bool + depends on KVM + ++config TDX ++ bool ++ depends on KVM ++ + config PC + bool + imply APPLESMC +@@ -26,6 +30,7 @@ config PC + imply QXL + imply SEV + imply SGX ++ imply TDX + imply TEST_DEVICES + imply TPM_CRB + imply TPM_TIS_ISA +diff --git a/qapi/qom.json b/qapi/qom.json +index 321ccd708a..530efeb7c5 100644 +--- a/qapi/qom.json ++++ b/qapi/qom.json +@@ -1008,6 +1008,19 @@ + '*host-data': 'str', + '*vcek-disabled': 'bool' } } + ++## ++# @TdxGuestProperties: ++# ++# Properties for tdx-guest objects. ++# ++# @attributes: The 'attributes' of a TD guest that is passed to ++# KVM_TDX_INIT_VM ++# ++# Since: 10.1 ++## ++{ 'struct': 'TdxGuestProperties', ++ 'data': { '*attributes': 'uint64' } } ++ + ## + # @ThreadContextProperties: + # +@@ -1092,6 +1105,7 @@ + 'sev-snp-guest', + 'thread-context', + 's390-pv-guest', ++ 'tdx-guest', + 'throttle-group', + 'tls-creds-anon', + 'tls-creds-psk', +@@ -1163,6 +1177,7 @@ + 'if': 'CONFIG_SECRET_KEYRING' }, + 'sev-guest': 'SevGuestProperties', + 'sev-snp-guest': 'SevSnpGuestProperties', ++ 'tdx-guest': 'TdxGuestProperties', + 'thread-context': 'ThreadContextProperties', + 'throttle-group': 'ThrottleGroupProperties', + 'tls-creds-anon': 'TlsCredsAnonProperties', +diff --git a/target/i386/kvm/meson.build b/target/i386/kvm/meson.build +index 3996cafaf2..466bccb9cb 100644 +--- a/target/i386/kvm/meson.build ++++ b/target/i386/kvm/meson.build +@@ -8,6 +8,8 @@ i386_kvm_ss.add(files( + + i386_kvm_ss.add(when: 'CONFIG_XEN_EMU', if_true: files('xen-emu.c')) + ++i386_kvm_ss.add(when: 'CONFIG_TDX', if_true: files('tdx.c')) ++ + i386_system_ss.add(when: 'CONFIG_HYPERV', if_true: files('hyperv.c'), if_false: files('hyperv-stub.c')) + + i386_system_ss.add_all(when: 'CONFIG_KVM', if_true: i386_kvm_ss) +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +new file mode 100644 +index 0000000000..ec84ae2947 +--- /dev/null ++++ b/target/i386/kvm/tdx.c +@@ -0,0 +1,43 @@ ++/* ++ * QEMU TDX support ++ * ++ * Copyright (c) 2025 Intel Corporation ++ * ++ * Author: ++ * Xiaoyao Li ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#include "qemu/osdep.h" ++#include "qom/object_interfaces.h" ++ ++#include "tdx.h" ++ ++/* tdx guest */ ++OBJECT_DEFINE_TYPE_WITH_INTERFACES(TdxGuest, ++ tdx_guest, ++ TDX_GUEST, ++ X86_CONFIDENTIAL_GUEST, ++ { TYPE_USER_CREATABLE }, ++ { NULL }) ++ ++static void tdx_guest_init(Object *obj) ++{ ++ ConfidentialGuestSupport *cgs = CONFIDENTIAL_GUEST_SUPPORT(obj); ++ TdxGuest *tdx = TDX_GUEST(obj); ++ ++ cgs->require_guest_memfd = true; ++ tdx->attributes = 0; ++ ++ object_property_add_uint64_ptr(obj, "attributes", &tdx->attributes, ++ OBJ_PROP_FLAG_READWRITE); ++} ++ ++static void tdx_guest_finalize(Object *obj) ++{ ++} ++ ++static void tdx_guest_class_init(ObjectClass *oc, void *data) ++{ ++} +diff --git a/target/i386/kvm/tdx.h b/target/i386/kvm/tdx.h +new file mode 100644 +index 0000000000..f3b7253361 +--- /dev/null ++++ b/target/i386/kvm/tdx.h +@@ -0,0 +1,21 @@ ++/* SPDX-License-Identifier: GPL-2.0-or-later */ ++ ++#ifndef QEMU_I386_TDX_H ++#define QEMU_I386_TDX_H ++ ++#include "confidential-guest.h" ++ ++#define TYPE_TDX_GUEST "tdx-guest" ++#define TDX_GUEST(obj) OBJECT_CHECK(TdxGuest, (obj), TYPE_TDX_GUEST) ++ ++typedef struct TdxGuestClass { ++ X86ConfidentialGuestClass parent_class; ++} TdxGuestClass; ++ ++typedef struct TdxGuest { ++ X86ConfidentialGuest parent_obj; ++ ++ uint64_t attributes; /* TD attributes */ ++} TdxGuest; ++ ++#endif /* QEMU_I386_TDX_H */ +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-Remove-unused-parameter-uint32_t-bit-in-feature.patch b/SOURCES/kvm-i386-Remove-unused-parameter-uint32_t-bit-in-feature.patch new file mode 100644 index 0000000..30aae17 --- /dev/null +++ b/SOURCES/kvm-i386-Remove-unused-parameter-uint32_t-bit-in-feature.patch @@ -0,0 +1,62 @@ +From 9d654537f0f667a36eb45d80fda283b31ace3d39 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 017/115] i386: Remove unused parameter "uint32_t bit" in + feature_word_description() + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [17/115] 37ee215225ee0757e51718df4548d4a43fe09e99 (bonzini/rhel-qemu-kvm) + +Parameter "uint32_t bit" is not used in function feature_word_description(), +so remove it. + +Signed-off-by: Lei Wang +Reviewed-by: Igor Mammedov +Reviewed-by: Xiaoyao Li +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Message-ID: <20241217123932.948789-2-xiaoyao.li@intel.com> +Signed-off-by: Paolo Bonzini +(cherry picked from commit bab32b8b4bf9da5d13386c8faa5a9389e63244b7) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 4 ++-- + 1 file changed, 2 insertions(+), 2 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index a97d042a2e..32e89f1a5c 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -5897,7 +5897,7 @@ static const TypeInfo max_x86_cpu_type_info = { + .class_init = max_x86_cpu_class_init, + }; + +-static char *feature_word_description(FeatureWordInfo *f, uint32_t bit) ++static char *feature_word_description(FeatureWordInfo *f) + { + assert(f->type == CPUID_FEATURE_WORD || f->type == MSR_FEATURE_WORD); + +@@ -5936,6 +5936,7 @@ static void mark_unavailable_features(X86CPU *cpu, FeatureWord w, uint64_t mask, + CPUX86State *env = &cpu->env; + FeatureWordInfo *f = &feature_word_info[w]; + int i; ++ g_autofree char *feat_word_str = feature_word_description(f); + + if (!cpu->force_features) { + env->features[w] &= ~mask; +@@ -5948,7 +5949,6 @@ static void mark_unavailable_features(X86CPU *cpu, FeatureWord w, uint64_t mask, + + for (i = 0; i < 64; ++i) { + if ((1ULL << i) & mask) { +- g_autofree char *feat_word_str = feature_word_description(f, i); + warn_report("%s: %s%s%s [bit %d]", + verbose_prefix, + feat_word_str, +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-apic-Skip-kvm_apic_put-for-TDX.patch b/SOURCES/kvm-i386-apic-Skip-kvm_apic_put-for-TDX.patch new file mode 100644 index 0000000..dec6bca --- /dev/null +++ b/SOURCES/kvm-i386-apic-Skip-kvm_apic_put-for-TDX.patch @@ -0,0 +1,63 @@ +From f5b6984efa1bf825410011b957b4f46fcfe963db Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 070/115] i386/apic: Skip kvm_apic_put() for TDX + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [70/115] d4e1631ebceb6441a608ff92c1964b64cf116094 (bonzini/rhel-qemu-kvm) + +KVM neithers allow writing to MSR_IA32_APICBASE for TDs, nor allow for +KVM_SET_LAPIC[*]. + +Note, KVM_GET_LAPIC is also disallowed for TDX. It is called in the path + + do_kvm_cpu_synchronize_state() + -> kvm_arch_get_registers() + -> kvm_get_apic() + +and it's already disllowed for confidential guest through +guest_state_protected. + +[*] https://lore.kernel.org/all/Z3w4Ku4Jq0CrtXne@google.com/ + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-42-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 62a1a8b89d90cd3fbee0e6d38e6a4c0d833e978a) +Signed-off-by: Paolo Bonzini +--- + hw/i386/kvm/apic.c | 5 +++++ + 1 file changed, 5 insertions(+) + +diff --git a/hw/i386/kvm/apic.c b/hw/i386/kvm/apic.c +index a72c28e8a7..9c12a9c856 100644 +--- a/hw/i386/kvm/apic.c ++++ b/hw/i386/kvm/apic.c +@@ -17,6 +17,7 @@ + #include "sysemu/hw_accel.h" + #include "sysemu/kvm.h" + #include "kvm/kvm_i386.h" ++#include "kvm/tdx.h" + + static inline void kvm_apic_set_reg(struct kvm_lapic_state *kapic, + int reg_id, uint32_t val) +@@ -141,6 +142,10 @@ static void kvm_apic_put(CPUState *cs, run_on_cpu_data data) + struct kvm_lapic_state kapic; + int ret; + ++ if (is_tdx_vm()) { ++ return; ++ } ++ + kvm_put_apicbase(s->cpu, s->apicbase); + kvm_put_apic_state(s, &kapic); + +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cgs-Introduce-x86_confidential_guest_check_feat.patch b/SOURCES/kvm-i386-cgs-Introduce-x86_confidential_guest_check_feat.patch new file mode 100644 index 0000000..7d75d73 --- /dev/null +++ b/SOURCES/kvm-i386-cgs-Introduce-x86_confidential_guest_check_feat.patch @@ -0,0 +1,80 @@ +From 0d8993cabc26807ef973630f38ec2b09557497fe Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 079/115] i386/cgs: Introduce + x86_confidential_guest_check_features() + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [79/115] 1fce5742b7746e6ba589c486fb1a6aec8ab8391a (bonzini/rhel-qemu-kvm) + +To do cgs specific feature checking. Note the feature checking in +x86_cpu_filter_features() is valid for non-cgs VMs. For cgs VMs like +TDX, what features can be supported has more restrictions. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-51-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit dc0b08b303ad34983b43936a4c978672e0f9a9d8) +Signed-off-by: Paolo Bonzini +--- + target/i386/confidential-guest.h | 13 +++++++++++++ + target/i386/kvm/kvm.c | 8 ++++++++ + 2 files changed, 21 insertions(+) + +diff --git a/target/i386/confidential-guest.h b/target/i386/confidential-guest.h +index 8a5cc7ecff..4e7eb43416 100644 +--- a/target/i386/confidential-guest.h ++++ b/target/i386/confidential-guest.h +@@ -42,6 +42,7 @@ struct X86ConfidentialGuestClass { + void (*cpu_instance_init)(X86ConfidentialGuest *cg, CPUState *cpu); + uint32_t (*adjust_cpuid_features)(X86ConfidentialGuest *cg, uint32_t feature, + uint32_t index, int reg, uint32_t value); ++ int (*check_features)(X86ConfidentialGuest *cg, CPUState *cs); + }; + + /** +@@ -91,4 +92,16 @@ static inline int x86_confidential_guest_adjust_cpuid_features(X86ConfidentialGu + } + } + ++static inline int x86_confidential_guest_check_features(X86ConfidentialGuest *cg, ++ CPUState *cs) ++{ ++ X86ConfidentialGuestClass *klass = X86_CONFIDENTIAL_GUEST_GET_CLASS(cg); ++ ++ if (klass->check_features) { ++ return klass->check_features(cg, cs); ++ } ++ ++ return 0; ++} ++ + #endif +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index 76352323e4..b6fddcd543 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -2081,6 +2081,14 @@ int kvm_arch_init_vcpu(CPUState *cs) + int r; + Error *local_err = NULL; + ++ if (current_machine->cgs) { ++ r = x86_confidential_guest_check_features( ++ X86_CONFIDENTIAL_GUEST(current_machine->cgs), cs); ++ if (r < 0) { ++ return r; ++ } ++ } ++ + memset(&cpuid_data, 0, sizeof(cpuid_data)); + + cpuid_i = 0; +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cgs-Rename-mask_cpuid_features-to-adjust_cpuid_.patch b/SOURCES/kvm-i386-cgs-Rename-mask_cpuid_features-to-adjust_cpuid_.patch new file mode 100644 index 0000000..2bb8084 --- /dev/null +++ b/SOURCES/kvm-i386-cgs-Rename-mask_cpuid_features-to-adjust_cpuid_.patch @@ -0,0 +1,116 @@ +From fa35367ae78505390b5915c9bf96542ffed1787d Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 072/115] i386/cgs: Rename *mask_cpuid_features() to + *adjust_cpuid_features() +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [72/115] 0633336351c17f73620bc7d13cee5ba53b100e13 (bonzini/rhel-qemu-kvm) + +Because for TDX case, there are also fixed-1 bits that enforced by TDX +module. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-44-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 695bfaee7153153708228946aa26c6d879599c04) +Signed-off-by: Paolo Bonzini +--- + target/i386/confidential-guest.h | 20 ++++++++++---------- + target/i386/kvm/kvm.c | 2 +- + target/i386/sev.c | 4 ++-- + 3 files changed, 13 insertions(+), 13 deletions(-) + +diff --git a/target/i386/confidential-guest.h b/target/i386/confidential-guest.h +index 38169ed68e..8a5cc7ecff 100644 +--- a/target/i386/confidential-guest.h ++++ b/target/i386/confidential-guest.h +@@ -40,8 +40,8 @@ struct X86ConfidentialGuestClass { + /* */ + int (*kvm_type)(X86ConfidentialGuest *cg); + void (*cpu_instance_init)(X86ConfidentialGuest *cg, CPUState *cpu); +- uint32_t (*mask_cpuid_features)(X86ConfidentialGuest *cg, uint32_t feature, uint32_t index, +- int reg, uint32_t value); ++ uint32_t (*adjust_cpuid_features)(X86ConfidentialGuest *cg, uint32_t feature, ++ uint32_t index, int reg, uint32_t value); + }; + + /** +@@ -71,21 +71,21 @@ static inline void x86_confidential_guest_cpu_instance_init(X86ConfidentialGuest + } + + /** +- * x86_confidential_guest_mask_cpuid_features: ++ * x86_confidential_guest_adjust_cpuid_features: + * +- * Removes unsupported features from a confidential guest's CPUID values, returns +- * the value with the bits removed. The bits removed should be those that KVM +- * provides independent of host-supported CPUID features, but are not supported by +- * the confidential computing firmware. ++ * Adjust the supported features from a confidential guest's CPUID values, ++ * returns the adjusted value. There are bits being removed that are not ++ * supported by the confidential computing firmware or bits being added that ++ * are forcibly exposed to guest by the confidential computing firmware. + */ +-static inline int x86_confidential_guest_mask_cpuid_features(X86ConfidentialGuest *cg, ++static inline int x86_confidential_guest_adjust_cpuid_features(X86ConfidentialGuest *cg, + uint32_t feature, uint32_t index, + int reg, uint32_t value) + { + X86ConfidentialGuestClass *klass = X86_CONFIDENTIAL_GUEST_GET_CLASS(cg); + +- if (klass->mask_cpuid_features) { +- return klass->mask_cpuid_features(cg, feature, index, reg, value); ++ if (klass->adjust_cpuid_features) { ++ return klass->adjust_cpuid_features(cg, feature, index, reg, value); + } else { + return value; + } +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index f3fe553151..5349ff4db7 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -565,7 +565,7 @@ uint32_t kvm_arch_get_supported_cpuid(KVMState *s, uint32_t function, + } + + if (current_machine->cgs) { +- ret = x86_confidential_guest_mask_cpuid_features( ++ ret = x86_confidential_guest_adjust_cpuid_features( + X86_CONFIDENTIAL_GUEST(current_machine->cgs), + function, index, reg, ret); + } +diff --git a/target/i386/sev.c b/target/i386/sev.c +index a0d271f898..24fcd078fc 100644 +--- a/target/i386/sev.c ++++ b/target/i386/sev.c +@@ -946,7 +946,7 @@ out: + } + + static uint32_t +-sev_snp_mask_cpuid_features(X86ConfidentialGuest *cg, uint32_t feature, uint32_t index, ++sev_snp_adjust_cpuid_features(X86ConfidentialGuest *cg, uint32_t feature, uint32_t index, + int reg, uint32_t value) + { + switch (feature) { +@@ -2404,7 +2404,7 @@ sev_snp_guest_class_init(ObjectClass *oc, void *data) + klass->launch_finish = sev_snp_launch_finish; + klass->launch_update_data = sev_snp_launch_update_data; + klass->kvm_init = sev_snp_kvm_init; +- x86_klass->mask_cpuid_features = sev_snp_mask_cpuid_features; ++ x86_klass->adjust_cpuid_features = sev_snp_adjust_cpuid_features; + x86_klass->kvm_type = sev_snp_kvm_type; + + object_class_property_add(oc, "policy", "uint64", +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-Cleanup-host_cpu_max_instance_init.patch b/SOURCES/kvm-i386-cpu-Cleanup-host_cpu_max_instance_init.patch new file mode 100644 index 0000000..b787654 --- /dev/null +++ b/SOURCES/kvm-i386-cpu-Cleanup-host_cpu_max_instance_init.patch @@ -0,0 +1,49 @@ +From 70ffde0038f36b6720b73136a4368b26f2bf6181 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:50 +0200 +Subject: [PATCH 107/115] i386/cpu: Cleanup host_cpu_max_instance_init() +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [107/115] 140ba66d2ff544b6ae498798e1f6ad3ade1791bc (bonzini/rhel-qemu-kvm) + +The implementation of host_cpu_max_instance_init() was merged into +host_cpu_instance_init() by commit 29f1ba338baf ("target/i386: merge +host_cpu_instance_init() and host_cpu_max_instance_init()"), while the +declaration of it remains in host-cpu.h. + +Clean it up. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Philippe Mathieu-Daudé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250716063117.602050-1-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 5fe6b9a854a91df86fdb794cbeb67d0656756137) +Signed-off-by: Paolo Bonzini +--- + target/i386/host-cpu.h | 1 - + 1 file changed, 1 deletion(-) + +diff --git a/target/i386/host-cpu.h b/target/i386/host-cpu.h +index b97ec01c9b..5b2ad491a8 100644 +--- a/target/i386/host-cpu.h ++++ b/target/i386/host-cpu.h +@@ -12,7 +12,6 @@ + + uint32_t host_cpu_phys_bits(void); + void host_cpu_instance_init(X86CPU *cpu); +-void host_cpu_max_instance_init(X86CPU *cpu); + bool host_cpu_realizefn(CPUState *cs, Error **errp); + + void host_cpu_vendor_fms(char *vendor, int *family, int *model, int *stepping); +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-Consolidate-the-helper-to-get-Host-s-vendor.patch b/SOURCES/kvm-i386-cpu-Consolidate-the-helper-to-get-Host-s-vendor.patch new file mode 100644 index 0000000..d566885 --- /dev/null +++ b/SOURCES/kvm-i386-cpu-Consolidate-the-helper-to-get-Host-s-vendor.patch @@ -0,0 +1,78 @@ +From 0a568fbac880efdda740808e6fbbd8be09cfb46a Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:45 +0200 +Subject: [PATCH 024/115] i386/cpu: Consolidate the helper to get Host's vendor + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [24/115] cd823d475dc0a0ec4135f0c9d5546db996579d30 (bonzini/rhel-qemu-kvm) + +Extend host_cpu_vendor_fms() to help more cases to get Host's vendor +information. + +Cc: Dongli Zhang +Signed-off-by: Zhao Liu +Link: https://lore.kernel.org/r/20250410075619.145792-1-zhao1.liu@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit ae39acef49e29169f90cd3a799d6cd0b50bc65d2) +Signed-off-by: Paolo Bonzini +--- + target/i386/host-cpu.c | 10 ++++++---- + target/i386/kvm/vmsr_energy.c | 3 +-- + 2 files changed, 7 insertions(+), 6 deletions(-) + +diff --git a/target/i386/host-cpu.c b/target/i386/host-cpu.c +index 03b9d1b169..4a77ecc1fc 100644 +--- a/target/i386/host-cpu.c ++++ b/target/i386/host-cpu.c +@@ -109,9 +109,13 @@ void host_cpu_vendor_fms(char *vendor, int *family, int *model, int *stepping) + { + uint32_t eax, ebx, ecx, edx; + +- host_cpuid(0x0, 0, &eax, &ebx, &ecx, &edx); ++ host_cpuid(0x0, 0, NULL, &ebx, &ecx, &edx); + x86_cpu_vendor_words2str(vendor, ebx, edx, ecx); + ++ if (!family && !model && !stepping) { ++ return; ++ } ++ + host_cpuid(0x1, 0, &eax, &ebx, &ecx, &edx); + if (family) { + *family = ((eax >> 8) & 0x0F) + ((eax >> 20) & 0xFF); +@@ -129,11 +133,9 @@ void host_cpu_instance_init(X86CPU *cpu) + X86CPUClass *xcc = X86_CPU_GET_CLASS(cpu); + + if (xcc->model) { +- uint32_t ebx = 0, ecx = 0, edx = 0; + char vendor[CPUID_VENDOR_SZ + 1]; + +- host_cpuid(0, 0, NULL, &ebx, &ecx, &edx); +- x86_cpu_vendor_words2str(vendor, ebx, edx, ecx); ++ host_cpu_vendor_fms(vendor, NULL, NULL, NULL); + object_property_set_str(OBJECT(cpu), "vendor", vendor, &error_abort); + } + } +diff --git a/target/i386/kvm/vmsr_energy.c b/target/i386/kvm/vmsr_energy.c +index 7e064c5aef..615f23b6cf 100644 +--- a/target/i386/kvm/vmsr_energy.c ++++ b/target/i386/kvm/vmsr_energy.c +@@ -29,10 +29,9 @@ char *vmsr_compute_default_paths(void) + + bool is_host_cpu_intel(void) + { +- int family, model, stepping; + char vendor[CPUID_VENDOR_SZ + 1]; + +- host_cpu_vendor_fms(vendor, &family, &model, &stepping); ++ host_cpu_vendor_fms(vendor, NULL, NULL, NULL); + + return strcmp(vendor, CPUID_VENDOR_INTEL); + } +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-Drop-cores_per_pkg-in-cpu_x86_cpuid.patch b/SOURCES/kvm-i386-cpu-Drop-cores_per_pkg-in-cpu_x86_cpuid.patch new file mode 100644 index 0000000..e7304fc --- /dev/null +++ b/SOURCES/kvm-i386-cpu-Drop-cores_per_pkg-in-cpu_x86_cpuid.patch @@ -0,0 +1,55 @@ +From da6c7cae87a945451617014a97c83b0e38879786 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 009/115] i386/cpu: Drop cores_per_pkg in cpu_x86_cpuid() + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [9/115] 1e8cca4b6784adde542654cd45b4f921fbc91fcd (bonzini/rhel-qemu-kvm) + +Local variable cores_per_pkg is only used to calculate threads_per_pkg. +No need for it. Drop it and open-code it instead. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20241219110125.1266461-4-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 00ec7be67c3981b486293aa8e0aef9534f229c5e) +Signed-off-by: Paolo Bonzini +(cherry picked from commit 589e1863d4e871e09af4ff176df97c23e2a33b8b) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 6 ++---- + 1 file changed, 2 insertions(+), 4 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 1fe492f33d..b639769ef3 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -6950,7 +6950,6 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + uint32_t limit; + uint32_t signature[3]; + X86CPUTopoInfo topo_info; +- uint32_t cores_per_pkg; + uint32_t threads_per_pkg; + + topo_info.dies_per_pkg = env->nr_dies; +@@ -6958,9 +6957,8 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + topo_info.cores_per_module = cs->nr_cores / env->nr_dies / env->nr_modules; + topo_info.threads_per_core = cs->nr_threads; + +- cores_per_pkg = topo_info.cores_per_module * topo_info.modules_per_die * +- topo_info.dies_per_pkg; +- threads_per_pkg = cores_per_pkg * topo_info.threads_per_core; ++ threads_per_pkg = topo_info.threads_per_core * topo_info.cores_per_module * ++ topo_info.modules_per_die * topo_info.dies_per_pkg; + + /* Calculate & apply limits for different index ranges */ + if (index >= 0xC0000000) { +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-Drop-the-check-of-phys_bits-in-host_cpu_rea.patch b/SOURCES/kvm-i386-cpu-Drop-the-check-of-phys_bits-in-host_cpu_rea.patch new file mode 100644 index 0000000..2809367 --- /dev/null +++ b/SOURCES/kvm-i386-cpu-Drop-the-check-of-phys_bits-in-host_cpu_rea.patch @@ -0,0 +1,79 @@ +From 8e732c71def28ac963a8f8d530c7dc063abc6d97 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 006/115] i386/cpu: Drop the check of phys_bits in + host_cpu_realizefn() + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [6/115] a53971f358752d7c850baee4f8f3c7912854f160 (bonzini/rhel-qemu-kvm) + +The check of cpu->phys_bits to be in range between +[32, TARGET_PHYS_ADDR_SPACE_BITS] in host_cpu_realizefn() +is duplicated with check in x86_cpu_realizefn(). + +Since the ckeck in x86_cpu_realizefn() is called later and can cover all +the x86 cases. Remove the one in host_cpu_realizefn(). + +Opportunistically adjust cpu->phys_bits directly in +host_cpu_adjust_phys_bits(), which matches more with the function name. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Igor Mammedov +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20240929085747.2023198-1-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 855bdb6c8a60ae20043531dc965fcb1ed171d7d9) +Signed-off-by: Paolo Bonzini +--- + target/i386/host-cpu.c | 16 +++------------- + 1 file changed, 3 insertions(+), 13 deletions(-) + +diff --git a/target/i386/host-cpu.c b/target/i386/host-cpu.c +index 8b8bf5afec..03b9d1b169 100644 +--- a/target/i386/host-cpu.c ++++ b/target/i386/host-cpu.c +@@ -42,7 +42,7 @@ static uint32_t host_cpu_phys_bits(void) + return host_phys_bits; + } + +-static uint32_t host_cpu_adjust_phys_bits(X86CPU *cpu) ++static void host_cpu_adjust_phys_bits(X86CPU *cpu) + { + uint32_t host_phys_bits = host_cpu_phys_bits(); + uint32_t phys_bits = cpu->phys_bits; +@@ -66,7 +66,7 @@ static uint32_t host_cpu_adjust_phys_bits(X86CPU *cpu) + } + } + +- return phys_bits; ++ cpu->phys_bits = phys_bits; + } + + bool host_cpu_realizefn(CPUState *cs, Error **errp) +@@ -75,17 +75,7 @@ bool host_cpu_realizefn(CPUState *cs, Error **errp) + CPUX86State *env = &cpu->env; + + if (env->features[FEAT_8000_0001_EDX] & CPUID_EXT2_LM) { +- uint32_t phys_bits = host_cpu_adjust_phys_bits(cpu); +- +- if (phys_bits && +- (phys_bits > TARGET_PHYS_ADDR_SPACE_BITS || +- phys_bits < 32)) { +- error_setg(errp, "phys-bits should be between 32 and %u " +- " (but is %u)", +- TARGET_PHYS_ADDR_SPACE_BITS, phys_bits); +- return false; +- } +- cpu->phys_bits = phys_bits; ++ host_cpu_adjust_phys_bits(cpu); + } + return true; + } +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-Drop-the-variable-smp_cores-and-smp_threads.patch b/SOURCES/kvm-i386-cpu-Drop-the-variable-smp_cores-and-smp_threads.patch new file mode 100644 index 0000000..1b658bc --- /dev/null +++ b/SOURCES/kvm-i386-cpu-Drop-the-variable-smp_cores-and-smp_threads.patch @@ -0,0 +1,68 @@ +From 368e0e988c60d0391c1e47db7e3f1aff7742f697 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 008/115] i386/cpu: Drop the variable smp_cores and smp_threads + in x86_cpu_pre_plug() + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [8/115] bc67c58c23402a05899559f8ecfa61218aa3ef2a (bonzini/rhel-qemu-kvm) + +No need to define smp_cores and smp_threads, just using ms->smp.cores +and ms->smp.threads is straightforward. It's also consistent with other +checks of socket/die/module. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20241219110125.1266461-3-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 81bd60625fc23cb8d4d0e682dcc4223d5e1ead84) +Signed-off-by: Paolo Bonzini +(cherry picked from commit c263000e490ddb361e0ef8a45342dea034036682) +Signed-off-by: Paolo Bonzini +--- + hw/i386/x86-common.c | 10 ++++------ + 1 file changed, 4 insertions(+), 6 deletions(-) + +diff --git a/hw/i386/x86-common.c b/hw/i386/x86-common.c +index 992ea1f25e..2806be98f3 100644 +--- a/hw/i386/x86-common.c ++++ b/hw/i386/x86-common.c +@@ -248,8 +248,6 @@ void x86_cpu_pre_plug(HotplugHandler *hotplug_dev, + CPUX86State *env = &cpu->env; + MachineState *ms = MACHINE(hotplug_dev); + X86MachineState *x86ms = X86_MACHINE(hotplug_dev); +- unsigned int smp_cores = ms->smp.cores; +- unsigned int smp_threads = ms->smp.threads; + X86CPUTopoInfo topo_info; + + if (!object_dynamic_cast(OBJECT(cpu), ms->cpu_type)) { +@@ -329,17 +327,17 @@ void x86_cpu_pre_plug(HotplugHandler *hotplug_dev, + if (cpu->core_id < 0) { + error_setg(errp, "CPU core-id is not set"); + return; +- } else if (cpu->core_id > (smp_cores - 1)) { ++ } else if (cpu->core_id > (ms->smp.cores - 1)) { + error_setg(errp, "Invalid CPU core-id: %u must be in range 0:%u", +- cpu->core_id, smp_cores - 1); ++ cpu->core_id, ms->smp.cores - 1); + return; + } + if (cpu->thread_id < 0) { + error_setg(errp, "CPU thread-id is not set"); + return; +- } else if (cpu->thread_id > (smp_threads - 1)) { ++ } else if (cpu->thread_id > (ms->smp.threads - 1)) { + error_setg(errp, "Invalid CPU thread-id: %u must be in range 0:%u", +- cpu->thread_id, smp_threads - 1); ++ cpu->thread_id, ms->smp.threads - 1); + return; + } + +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-Extract-a-common-fucntion-to-setup-value-of.patch b/SOURCES/kvm-i386-cpu-Extract-a-common-fucntion-to-setup-value-of.patch new file mode 100644 index 0000000..ff72ddf --- /dev/null +++ b/SOURCES/kvm-i386-cpu-Extract-a-common-fucntion-to-setup-value-of.patch @@ -0,0 +1,112 @@ +From 54403f52f9be5f4d05f1cc866b397820340099ab Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 007/115] i386/cpu: Extract a common fucntion to setup value of + MSR_CORE_THREAD_COUNT + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [7/115] d61186d383365a629a8d74a5618b4df6b2723a88 (bonzini/rhel-qemu-kvm) + +There are duplicated code to setup the value of MSR_CORE_THREAD_COUNT. +Extract a common function for it. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20241219110125.1266461-2-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit d3bb5d0d4f5d4ad7dc6c02ea5fea51ca2f946593) +Signed-off-by: Paolo Bonzini +(cherry picked from commit 66f4008d32a6cddd1ada5906487ecc9fb0b62fed) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu-sysemu.c | 11 +++++++++++ + target/i386/cpu.h | 2 ++ + target/i386/hvf/x86_emu.c | 3 +-- + target/i386/kvm/kvm.c | 5 +---- + target/i386/tcg/sysemu/misc_helper.c | 3 +-- + 5 files changed, 16 insertions(+), 8 deletions(-) + +diff --git a/target/i386/cpu-sysemu.c b/target/i386/cpu-sysemu.c +index 227ac021f6..4e9df0bc01 100644 +--- a/target/i386/cpu-sysemu.c ++++ b/target/i386/cpu-sysemu.c +@@ -309,3 +309,14 @@ void x86_cpu_get_crash_info_qom(Object *obj, Visitor *v, + errp); + qapi_free_GuestPanicInformation(panic_info); + } ++ ++uint64_t cpu_x86_get_msr_core_thread_count(X86CPU *cpu) ++{ ++ CPUState *cs = CPU(cpu); ++ uint64_t val; ++ ++ val = cs->nr_threads * cs->nr_cores; /* thread count, bits 15..0 */ ++ val |= ((uint32_t)cs->nr_cores << 16); /* core count, bits 31..16 */ ++ ++ return val; ++} +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index 5924761551..8c9216d9d0 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -2368,6 +2368,8 @@ static inline void cpu_x86_load_seg_cache_sipi(X86CPU *cpu, + cs->halted = 0; + } + ++uint64_t cpu_x86_get_msr_core_thread_count(X86CPU *cpu); ++ + int cpu_x86_get_descr_debug(CPUX86State *env, unsigned int selector, + target_ulong *base, unsigned int *limit, + unsigned int *flags); +diff --git a/target/i386/hvf/x86_emu.c b/target/i386/hvf/x86_emu.c +index 38c782b8e3..425f1afda8 100644 +--- a/target/i386/hvf/x86_emu.c ++++ b/target/i386/hvf/x86_emu.c +@@ -745,8 +745,7 @@ void simulate_rdmsr(CPUX86State *env) + val = env->mtrr_deftype; + break; + case MSR_CORE_THREAD_COUNT: +- val = cs->nr_threads * cs->nr_cores; /* thread count, bits 15..0 */ +- val |= ((uint32_t)cs->nr_cores << 16); /* core count, bits 31..16 */ ++ val = cpu_x86_get_msr_core_thread_count(cpu); + break; + default: + /* fprintf(stderr, "%s: unknown msr 0x%x\n", __func__, msr); */ +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index b02aec915c..fe6b34bb10 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -2586,10 +2586,7 @@ static bool kvm_rdmsr_core_thread_count(X86CPU *cpu, + uint32_t msr, + uint64_t *val) + { +- CPUState *cs = CPU(cpu); +- +- *val = cs->nr_threads * cs->nr_cores; /* thread count, bits 15..0 */ +- *val |= ((uint32_t)cs->nr_cores << 16); /* core count, bits 31..16 */ ++ *val = cpu_x86_get_msr_core_thread_count(cpu); + + return true; + } +diff --git a/target/i386/tcg/sysemu/misc_helper.c b/target/i386/tcg/sysemu/misc_helper.c +index 094aa56a20..ff7b201b44 100644 +--- a/target/i386/tcg/sysemu/misc_helper.c ++++ b/target/i386/tcg/sysemu/misc_helper.c +@@ -468,8 +468,7 @@ void helper_rdmsr(CPUX86State *env) + val = x86_cpu->ucode_rev; + break; + case MSR_CORE_THREAD_COUNT: { +- CPUState *cs = CPU(x86_cpu); +- val = (cs->nr_threads * cs->nr_cores) | (cs->nr_cores << 16); ++ val = cpu_x86_get_msr_core_thread_count(x86_cpu); + break; + } + case MSR_APIC_START ... MSR_APIC_END: { +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-Hoist-check-of-CPUID_EXT3_TOPOEXT-against-t.patch b/SOURCES/kvm-i386-cpu-Hoist-check-of-CPUID_EXT3_TOPOEXT-against-t.patch new file mode 100644 index 0000000..00f22c2 --- /dev/null +++ b/SOURCES/kvm-i386-cpu-Hoist-check-of-CPUID_EXT3_TOPOEXT-against-t.patch @@ -0,0 +1,80 @@ +From 97edfb8e45b34e3909f387162785f0aa8979aba6 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 013/115] i386/cpu: Hoist check of CPUID_EXT3_TOPOEXT against + threads_per_core + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [13/115] e8e81cac13bb9363c167e92a8478efb97ceab79a (bonzini/rhel-qemu-kvm) + +Now it changes to use env->topo_info.threads_per_core and doesn't depend +on qemu_init_vcpu() anymore. Put it together with other feature checks +before qemu_init_vcpu() + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20241219110125.1266461-8-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 473d79b56a1645be90b890f9623b27acd0afba49) +Signed-off-by: Paolo Bonzini +(cherry picked from commit 8098c705a5d8f82d2a772194ebdbef87784b0461) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 30 +++++++++++++++--------------- + 1 file changed, 15 insertions(+), 15 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 554455169a..2295149bfa 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -8327,6 +8327,21 @@ static void x86_cpu_realizefn(DeviceState *dev, Error **errp) + */ + cpu->mwait.ecx |= CPUID_MWAIT_EMX | CPUID_MWAIT_IBE; + ++ /* ++ * Most Intel and certain AMD CPUs support hyperthreading. Even though QEMU ++ * fixes this issue by adjusting CPUID_0000_0001_EBX and CPUID_8000_0008_ECX ++ * based on inputs (sockets,cores,threads), it is still better to give ++ * users a warning. ++ */ ++ if (IS_AMD_CPU(env) && ++ !(env->features[FEAT_8000_0001_ECX] & CPUID_EXT3_TOPOEXT) && ++ env->topo_info.threads_per_core > 1) { ++ warn_report_once("This family of AMD CPU doesn't support " ++ "hyperthreading(%d). Please configure -smp " ++ "options properly or try enabling topoext " ++ "feature.", env->topo_info.threads_per_core); ++ } ++ + /* For 64bit systems think about the number of physical bits to present. + * ideally this should be the same as the host; anything other than matching + * the host can cause incorrect guest behaviour. +@@ -8430,21 +8445,6 @@ static void x86_cpu_realizefn(DeviceState *dev, Error **errp) + + qemu_init_vcpu(cs); + +- /* +- * Most Intel and certain AMD CPUs support hyperthreading. Even though QEMU +- * fixes this issue by adjusting CPUID_0000_0001_EBX and CPUID_8000_0008_ECX +- * based on inputs (sockets,cores,threads), it is still better to give +- * users a warning. +- */ +- if (IS_AMD_CPU(env) && +- !(env->features[FEAT_8000_0001_ECX] & CPUID_EXT3_TOPOEXT) && +- env->topo_info.threads_per_core > 1) { +- warn_report_once("This family of AMD CPU doesn't support " +- "hyperthreading(%d). Please configure -smp " +- "options properly or try enabling topoext " +- "feature.", env->topo_info.threads_per_core); +- } +- + #ifndef CONFIG_USER_ONLY + x86_cpu_apic_realize(cpu, &local_err); + if (local_err != NULL) { +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-Introduce-enable_cpuid_0x1f-to-force-exposi.patch b/SOURCES/kvm-i386-cpu-Introduce-enable_cpuid_0x1f-to-force-exposi.patch new file mode 100644 index 0000000..25db7b1 --- /dev/null +++ b/SOURCES/kvm-i386-cpu-Introduce-enable_cpuid_0x1f-to-force-exposi.patch @@ -0,0 +1,105 @@ +From c45bc21b4c191ceb2523bfeac4ff2eb042469d89 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 062/115] i386/cpu: Introduce enable_cpuid_0x1f to force + exposing CPUID 0x1f + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [62/115] 046ef898225f70d84fbcf20266aff8478c6fb428 (bonzini/rhel-qemu-kvm) + +Currently, QEMU exposes CPUID 0x1f to guest only when necessary, i.e., +when topology level that cannot be enumerated by leaf 0xB, e.g., die or +module level, are configured for the guest, e.g., -smp xx,dies=2. + +However, TDX architecture forces to require CPUID 0x1f to configure CPU +topology. + +Introduce a bool flag, enable_cpuid_0x1f, in CPU for the case that +requires CPUID leaf 0x1f to be exposed to guest. + +Introduce a new function x86_has_cpuid_0x1f(), which is the wrapper of +cpu->enable_cpuid_0x1f and x86_has_extended_topo() to check if it needs +to enable cpuid leaf 0x1f for the guest. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-34-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit ab8bd85adf75900edc2764d0ebe8b53867cc54aa) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 4 ++-- + target/i386/cpu.h | 9 +++++++++ + target/i386/kvm/kvm.c | 2 +- + 3 files changed, 12 insertions(+), 3 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index ee6f1c0627..ab34626b19 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -7177,7 +7177,7 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + break; + case 0x1F: + /* V2 Extended Topology Enumeration Leaf */ +- if (!x86_has_extended_topo(env->avail_cpu_topo)) { ++ if (!x86_has_cpuid_0x1f(cpu)) { + *eax = *ebx = *ecx = *edx = 0; + break; + } +@@ -8035,7 +8035,7 @@ void x86_cpu_expand_features(X86CPU *cpu, Error **errp) + * cpu->vendor_cpuid_only has been unset for compatibility with older + * machine types. + */ +- if (x86_has_extended_topo(env->avail_cpu_topo) && ++ if (x86_has_cpuid_0x1f(cpu) && + (IS_INTEL_CPU(env) || !cpu->vendor_cpuid_only)) { + x86_cpu_adjust_level(cpu, &env->cpuid_min_level, 0x1F); + } +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index ee1a1b6622..83fa89bf0a 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -2168,6 +2168,9 @@ struct ArchCPU { + /* Compatibility bits for old machine types: */ + bool enable_cpuid_0xb; + ++ /* Force to enable cpuid 0x1f */ ++ bool enable_cpuid_0x1f; ++ + /* Enable auto level-increase for all CPUID leaves */ + bool full_cpuid_auto_level; + +@@ -2429,6 +2432,12 @@ void host_cpuid(uint32_t function, uint32_t count, + uint32_t *eax, uint32_t *ebx, uint32_t *ecx, uint32_t *edx); + bool cpu_has_x2apic_feature(CPUX86State *env); + ++static inline bool x86_has_cpuid_0x1f(X86CPU *cpu) ++{ ++ return cpu->enable_cpuid_0x1f || ++ x86_has_extended_topo(cpu->env.avail_cpu_topo); ++} ++ + /* helper.c */ + void x86_cpu_set_a20(X86CPU *cpu, int a20_state); + void cpu_sync_avx_hflag(CPUX86State *env); +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index 4bda4f5525..f4809ee004 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -1861,7 +1861,7 @@ uint32_t kvm_x86_build_cpuid(CPUX86State *env, struct kvm_cpuid_entry2 *entries, + break; + } + case 0x1f: +- if (!x86_has_extended_topo(env->avail_cpu_topo)) { ++ if (!x86_has_cpuid_0x1f(env_archcpu(env))) { + cpuid_i--; + break; + } +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-Move-adjustment-of-CPUID_EXT_PDCM-before-fe.patch b/SOURCES/kvm-i386-cpu-Move-adjustment-of-CPUID_EXT_PDCM-before-fe.patch new file mode 100644 index 0000000..f720123 --- /dev/null +++ b/SOURCES/kvm-i386-cpu-Move-adjustment-of-CPUID_EXT_PDCM-before-fe.patch @@ -0,0 +1,59 @@ +From faffdc7fae7e90e9b4bdbb36ab2e753ebbf26732 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:49 +0200 +Subject: [PATCH 087/115] i386/cpu: Move adjustment of CPUID_EXT_PDCM before + feature_dependencies[] check + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [87/115] 8e3628cfafdd307c5a38cbd2ec52ce3f679c1796 (bonzini/rhel-qemu-kvm) + +There is one entry relates to CPUID_EXT_PDCM in feature_dependencies[]. +So it needs to get correct value of CPUID_EXT_PDCM before using +feature_dependencies[] to apply dependencies. + +Besides, it also ensures CPUID_EXT_PDCM value is tracked in +env->features[FEAT_1_ECX]. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250304052450.465445-2-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit e68ec2980901c8e7f948f3305770962806c53f0b) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 7 ++++--- + 1 file changed, 4 insertions(+), 3 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 433d0a0418..a9d6811032 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -7026,9 +7026,6 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + if (threads_per_pkg > 1) { + *ebx |= threads_per_pkg << 16; + } +- if (!cpu->enable_pmu) { +- *ecx &= ~CPUID_EXT_PDCM; +- } + break; + case 2: + /* cache info: needed for Pentium Pro compatibility */ +@@ -8012,6 +8009,10 @@ void x86_cpu_expand_features(X86CPU *cpu, Error **errp) + } + } + ++ if (!cpu->enable_pmu) { ++ env->features[FEAT_1_ECX] &= ~CPUID_EXT_PDCM; ++ } ++ + for (i = 0; i < ARRAY_SIZE(feature_dependencies); i++) { + FeatureDep *d = &feature_dependencies[i]; + if (!(env->features[d->from.index] & d->from.mask)) { +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-Move-x86_ext_save_areas-initialization-to-..patch b/SOURCES/kvm-i386-cpu-Move-x86_ext_save_areas-initialization-to-..patch new file mode 100644 index 0000000..c668f8f --- /dev/null +++ b/SOURCES/kvm-i386-cpu-Move-x86_ext_save_areas-initialization-to-..patch @@ -0,0 +1,91 @@ +From eb67f5f683c98892292aa40695d7ec59a17c122f Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:50 +0200 +Subject: [PATCH 105/115] i386/cpu: Move x86_ext_save_areas[] initialization to + .instance_init + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [105/115] 289d5143a71cd0dc24c6d8e32d510657d5c77091 (bonzini/rhel-qemu-kvm) + +In x86_cpu_post_initfn(), the initialization of x86_ext_save_areas[] +marks the unsupported xsave areas based on Host support. + +This step must be done before accel_cpu_instance_init(), otherwise, +KVM's assertion on host xsave support would fail: + +qemu-system-x86_64: ../target/i386/kvm/kvm-cpu.c:149: +kvm_cpu_xsave_init: Assertion `esa->size == eax' failed. + +(on AMD EPYC 7302 16-Core Processor) + +Move x86_ext_save_areas[] initialization to .instance_init and place it +before accel_cpu_instance_init(). + +Fixes: commit 5f158abef44c ("target/i386: move accel_cpu_instance_init to .instance_init") +Reported-by: Paolo Abeni +Tested-by: Paolo Abeni +Signed-off-by: Zhao Liu +Link: https://lore.kernel.org/r/20250717023933.2502109-1-zhao1.liu@intel.com +Reviewed-by: Xiaoyao Li +Signed-off-by: Paolo Bonzini +(cherry picked from commit e52af92e9e6f8fc00f2ae6b63214b3d6213b3cec) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 22 +++++++++++++++------- + 1 file changed, 15 insertions(+), 7 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index d0161f922c..ee753351fc 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -8617,6 +8617,16 @@ static void x86_cpu_register_feature_bit_props(X86CPUClass *xcc, + } + + static void x86_cpu_post_initfn(Object *obj) ++{ ++#ifndef CONFIG_USER_ONLY ++ if (current_machine && current_machine->cgs) { ++ x86_confidential_guest_cpu_instance_init( ++ X86_CONFIDENTIAL_GUEST(current_machine->cgs), (CPU(obj))); ++ } ++#endif ++} ++ ++static void x86_cpu_init_xsave(void) + { + static bool first = true; + uint64_t supported_xcr0; +@@ -8637,13 +8647,6 @@ static void x86_cpu_post_initfn(Object *obj) + } + } + } +- +-#ifndef CONFIG_USER_ONLY +- if (current_machine && current_machine->cgs) { +- x86_confidential_guest_cpu_instance_init( +- X86_CONFIDENTIAL_GUEST(current_machine->cgs), (CPU(obj))); +- } +-#endif + } + + static void x86_cpu_init_default_topo(X86CPU *cpu) +@@ -8713,6 +8716,11 @@ static void x86_cpu_initfn(Object *obj) + x86_cpu_load_model(cpu, xcc->model); + } + ++ /* ++ * accel's cpu_instance_init may have the xsave check, ++ * so x86_ext_save_areas[] must be initialized before this. ++ */ ++ x86_cpu_init_xsave(); + accel_cpu_instance_init(CPU(obj)); + } + +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-Rename-enable_cpuid_0x1f-to-force_cpuid_0x1.patch b/SOURCES/kvm-i386-cpu-Rename-enable_cpuid_0x1f-to-force_cpuid_0x1.patch new file mode 100644 index 0000000..0d3aa45 --- /dev/null +++ b/SOURCES/kvm-i386-cpu-Rename-enable_cpuid_0x1f-to-force_cpuid_0x1.patch @@ -0,0 +1,73 @@ +From 07bba3fcfd6b8eb6e833e7d675be3ccc667423de Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:49 +0200 +Subject: [PATCH 089/115] i386/cpu: Rename enable_cpuid_0x1f to + force_cpuid_0x1f +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [89/115] d1152aea8e699b1563750d286b41a47d9036429e (bonzini/rhel-qemu-kvm) + +The name of "enable_cpuid_0x1f" isn't right to its behavior because the +leaf 0x1f can be enabled even when "enable_cpuid_0x1f" is false. + +Rename it to "force_cpuid_0x1f" to better reflect its behavior. + +Suggested-by: Igor Mammedov +Signed-off-by: Xiaoyao Li +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Igor Mammedov +Link: https://lore.kernel.org/r/20250603050305.1704586-2-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 90d2bbd1f6edfa22a056070ee62ded55099cd56d) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.h | 4 ++-- + target/i386/kvm/tdx.c | 2 +- + 2 files changed, 3 insertions(+), 3 deletions(-) + +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index 2e73945b28..a08931f969 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -2195,7 +2195,7 @@ struct ArchCPU { + bool enable_cpuid_0xb; + + /* Force to enable cpuid 0x1f */ +- bool enable_cpuid_0x1f; ++ bool force_cpuid_0x1f; + + /* Enable auto level-increase for all CPUID leaves */ + bool full_cpuid_auto_level; +@@ -2465,7 +2465,7 @@ void mark_forced_on_features(X86CPU *cpu, FeatureWord w, uint64_t mask, + + static inline bool x86_has_cpuid_0x1f(X86CPU *cpu) + { +- return cpu->enable_cpuid_0x1f || ++ return cpu->force_cpuid_0x1f || + x86_has_extended_topo(cpu->env.avail_cpu_topo); + } + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 3099e40baa..ca3641441c 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -752,7 +752,7 @@ static void tdx_cpu_instance_init(X86ConfidentialGuest *cg, CPUState *cpu) + /* invtsc is fixed1 for TD guest */ + object_property_set_bool(OBJECT(cpu), "invtsc", true, &error_abort); + +- x86cpu->enable_cpuid_0x1f = true; ++ x86cpu->force_cpuid_0x1f = true; + } + + static uint32_t tdx_adjust_cpuid_features(X86ConfidentialGuest *cg, +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-Set-and-track-CPUID_EXT3_CMP_LEG-in-env-fea.patch b/SOURCES/kvm-i386-cpu-Set-and-track-CPUID_EXT3_CMP_LEG-in-env-fea.patch new file mode 100644 index 0000000..a9c1cb4 --- /dev/null +++ b/SOURCES/kvm-i386-cpu-Set-and-track-CPUID_EXT3_CMP_LEG-in-env-fea.patch @@ -0,0 +1,69 @@ +From 90ccca2272ea05adc4ecedbbac70692f72831533 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 016/115] i386/cpu: Set and track CPUID_EXT3_CMP_LEG in + env->features[FEAT_8000_0001_ECX] + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [16/115] 8d6efb657ea34f7adccb5f8648cf0d10ba126299 (bonzini/rhel-qemu-kvm) + +The correct usage is tracking and maintaining features in env->features[] +instead of manually set it in cpu_x86_cpuid(). + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20241219110125.1266461-11-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 99a637a86f55c8486b06c698656befdf012eec4d) +Signed-off-by: Paolo Bonzini +(cherry picked from commit eb7304fc81c136a475a951720b76afa1f128ae95) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 20 +++++++++----------- + 1 file changed, 9 insertions(+), 11 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index e20977411d..a97d042a2e 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -7404,17 +7404,6 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + *ecx = env->features[FEAT_8000_0001_ECX]; + *edx = env->features[FEAT_8000_0001_EDX]; + +- /* The Linux kernel checks for the CMPLegacy bit and +- * discards multiple thread information if it is set. +- * So don't set it here for Intel to make Linux guests happy. +- */ +- if (threads_per_pkg > 1) { +- if (env->cpuid_vendor1 != CPUID_VENDOR_INTEL_1 || +- env->cpuid_vendor2 != CPUID_VENDOR_INTEL_2 || +- env->cpuid_vendor3 != CPUID_VENDOR_INTEL_3) { +- *ecx |= 1 << 1; /* CmpLegacy bit */ +- } +- } + if (tcg_enabled() && env->cpuid_vendor1 == CPUID_VENDOR_INTEL_1 && + !(env->hflags & HF_LMA_MASK)) { + *edx &= ~CPUID_EXT2_SYSCALL; +@@ -7976,6 +7965,15 @@ void x86_cpu_expand_features(X86CPU *cpu, Error **errp) + + if (x86_threads_per_pkg(&env->topo_info) > 1) { + env->features[FEAT_1_EDX] |= CPUID_HT; ++ ++ /* ++ * The Linux kernel checks for the CMPLegacy bit and ++ * discards multiple thread information if it is set. ++ * So don't set it here for Intel to make Linux guests happy. ++ */ ++ if (!IS_INTEL_CPU(env)) { ++ env->features[FEAT_8000_0001_ECX] |= CPUID_EXT3_CMP_LEG; ++ } + } + + for (i = 0; i < ARRAY_SIZE(feature_dependencies); i++) { +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-Set-up-CPUID_HT-in-x86_cpu_expand_features-.patch b/SOURCES/kvm-i386-cpu-Set-up-CPUID_HT-in-x86_cpu_expand_features-.patch new file mode 100644 index 0000000..4c7342e --- /dev/null +++ b/SOURCES/kvm-i386-cpu-Set-up-CPUID_HT-in-x86_cpu_expand_features-.patch @@ -0,0 +1,59 @@ +From 673bbd9c79946356d2863f0944a576d100750c64 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 015/115] i386/cpu: Set up CPUID_HT in + x86_cpu_expand_features() instead of cpu_x86_cpuid() + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [15/115] c763ec424a0cc64f46fd27aa2f03fd1b2933c23b (bonzini/rhel-qemu-kvm) + +Currently CPUID_HT is evaluated in cpu_x86_cpuid() each time. It's not a +correct usage of how feature bit is maintained and evaluated. The +expected practice is that features are tracked in env->features[] and +cpu_x86_cpuid() should be the consumer of env->features[]. + +Track CPUID_HT in env->features[FEAT_1_EDX] instead and evaluate it in +cpu's realizefn(). + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20241219110125.1266461-10-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit c6bd2dd634208ca717b6dc010064fe34d1359080) +Signed-off-by: Paolo Bonzini +(cherry picked from commit 4e3da2efde9f16460e90246bec15b29739930bb1) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 5 ++++- + 1 file changed, 4 insertions(+), 1 deletion(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 2295149bfa..e20977411d 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -6989,7 +6989,6 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + *edx = env->features[FEAT_1_EDX]; + if (threads_per_pkg > 1) { + *ebx |= threads_per_pkg << 16; +- *edx |= CPUID_HT; + } + if (!cpu->enable_pmu) { + *ecx &= ~CPUID_EXT_PDCM; +@@ -7975,6 +7974,10 @@ void x86_cpu_expand_features(X86CPU *cpu, Error **errp) + } + } + ++ if (x86_threads_per_pkg(&env->topo_info) > 1) { ++ env->features[FEAT_1_EDX] |= CPUID_HT; ++ } ++ + for (i = 0; i < ARRAY_SIZE(feature_dependencies); i++) { + FeatureDep *d = &feature_dependencies[i]; + if (!(env->features[d->from.index] & d->from.mask)) { +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-Track-a-X86CPUTopoInfo-directly-in-CPUX86St.patch b/SOURCES/kvm-i386-cpu-Track-a-X86CPUTopoInfo-directly-in-CPUX86St.patch new file mode 100644 index 0000000..05b5ca0 --- /dev/null +++ b/SOURCES/kvm-i386-cpu-Track-a-X86CPUTopoInfo-directly-in-CPUX86St.patch @@ -0,0 +1,301 @@ +From 03d6e64ed980335602368d3b470e33cb21af307d Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 012/115] i386/cpu: Track a X86CPUTopoInfo directly in + CPUX86State + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [12/115] 4d23b4e386ec4179739869c1ace2bb2b9f728bde (bonzini/rhel-qemu-kvm) + +The name of nr_modules/nr_dies are ambiguous and they mislead people. + +The purpose of them is to record and form the topology information. So +just maintain a X86CPUTopoInfo member in CPUX86State instead. Then +nr_modules and nr_dies can be dropped. + +As the benefit, x86 can switch to use information in +CPUX86State::topo_info and get rid of the nr_cores and nr_threads in +CPUState. This helps remove the dependency on qemu_init_vcpu(), so that +x86 can get and use topology info earlier in x86_cpu_realizefn(); drop +the comment that highlighted the depedency. + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20241219110125.1266461-7-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 84b71a131c1bc84c36fafb63271080ecf9f2ff7a) +Signed-off-by: Paolo Bonzini +(cherry picked from commit f598befcd2cddbf01f7606102400cec4acd35d48) +Signed-off-by: Paolo Bonzini +--- + hw/i386/x86-common.c | 12 ++++------ + target/i386/cpu-sysemu.c | 6 ++--- + target/i386/cpu.c | 51 +++++++++++++++++----------------------- + target/i386/cpu.h | 6 +---- + 4 files changed, 30 insertions(+), 45 deletions(-) + +diff --git a/hw/i386/x86-common.c b/hw/i386/x86-common.c +index 2806be98f3..562215990d 100644 +--- a/hw/i386/x86-common.c ++++ b/hw/i386/x86-common.c +@@ -248,7 +248,7 @@ void x86_cpu_pre_plug(HotplugHandler *hotplug_dev, + CPUX86State *env = &cpu->env; + MachineState *ms = MACHINE(hotplug_dev); + X86MachineState *x86ms = X86_MACHINE(hotplug_dev); +- X86CPUTopoInfo topo_info; ++ X86CPUTopoInfo *topo_info = &env->topo_info; + + if (!object_dynamic_cast(OBJECT(cpu), ms->cpu_type)) { + error_setg(errp, "Invalid CPU type, expected cpu type: '%s'", +@@ -267,15 +267,13 @@ void x86_cpu_pre_plug(HotplugHandler *hotplug_dev, + } + } + +- init_topo_info(&topo_info, x86ms); ++ init_topo_info(topo_info, x86ms); + + if (ms->smp.modules > 1) { +- env->nr_modules = ms->smp.modules; + set_bit(CPU_TOPO_LEVEL_MODULE, env->avail_cpu_topo); + } + + if (ms->smp.dies > 1) { +- env->nr_dies = ms->smp.dies; + set_bit(CPU_TOPO_LEVEL_DIE, env->avail_cpu_topo); + } + +@@ -346,12 +344,12 @@ void x86_cpu_pre_plug(HotplugHandler *hotplug_dev, + topo_ids.module_id = cpu->module_id; + topo_ids.core_id = cpu->core_id; + topo_ids.smt_id = cpu->thread_id; +- cpu->apic_id = x86_apicid_from_topo_ids(&topo_info, &topo_ids); ++ cpu->apic_id = x86_apicid_from_topo_ids(topo_info, &topo_ids); + } + + cpu_slot = x86_find_cpu_slot(MACHINE(x86ms), cpu->apic_id, &idx); + if (!cpu_slot) { +- x86_topo_ids_from_apicid(cpu->apic_id, &topo_info, &topo_ids); ++ x86_topo_ids_from_apicid(cpu->apic_id, topo_info, &topo_ids); + + error_setg(errp, + "Invalid CPU [socket: %u, die: %u, module: %u, core: %u, thread: %u]" +@@ -374,7 +372,7 @@ void x86_cpu_pre_plug(HotplugHandler *hotplug_dev, + /* TODO: move socket_id/core_id/thread_id checks into x86_cpu_realizefn() + * once -smp refactoring is complete and there will be CPU private + * CPUState::nr_cores and CPUState::nr_threads fields instead of globals */ +- x86_topo_ids_from_apicid(cpu->apic_id, &topo_info, &topo_ids); ++ x86_topo_ids_from_apicid(cpu->apic_id, topo_info, &topo_ids); + if (cpu->socket_id != -1 && cpu->socket_id != topo_ids.pkg_id) { + error_setg(errp, "property socket-id: %u doesn't match set apic-id:" + " 0x%x (socket-id: %u)", cpu->socket_id, cpu->apic_id, +diff --git a/target/i386/cpu-sysemu.c b/target/i386/cpu-sysemu.c +index 4e9df0bc01..31b37c6325 100644 +--- a/target/i386/cpu-sysemu.c ++++ b/target/i386/cpu-sysemu.c +@@ -312,11 +312,11 @@ void x86_cpu_get_crash_info_qom(Object *obj, Visitor *v, + + uint64_t cpu_x86_get_msr_core_thread_count(X86CPU *cpu) + { +- CPUState *cs = CPU(cpu); ++ CPUX86State *env = &cpu->env; + uint64_t val; + +- val = cs->nr_threads * cs->nr_cores; /* thread count, bits 15..0 */ +- val |= ((uint32_t)cs->nr_cores << 16); /* core count, bits 31..16 */ ++ val = x86_threads_per_pkg(&env->topo_info); /* thread count, bits 15..0 */ ++ val |= x86_cores_per_pkg(&env->topo_info) << 16; /* core count, bits 31..16 */ + + return val; + } +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 1c79eb9a06..554455169a 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -6947,15 +6947,10 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + CPUState *cs = env_cpu(env); + uint32_t limit; + uint32_t signature[3]; +- X86CPUTopoInfo topo_info; ++ X86CPUTopoInfo *topo_info = &env->topo_info; + uint32_t threads_per_pkg; + +- topo_info.dies_per_pkg = env->nr_dies; +- topo_info.modules_per_die = env->nr_modules; +- topo_info.cores_per_module = cs->nr_cores / env->nr_dies / env->nr_modules; +- topo_info.threads_per_core = cs->nr_threads; +- +- threads_per_pkg = x86_threads_per_pkg(&topo_info); ++ threads_per_pkg = x86_threads_per_pkg(topo_info); + + /* Calculate & apply limits for different index ranges */ + if (index >= 0xC0000000) { +@@ -7032,12 +7027,12 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + int host_vcpus_per_cache = 1 + ((*eax & 0x3FFC000) >> 14); + + *eax &= ~0xFC000000; +- *eax |= max_core_ids_in_package(&topo_info) << 26; ++ *eax |= max_core_ids_in_package(topo_info) << 26; + if (host_vcpus_per_cache > threads_per_pkg) { + *eax &= ~0x3FFC000; + + /* Share the cache at package level. */ +- *eax |= max_thread_ids_for_cache(&topo_info, ++ *eax |= max_thread_ids_for_cache(topo_info, + CPU_TOPO_LEVEL_PACKAGE) << 14; + } + } +@@ -7049,7 +7044,7 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + switch (count) { + case 0: /* L1 dcache info */ + encode_cache_cpuid4(env->cache_info_cpuid4.l1d_cache, +- &topo_info, ++ topo_info, + eax, ebx, ecx, edx); + if (!cpu->l1_cache_per_core) { + *eax &= ~MAKE_64BIT_MASK(14, 12); +@@ -7057,7 +7052,7 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + break; + case 1: /* L1 icache info */ + encode_cache_cpuid4(env->cache_info_cpuid4.l1i_cache, +- &topo_info, ++ topo_info, + eax, ebx, ecx, edx); + if (!cpu->l1_cache_per_core) { + *eax &= ~MAKE_64BIT_MASK(14, 12); +@@ -7065,13 +7060,13 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + break; + case 2: /* L2 cache info */ + encode_cache_cpuid4(env->cache_info_cpuid4.l2_cache, +- &topo_info, ++ topo_info, + eax, ebx, ecx, edx); + break; + case 3: /* L3 cache info */ + if (cpu->enable_l3_cache) { + encode_cache_cpuid4(env->cache_info_cpuid4.l3_cache, +- &topo_info, ++ topo_info, + eax, ebx, ecx, edx); + break; + } +@@ -7154,12 +7149,12 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + + switch (count) { + case 0: +- *eax = apicid_core_offset(&topo_info); +- *ebx = topo_info.threads_per_core; ++ *eax = apicid_core_offset(topo_info); ++ *ebx = topo_info->threads_per_core; + *ecx |= CPUID_B_ECX_TOPO_LEVEL_SMT << 8; + break; + case 1: +- *eax = apicid_pkg_offset(&topo_info); ++ *eax = apicid_pkg_offset(topo_info); + *ebx = threads_per_pkg; + *ecx |= CPUID_B_ECX_TOPO_LEVEL_CORE << 8; + break; +@@ -7185,7 +7180,7 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + break; + } + +- encode_topo_cpuid1f(env, count, &topo_info, eax, ebx, ecx, edx); ++ encode_topo_cpuid1f(env, count, topo_info, eax, ebx, ecx, edx); + break; + case 0xD: { + /* Processor Extended State */ +@@ -7488,7 +7483,7 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + * thread ID within a package". + * Bits 7:0 is "The number of threads in the package is NC+1" + */ +- *ecx = (apicid_pkg_offset(&topo_info) << 12) | ++ *ecx = (apicid_pkg_offset(topo_info) << 12) | + (threads_per_pkg - 1); + } else { + *ecx = 0; +@@ -7517,19 +7512,19 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + switch (count) { + case 0: /* L1 dcache info */ + encode_cache_cpuid8000001d(env->cache_info_amd.l1d_cache, +- &topo_info, eax, ebx, ecx, edx); ++ topo_info, eax, ebx, ecx, edx); + break; + case 1: /* L1 icache info */ + encode_cache_cpuid8000001d(env->cache_info_amd.l1i_cache, +- &topo_info, eax, ebx, ecx, edx); ++ topo_info, eax, ebx, ecx, edx); + break; + case 2: /* L2 cache info */ + encode_cache_cpuid8000001d(env->cache_info_amd.l2_cache, +- &topo_info, eax, ebx, ecx, edx); ++ topo_info, eax, ebx, ecx, edx); + break; + case 3: /* L3 cache info */ + encode_cache_cpuid8000001d(env->cache_info_amd.l3_cache, +- &topo_info, eax, ebx, ecx, edx); ++ topo_info, eax, ebx, ecx, edx); + break; + default: /* end of info */ + *eax = *ebx = *ecx = *edx = 0; +@@ -7541,7 +7536,7 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + break; + case 0x8000001E: + if (cpu->core_id <= 255) { +- encode_topo_cpuid8000001e(cpu, &topo_info, eax, ebx, ecx, edx); ++ encode_topo_cpuid8000001e(cpu, topo_info, eax, ebx, ecx, edx); + } else { + *eax = 0; + *ebx = 0; +@@ -8440,17 +8435,14 @@ static void x86_cpu_realizefn(DeviceState *dev, Error **errp) + * fixes this issue by adjusting CPUID_0000_0001_EBX and CPUID_8000_0008_ECX + * based on inputs (sockets,cores,threads), it is still better to give + * users a warning. +- * +- * NOTE: the following code has to follow qemu_init_vcpu(). Otherwise +- * cs->nr_threads hasn't be populated yet and the checking is incorrect. + */ + if (IS_AMD_CPU(env) && + !(env->features[FEAT_8000_0001_ECX] & CPUID_EXT3_TOPOEXT) && +- cs->nr_threads > 1) { ++ env->topo_info.threads_per_core > 1) { + warn_report_once("This family of AMD CPU doesn't support " + "hyperthreading(%d). Please configure -smp " + "options properly or try enabling topoext " +- "feature.", cs->nr_threads); ++ "feature.", env->topo_info.threads_per_core); + } + + #ifndef CONFIG_USER_ONLY +@@ -8611,8 +8603,7 @@ static void x86_cpu_init_default_topo(X86CPU *cpu) + { + CPUX86State *env = &cpu->env; + +- env->nr_modules = 1; +- env->nr_dies = 1; ++ env->topo_info = (X86CPUTopoInfo) {1, 1, 1, 1}; + + /* SMT, core and package levels are set by default. */ + set_bit(CPU_TOPO_LEVEL_SMT, env->avail_cpu_topo); +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index 8c9216d9d0..cc42a2c520 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -2024,11 +2024,7 @@ typedef struct CPUArchState { + + TPRAccess tpr_access_type; + +- /* Number of dies within this CPU package. */ +- unsigned nr_dies; +- +- /* Number of modules within one die. */ +- unsigned nr_modules; ++ X86CPUTopoInfo topo_info; + + /* Bitmap of available CPU topology levels for this CPU. */ + DECLARE_BITMAP(avail_cpu_topo, CPU_TOPO_LEVEL_MAX); +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-cpu-introduce-x86_confidential_guest_cpu_instan.patch b/SOURCES/kvm-i386-cpu-introduce-x86_confidential_guest_cpu_instan.patch new file mode 100644 index 0000000..abde4c4 --- /dev/null +++ b/SOURCES/kvm-i386-cpu-introduce-x86_confidential_guest_cpu_instan.patch @@ -0,0 +1,87 @@ +From 831b04c3f8f66705a01b4fd455dbe53fa3193a4e Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 060/115] i386/cpu: introduce + x86_confidential_guest_cpu_instance_init() + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [60/115] be57d5108231bb12dc6c9ffd8c9c639b87b1f15c (bonzini/rhel-qemu-kvm) + +To allow execute confidential guest specific cpu init operations. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-32-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 8583c53e2b619b1b9569d3f2d3f3cb2904a573ad) +Signed-off-by: Paolo Bonzini + +Conflict: include/system/ is still include/sysemu/ +--- + target/i386/confidential-guest.h | 11 +++++++++++ + target/i386/cpu.c | 8 ++++++++ + 2 files changed, 19 insertions(+) + +diff --git a/target/i386/confidential-guest.h b/target/i386/confidential-guest.h +index 7342d2843a..38169ed68e 100644 +--- a/target/i386/confidential-guest.h ++++ b/target/i386/confidential-guest.h +@@ -39,6 +39,7 @@ struct X86ConfidentialGuestClass { + + /* */ + int (*kvm_type)(X86ConfidentialGuest *cg); ++ void (*cpu_instance_init)(X86ConfidentialGuest *cg, CPUState *cpu); + uint32_t (*mask_cpuid_features)(X86ConfidentialGuest *cg, uint32_t feature, uint32_t index, + int reg, uint32_t value); + }; +@@ -59,6 +60,16 @@ static inline int x86_confidential_guest_kvm_type(X86ConfidentialGuest *cg) + } + } + ++static inline void x86_confidential_guest_cpu_instance_init(X86ConfidentialGuest *cg, ++ CPUState *cpu) ++{ ++ X86ConfidentialGuestClass *klass = X86_CONFIDENTIAL_GUEST_GET_CLASS(cg); ++ ++ if (klass->cpu_instance_init) { ++ klass->cpu_instance_init(cg, cpu); ++ } ++} ++ + /** + * x86_confidential_guest_mask_cpuid_features: + * +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 4eef3d1dbd..ee6f1c0627 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -36,6 +36,7 @@ + #include "hw/qdev-properties.h" + #include "hw/i386/topology.h" + #ifndef CONFIG_USER_ONLY ++#include "confidential-guest.h" + #include "sysemu/reset.h" + #include "qapi/qapi-commands-machine-target.h" + #include "exec/address-spaces.h" +@@ -8600,6 +8601,13 @@ static void x86_cpu_post_initfn(Object *obj) + } + + accel_cpu_instance_init(CPU(obj)); ++ ++#ifndef CONFIG_USER_ONLY ++ if (current_machine && current_machine->cgs) { ++ x86_confidential_guest_cpu_instance_init( ++ X86_CONFIDENTIAL_GUEST(current_machine->cgs), (CPU(obj))); ++ } ++#endif + } + + static void x86_cpu_init_default_topo(X86CPU *cpu) +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdvf-Fix-build-on-32-bit-host.patch b/SOURCES/kvm-i386-tdvf-Fix-build-on-32-bit-host.patch new file mode 100644 index 0000000..7120003 --- /dev/null +++ b/SOURCES/kvm-i386-tdvf-Fix-build-on-32-bit-host.patch @@ -0,0 +1,55 @@ +From d77454aef08828ab25881dbd9acb25f3d6f59d10 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:49 +0200 +Subject: [PATCH 086/115] i386/tdvf: Fix build on 32-bit host +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [86/115] 99b8d992079a248091911010238eb20588acca6a (bonzini/rhel-qemu-kvm) + +Use PRI formats where required. + +Cc: Isaku Yamahata +Signed-off-by: Cédric Le Goater +Link: https://lore.kernel.org/r/20250602173101.1052983-3-clg@redhat.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 6f1035fc65406c4e72e1dbd76e64924415edd616) +Signed-off-by: Paolo Bonzini +--- + hw/i386/tdvf.c | 6 +++--- + 1 file changed, 3 insertions(+), 3 deletions(-) + +diff --git a/hw/i386/tdvf.c b/hw/i386/tdvf.c +index 88453cf3a5..6fcc794880 100644 +--- a/hw/i386/tdvf.c ++++ b/hw/i386/tdvf.c +@@ -101,16 +101,16 @@ static int tdvf_parse_and_check_section_entry(const TdvfSectionEntry *src, + + /* sanity check */ + if (entry->size < entry->data_len) { +- error_report("Broken metadata RawDataSize 0x%x MemoryDataSize 0x%lx", ++ error_report("Broken metadata RawDataSize 0x%x MemoryDataSize 0x%"PRIx64, + entry->data_len, entry->size); + return -1; + } + if (!QEMU_IS_ALIGNED(entry->address, TDVF_ALIGNMENT)) { +- error_report("MemoryAddress 0x%lx not page aligned", entry->address); ++ error_report("MemoryAddress 0x%"PRIx64" not page aligned", entry->address); + return -1; + } + if (!QEMU_IS_ALIGNED(entry->size, TDVF_ALIGNMENT)) { +- error_report("MemoryDataSize 0x%lx not page aligned", entry->size); ++ error_report("MemoryDataSize 0x%"PRIx64" not page aligned", entry->size); + return -1; + } + +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdvf-Introduce-function-to-parse-TDVF-metadata.patch b/SOURCES/kvm-i386-tdvf-Introduce-function-to-parse-TDVF-metadata.patch new file mode 100644 index 0000000..952fcb8 --- /dev/null +++ b/SOURCES/kvm-i386-tdvf-Introduce-function-to-parse-TDVF-metadata.patch @@ -0,0 +1,315 @@ +From e31406cad5a27e2a807dfacb92d9d44efd1615fe Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 046/115] i386/tdvf: Introduce function to parse TDVF metadata + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [46/115] 5eed49eea4f80f57a310e69a47ee220bd41f6188 (bonzini/rhel-qemu-kvm) + +TDX VM needs to boot with its specialized firmware, Trusted Domain +Virtual Firmware (TDVF). QEMU needs to parse TDVF and map it in TD +guest memory prior to running the TDX VM. + +A TDVF Metadata in TDVF image describes the structure of firmware. +QEMU refers to it to setup memory for TDVF. Introduce function +tdvf_parse_metadata() to parse the metadata from TDVF image and store +the info of each TDVF section. + +TDX metadata is located by a TDX metadata offset block, which is a +GUID-ed structure. The data portion of the GUID structure contains +only an 4-byte field that is the offset of TDX metadata to the end +of firmware file. + +Select X86_FW_OVMF when TDX is enable to leverage existing functions +to parse and search OVMF's GUID-ed structures. + +Signed-off-by: Isaku Yamahata +Co-developed-by: Xiaoyao Li +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-18-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit b65a6011d16c4f7cb2eb227ab1bc735850475288) +Signed-off-by: Paolo Bonzini + +Conflicts: system/ -> sysemu/ +--- + hw/i386/Kconfig | 1 + + hw/i386/meson.build | 1 + + hw/i386/tdvf.c | 188 +++++++++++++++++++++++++++++++++++++++++ + include/hw/i386/tdvf.h | 38 +++++++++ + 4 files changed, 228 insertions(+) + create mode 100644 hw/i386/tdvf.c + create mode 100644 include/hw/i386/tdvf.h + +diff --git a/hw/i386/Kconfig b/hw/i386/Kconfig +index edd61cd2aa..31e50e2ebf 100644 +--- a/hw/i386/Kconfig ++++ b/hw/i386/Kconfig +@@ -12,6 +12,7 @@ config SGX + + config TDX + bool ++ select X86_FW_OVMF + depends on KVM + + config PC +diff --git a/hw/i386/meson.build b/hw/i386/meson.build +index 03aad10df7..d6d8023664 100644 +--- a/hw/i386/meson.build ++++ b/hw/i386/meson.build +@@ -31,6 +31,7 @@ i386_ss.add(when: 'CONFIG_PC', if_true: files( + 'port92.c')) + i386_ss.add(when: 'CONFIG_X86_FW_OVMF', if_true: files('pc_sysfw_ovmf.c'), + if_false: files('pc_sysfw_ovmf-stubs.c')) ++i386_ss.add(when: 'CONFIG_TDX', if_true: files('tdvf.c')) + + subdir('kvm') + subdir('xen') +diff --git a/hw/i386/tdvf.c b/hw/i386/tdvf.c +new file mode 100644 +index 0000000000..824a387d42 +--- /dev/null ++++ b/hw/i386/tdvf.c +@@ -0,0 +1,188 @@ ++/* ++ * Copyright (c) 2025 Intel Corporation ++ * Author: Isaku Yamahata ++ * ++ * Xiaoyao Li ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#include "qemu/osdep.h" ++#include "qemu/error-report.h" ++ ++#include "hw/i386/pc.h" ++#include "hw/i386/tdvf.h" ++#include "sysemu/kvm.h" ++ ++#define TDX_METADATA_OFFSET_GUID "e47a6535-984a-4798-865e-4685a7bf8ec2" ++#define TDX_METADATA_VERSION 1 ++#define TDVF_SIGNATURE 0x46564454 /* TDVF as little endian */ ++#define TDVF_ALIGNMENT 4096 ++ ++/* ++ * the raw structs read from TDVF keeps the name convention in ++ * TDVF Design Guide spec. ++ */ ++typedef struct { ++ uint32_t DataOffset; ++ uint32_t RawDataSize; ++ uint64_t MemoryAddress; ++ uint64_t MemoryDataSize; ++ uint32_t Type; ++ uint32_t Attributes; ++} TdvfSectionEntry; ++ ++typedef struct { ++ uint32_t Signature; ++ uint32_t Length; ++ uint32_t Version; ++ uint32_t NumberOfSectionEntries; ++ TdvfSectionEntry SectionEntries[]; ++} TdvfMetadata; ++ ++struct tdx_metadata_offset { ++ uint32_t offset; ++}; ++ ++static TdvfMetadata *tdvf_get_metadata(void *flash_ptr, int size) ++{ ++ TdvfMetadata *metadata; ++ uint32_t offset = 0; ++ uint8_t *data; ++ ++ if ((uint32_t) size != size) { ++ return NULL; ++ } ++ ++ if (pc_system_ovmf_table_find(TDX_METADATA_OFFSET_GUID, &data, NULL)) { ++ offset = size - le32_to_cpu(((struct tdx_metadata_offset *)data)->offset); ++ ++ if (offset + sizeof(*metadata) > size) { ++ return NULL; ++ } ++ } else { ++ error_report("Cannot find TDX_METADATA_OFFSET_GUID"); ++ return NULL; ++ } ++ ++ metadata = flash_ptr + offset; ++ ++ /* Finally, verify the signature to determine if this is a TDVF image. */ ++ metadata->Signature = le32_to_cpu(metadata->Signature); ++ if (metadata->Signature != TDVF_SIGNATURE) { ++ error_report("Invalid TDVF signature in metadata!"); ++ return NULL; ++ } ++ ++ /* Sanity check that the TDVF doesn't overlap its own metadata. */ ++ metadata->Length = le32_to_cpu(metadata->Length); ++ if (offset + metadata->Length > size) { ++ return NULL; ++ } ++ ++ /* Only version 1 is supported/defined. */ ++ metadata->Version = le32_to_cpu(metadata->Version); ++ if (metadata->Version != TDX_METADATA_VERSION) { ++ return NULL; ++ } ++ ++ return metadata; ++} ++ ++static int tdvf_parse_and_check_section_entry(const TdvfSectionEntry *src, ++ TdxFirmwareEntry *entry) ++{ ++ entry->data_offset = le32_to_cpu(src->DataOffset); ++ entry->data_len = le32_to_cpu(src->RawDataSize); ++ entry->address = le64_to_cpu(src->MemoryAddress); ++ entry->size = le64_to_cpu(src->MemoryDataSize); ++ entry->type = le32_to_cpu(src->Type); ++ entry->attributes = le32_to_cpu(src->Attributes); ++ ++ /* sanity check */ ++ if (entry->size < entry->data_len) { ++ error_report("Broken metadata RawDataSize 0x%x MemoryDataSize 0x%lx", ++ entry->data_len, entry->size); ++ return -1; ++ } ++ if (!QEMU_IS_ALIGNED(entry->address, TDVF_ALIGNMENT)) { ++ error_report("MemoryAddress 0x%lx not page aligned", entry->address); ++ return -1; ++ } ++ if (!QEMU_IS_ALIGNED(entry->size, TDVF_ALIGNMENT)) { ++ error_report("MemoryDataSize 0x%lx not page aligned", entry->size); ++ return -1; ++ } ++ ++ switch (entry->type) { ++ case TDVF_SECTION_TYPE_BFV: ++ case TDVF_SECTION_TYPE_CFV: ++ /* The sections that must be copied from firmware image to TD memory */ ++ if (entry->data_len == 0) { ++ error_report("%d section with RawDataSize == 0", entry->type); ++ return -1; ++ } ++ break; ++ case TDVF_SECTION_TYPE_TD_HOB: ++ case TDVF_SECTION_TYPE_TEMP_MEM: ++ /* The sections that no need to be copied from firmware image */ ++ if (entry->data_len != 0) { ++ error_report("%d section with RawDataSize 0x%x != 0", ++ entry->type, entry->data_len); ++ return -1; ++ } ++ break; ++ default: ++ error_report("TDVF contains unsupported section type %d", entry->type); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++int tdvf_parse_metadata(TdxFirmware *fw, void *flash_ptr, int size) ++{ ++ g_autofree TdvfSectionEntry *sections = NULL; ++ TdvfMetadata *metadata; ++ ssize_t entries_size; ++ int i; ++ ++ metadata = tdvf_get_metadata(flash_ptr, size); ++ if (!metadata) { ++ return -EINVAL; ++ } ++ ++ /* load and parse metadata entries */ ++ fw->nr_entries = le32_to_cpu(metadata->NumberOfSectionEntries); ++ if (fw->nr_entries < 2) { ++ error_report("Invalid number of fw entries (%u) in TDVF Metadata", ++ fw->nr_entries); ++ return -EINVAL; ++ } ++ ++ entries_size = fw->nr_entries * sizeof(TdvfSectionEntry); ++ if (metadata->Length != sizeof(*metadata) + entries_size) { ++ error_report("TDVF metadata len (0x%x) mismatch, expected (0x%x)", ++ metadata->Length, ++ (uint32_t)(sizeof(*metadata) + entries_size)); ++ return -EINVAL; ++ } ++ ++ fw->entries = g_new(TdxFirmwareEntry, fw->nr_entries); ++ sections = g_new(TdvfSectionEntry, fw->nr_entries); ++ ++ memcpy(sections, (void *)metadata + sizeof(*metadata), entries_size); ++ ++ for (i = 0; i < fw->nr_entries; i++) { ++ if (tdvf_parse_and_check_section_entry(§ions[i], &fw->entries[i])) { ++ goto err; ++ } ++ } ++ ++ return 0; ++ ++err: ++ fw->entries = 0; ++ g_free(fw->entries); ++ return -EINVAL; ++} +diff --git a/include/hw/i386/tdvf.h b/include/hw/i386/tdvf.h +new file mode 100644 +index 0000000000..7ebcac42a3 +--- /dev/null ++++ b/include/hw/i386/tdvf.h +@@ -0,0 +1,38 @@ ++/* ++ * Copyright (c) 2025 Intel Corporation ++ * Author: Isaku Yamahata ++ * ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#ifndef HW_I386_TDVF_H ++#define HW_I386_TDVF_H ++ ++#include "qemu/osdep.h" ++ ++#define TDVF_SECTION_TYPE_BFV 0 ++#define TDVF_SECTION_TYPE_CFV 1 ++#define TDVF_SECTION_TYPE_TD_HOB 2 ++#define TDVF_SECTION_TYPE_TEMP_MEM 3 ++ ++#define TDVF_SECTION_ATTRIBUTES_MR_EXTEND (1U << 0) ++#define TDVF_SECTION_ATTRIBUTES_PAGE_AUG (1U << 1) ++ ++typedef struct TdxFirmwareEntry { ++ uint32_t data_offset; ++ uint32_t data_len; ++ uint64_t address; ++ uint64_t size; ++ uint32_t type; ++ uint32_t attributes; ++} TdxFirmwareEntry; ++ ++typedef struct TdxFirmware { ++ uint32_t nr_entries; ++ TdxFirmwareEntry *entries; ++} TdxFirmware; ++ ++int tdvf_parse_metadata(TdxFirmware *fw, void *flash_ptr, int size); ++ ++#endif /* HW_I386_TDVF_H */ +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Add-TDVF-memory-via-KVM_TDX_INIT_MEM_REGION.patch b/SOURCES/kvm-i386-tdx-Add-TDVF-memory-via-KVM_TDX_INIT_MEM_REGION.patch new file mode 100644 index 0000000..81944df --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Add-TDVF-memory-via-KVM_TDX_INIT_MEM_REGION.patch @@ -0,0 +1,106 @@ +From 62fd2fea6ebca35e3bd12685ce5b10635375968b Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:25:28 +0200 +Subject: [PATCH 053/115] i386/tdx: Add TDVF memory via KVM_TDX_INIT_MEM_REGION + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [53/115] 44c8cbffafa5b307c5e28c6ad76abb496287cf47 (bonzini/rhel-qemu-kvm) + +TDVF firmware (CODE and VARS) needs to be copied to TD's private +memory via KVM_TDX_INIT_MEM_REGION, as well as TD HOB and TEMP memory. + +If the TDVF section has TDVF_SECTION_ATTRIBUTES_MR_EXTEND set in the +flag, calling KVM_TDX_EXTEND_MEMORY to extend the measurement. + +After populating the TDVF memory, the original image located in shared +ramblock can be discarded. + +Signed-off-by: Isaku Yamahata +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-25-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit ebc2d2b497c59414ac3c91de32bc546d27940e74) +Signed-off-by: Paolo Bonzini + +Conflicts: system/ -> sysemu/,exec/ +--- + target/i386/kvm/tdx.c | 42 ++++++++++++++++++++++++++++++++++++++++++ + 1 file changed, 42 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index db5d58b600..8f0826ac11 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -17,6 +17,7 @@ + #include "qom/object_interfaces.h" + #include "crypto/hash.h" + #include "sysemu/sysemu.h" ++#include "exec/ramblock.h" + + #include "hw/i386/e820_memory_layout.h" + #include "hw/i386/tdvf.h" +@@ -262,6 +263,9 @@ static void tdx_finalize_vm(Notifier *notifier, void *unused) + { + TdxFirmware *tdvf = &tdx_guest->tdvf; + TdxFirmwareEntry *entry; ++ RAMBlock *ram_block; ++ Error *local_err = NULL; ++ int r; + + tdx_init_ram_entries(); + +@@ -297,6 +301,44 @@ static void tdx_finalize_vm(Notifier *notifier, void *unused) + sizeof(TdxRamEntry), &tdx_ram_entry_compare); + + tdvf_hob_create(tdx_guest, tdx_get_hob_entry(tdx_guest)); ++ ++ for_each_tdx_fw_entry(tdvf, entry) { ++ struct kvm_tdx_init_mem_region region; ++ uint32_t flags; ++ ++ region = (struct kvm_tdx_init_mem_region) { ++ .source_addr = (uint64_t)entry->mem_ptr, ++ .gpa = entry->address, ++ .nr_pages = entry->size >> 12, ++ }; ++ ++ flags = entry->attributes & TDVF_SECTION_ATTRIBUTES_MR_EXTEND ? ++ KVM_TDX_MEASURE_MEMORY_REGION : 0; ++ ++ do { ++ error_free(local_err); ++ local_err = NULL; ++ r = tdx_vcpu_ioctl(first_cpu, KVM_TDX_INIT_MEM_REGION, flags, ++ ®ion, &local_err); ++ } while (r == -EAGAIN || r == -EINTR); ++ if (r < 0) { ++ error_report_err(local_err); ++ exit(1); ++ } ++ ++ if (entry->type == TDVF_SECTION_TYPE_TD_HOB || ++ entry->type == TDVF_SECTION_TYPE_TEMP_MEM) { ++ qemu_ram_munmap(-1, entry->mem_ptr, entry->size); ++ entry->mem_ptr = NULL; ++ } ++ } ++ ++ /* ++ * TDVF image has been copied into private region above via ++ * KVM_MEMORY_MAPPING. It becomes useless. ++ */ ++ ram_block = tdx_guest->tdvf_mr->ram_block; ++ ram_block_discard_range(ram_block, 0, ram_block->max_length); + } + + static Notifier tdx_machine_done_notify = { +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Add-TDX-fixed1-bits-to-supported-CPUIDs.patch b/SOURCES/kvm-i386-tdx-Add-TDX-fixed1-bits-to-supported-CPUIDs.patch new file mode 100644 index 0000000..94b41df --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Add-TDX-fixed1-bits-to-supported-CPUIDs.patch @@ -0,0 +1,249 @@ +From 70f7099dfd895a7ffeee3f66e188c1853d3885d6 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 074/115] i386/tdx: Add TDX fixed1 bits to supported CPUIDs + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [74/115] 81fb39c826363f892e46386883c69c0e862d7850 (bonzini/rhel-qemu-kvm) + +TDX architecture forcibly sets some CPUID bits for TD guest that VMM +cannot disable it. They are fixed1 bits. + +Fixed1 bits are not covered by tdx_caps.cpuid (which only contains the +directly configurable bits), while fixed1 bits are supported for TD guest +obviously. + +Add fixed1 bits to tdx_supported_cpuid. Besides, set all the fixed1 +bits to the initial set of KVM's support since KVM might not report them +as supported. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-46-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 0ba06e46d09b84a2cb97a268da5576aaca3a24ca) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.h | 2 + + target/i386/kvm/kvm_i386.h | 7 ++ + target/i386/kvm/tdx.c | 134 +++++++++++++++++++++++++++++++++++++ + target/i386/sev.c | 8 --- + 4 files changed, 143 insertions(+), 8 deletions(-) + +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index 601e828577..529f24df00 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -924,6 +924,8 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w); + #define CPUID_7_0_EDX_FSRM (1U << 4) + /* AVX512 Vector Pair Intersection to a Pair of Mask Registers */ + #define CPUID_7_0_EDX_AVX512_VP2INTERSECT (1U << 8) ++ /* "md_clear" VERW clears CPU buffers */ ++#define CPUID_7_0_EDX_MD_CLEAR (1U << 10) + /* SERIALIZE instruction */ + #define CPUID_7_0_EDX_SERIALIZE (1U << 14) + /* TSX Suspend Load Address Tracking instruction */ +diff --git a/target/i386/kvm/kvm_i386.h b/target/i386/kvm/kvm_i386.h +index 797610496a..f1d55d5b75 100644 +--- a/target/i386/kvm/kvm_i386.h ++++ b/target/i386/kvm/kvm_i386.h +@@ -44,6 +44,13 @@ void kvm_request_xsave_components(X86CPU *cpu, uint64_t mask); + + #ifdef CONFIG_KVM + ++#include ++ ++typedef struct KvmCpuidInfo { ++ struct kvm_cpuid2 cpuid; ++ struct kvm_cpuid_entry2 entries[KVM_MAX_CPUID_ENTRIES]; ++} KvmCpuidInfo; ++ + bool kvm_is_vm_type_supported(int type); + bool kvm_has_adjust_clock_stable(void); + bool kvm_has_exception_payload(void); +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 4949d01f22..6fa30c3ec4 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -367,6 +367,133 @@ static Notifier tdx_machine_done_notify = { + .notify = tdx_finalize_vm, + }; + ++/* ++ * Some CPUID bits change from fixed1 to configurable bits when TDX module ++ * supports TDX_FEATURES0.VE_REDUCTION. e.g., MCA/MCE/MTRR/CORE_CAPABILITY. ++ * ++ * To make QEMU work with all the versions of TDX module, keep the fixed1 bits ++ * here if they are ever fixed1 bits in any of the version though not fixed1 in ++ * the latest version. Otherwise, with the older version of TDX module, QEMU may ++ * treat the fixed1 bit as unsupported. ++ * ++ * For newer TDX module, it does no harm to keep them in tdx_fixed1_bits even ++ * though they changed to configurable bits. Because tdx_fixed1_bits is used to ++ * setup the supported bits. ++ */ ++KvmCpuidInfo tdx_fixed1_bits = { ++ .cpuid.nent = 8, ++ .entries[0] = { ++ .function = 0x1, ++ .index = 0, ++ .ecx = CPUID_EXT_SSE3 | CPUID_EXT_PCLMULQDQ | CPUID_EXT_DTES64 | ++ CPUID_EXT_DSCPL | CPUID_EXT_SSSE3 | CPUID_EXT_CX16 | ++ CPUID_EXT_PDCM | CPUID_EXT_PCID | CPUID_EXT_SSE41 | ++ CPUID_EXT_SSE42 | CPUID_EXT_X2APIC | CPUID_EXT_MOVBE | ++ CPUID_EXT_POPCNT | CPUID_EXT_AES | CPUID_EXT_XSAVE | ++ CPUID_EXT_RDRAND | CPUID_EXT_HYPERVISOR, ++ .edx = CPUID_FP87 | CPUID_VME | CPUID_DE | CPUID_PSE | CPUID_TSC | ++ CPUID_MSR | CPUID_PAE | CPUID_MCE | CPUID_CX8 | CPUID_APIC | ++ CPUID_SEP | CPUID_MTRR | CPUID_PGE | CPUID_MCA | CPUID_CMOV | ++ CPUID_PAT | CPUID_CLFLUSH | CPUID_DTS | CPUID_MMX | CPUID_FXSR | ++ CPUID_SSE | CPUID_SSE2, ++ }, ++ .entries[1] = { ++ .function = 0x6, ++ .index = 0, ++ .eax = CPUID_6_EAX_ARAT, ++ }, ++ .entries[2] = { ++ .function = 0x7, ++ .index = 0, ++ .flags = KVM_CPUID_FLAG_SIGNIFCANT_INDEX, ++ .ebx = CPUID_7_0_EBX_FSGSBASE | CPUID_7_0_EBX_FDP_EXCPTN_ONLY | ++ CPUID_7_0_EBX_SMEP | CPUID_7_0_EBX_INVPCID | ++ CPUID_7_0_EBX_ZERO_FCS_FDS | CPUID_7_0_EBX_RDSEED | ++ CPUID_7_0_EBX_SMAP | CPUID_7_0_EBX_CLFLUSHOPT | ++ CPUID_7_0_EBX_CLWB | CPUID_7_0_EBX_SHA_NI, ++ .ecx = CPUID_7_0_ECX_BUS_LOCK_DETECT | CPUID_7_0_ECX_MOVDIRI | ++ CPUID_7_0_ECX_MOVDIR64B, ++ .edx = CPUID_7_0_EDX_MD_CLEAR | CPUID_7_0_EDX_SPEC_CTRL | ++ CPUID_7_0_EDX_STIBP | CPUID_7_0_EDX_FLUSH_L1D | ++ CPUID_7_0_EDX_ARCH_CAPABILITIES | CPUID_7_0_EDX_CORE_CAPABILITY | ++ CPUID_7_0_EDX_SPEC_CTRL_SSBD, ++ }, ++ .entries[3] = { ++ .function = 0x7, ++ .index = 2, ++ .flags = KVM_CPUID_FLAG_SIGNIFCANT_INDEX, ++ .edx = CPUID_7_2_EDX_PSFD | CPUID_7_2_EDX_IPRED_CTRL | ++ CPUID_7_2_EDX_RRSBA_CTRL | CPUID_7_2_EDX_BHI_CTRL, ++ }, ++ .entries[4] = { ++ .function = 0xD, ++ .index = 0, ++ .flags = KVM_CPUID_FLAG_SIGNIFCANT_INDEX, ++ .eax = XSTATE_FP_MASK | XSTATE_SSE_MASK, ++ }, ++ .entries[5] = { ++ .function = 0xD, ++ .index = 1, ++ .flags = KVM_CPUID_FLAG_SIGNIFCANT_INDEX, ++ .eax = CPUID_XSAVE_XSAVEOPT | CPUID_XSAVE_XSAVEC| ++ CPUID_XSAVE_XGETBV1 | CPUID_XSAVE_XSAVES, ++ }, ++ .entries[6] = { ++ .function = 0x80000001, ++ .index = 0, ++ .ecx = CPUID_EXT3_LAHF_LM | CPUID_EXT3_ABM | CPUID_EXT3_3DNOWPREFETCH, ++ /* ++ * Strictly speaking, SYSCALL is not fixed1 bit since it depends on ++ * the CPU to be in 64-bit mode. But here fixed1 is used to serve the ++ * purpose of supported bits for TDX. In this sense, SYACALL is always ++ * supported. ++ */ ++ .edx = CPUID_EXT2_SYSCALL | CPUID_EXT2_NX | CPUID_EXT2_PDPE1GB | ++ CPUID_EXT2_RDTSCP | CPUID_EXT2_LM, ++ }, ++ .entries[7] = { ++ .function = 0x80000007, ++ .index = 0, ++ .edx = CPUID_APM_INVTSC, ++ }, ++}; ++ ++static struct kvm_cpuid_entry2 *find_in_supported_entry(uint32_t function, ++ uint32_t index) ++{ ++ struct kvm_cpuid_entry2 *e; ++ ++ e = cpuid_find_entry(tdx_supported_cpuid, function, index); ++ if (!e) { ++ if (tdx_supported_cpuid->nent >= KVM_MAX_CPUID_ENTRIES) { ++ error_report("tdx_supported_cpuid requries more space than %d entries", ++ KVM_MAX_CPUID_ENTRIES); ++ exit(1); ++ } ++ e = &tdx_supported_cpuid->entries[tdx_supported_cpuid->nent++]; ++ e->function = function; ++ e->index = index; ++ } ++ ++ return e; ++} ++ ++static void tdx_add_supported_cpuid_by_fixed1_bits(void) ++{ ++ struct kvm_cpuid_entry2 *e, *e1; ++ int i; ++ ++ for (i = 0; i < tdx_fixed1_bits.cpuid.nent; i++) { ++ e = &tdx_fixed1_bits.entries[i]; ++ ++ e1 = find_in_supported_entry(e->function, e->index); ++ e1->eax |= e->eax; ++ e1->ebx |= e->ebx; ++ e1->ecx |= e->ecx; ++ e1->edx |= e->edx; ++ } ++} ++ + static void tdx_setup_supported_cpuid(void) + { + if (tdx_supported_cpuid) { +@@ -379,6 +506,8 @@ static void tdx_setup_supported_cpuid(void) + memcpy(tdx_supported_cpuid->entries, tdx_caps->cpuid.entries, + tdx_caps->cpuid.nent * sizeof(struct kvm_cpuid_entry2)); + tdx_supported_cpuid->nent = tdx_caps->cpuid.nent; ++ ++ tdx_add_supported_cpuid_by_fixed1_bits(); + } + + static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) +@@ -463,6 +592,11 @@ static uint32_t tdx_adjust_cpuid_features(X86ConfidentialGuest *cg, + { + struct kvm_cpuid_entry2 *e; + ++ e = cpuid_find_entry(&tdx_fixed1_bits.cpuid, feature, index); ++ if (e) { ++ value |= cpuid_entry_get_reg(e, reg); ++ } ++ + if (is_feature_word_cpuid(feature, index, reg)) { + e = cpuid_find_entry(tdx_supported_cpuid, feature, index); + if (e) { +diff --git a/target/i386/sev.c b/target/i386/sev.c +index 24fcd078fc..edbad9bb92 100644 +--- a/target/i386/sev.c ++++ b/target/i386/sev.c +@@ -211,14 +211,6 @@ static const char *const sev_fw_errlist[] = { + + #define SEV_FW_MAX_ERROR ARRAY_SIZE(sev_fw_errlist) + +-/* doesn't expose this, so re-use the max from kvm.c */ +-#define KVM_MAX_CPUID_ENTRIES 100 +- +-typedef struct KvmCpuidInfo { +- struct kvm_cpuid2 cpuid; +- struct kvm_cpuid_entry2 entries[KVM_MAX_CPUID_ENTRIES]; +-} KvmCpuidInfo; +- + #define SNP_CPUID_FUNCTION_MAXCOUNT 64 + #define SNP_CPUID_FUNCTION_UNKNOWN 0xFFFFFFFF + +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Add-XFD-to-supported-bit-of-TDX.patch b/SOURCES/kvm-i386-tdx-Add-XFD-to-supported-bit-of-TDX.patch new file mode 100644 index 0000000..e3e00bf --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Add-XFD-to-supported-bit-of-TDX.patch @@ -0,0 +1,60 @@ +From 2c833c9f0b720bcf4267f9025344586b4eaa4906 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 077/115] i386/tdx: Add XFD to supported bit of TDX + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [77/115] 12485d15d9f665ad75d51f92380997df371180e4 (bonzini/rhel-qemu-kvm) + +Just mark XFD as always supported for TDX. This simple solution relies +on the fact KVM will report XFD as 0 when it's not supported by the +hardware. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-49-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 9f5771c57dbe92d46361afd992a5851c846d0322) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.h | 1 + + target/i386/kvm/tdx.c | 6 ++++++ + 2 files changed, 7 insertions(+) + +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index 3a7a409809..19645eb6f8 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -1100,6 +1100,7 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w); + #define CPUID_XSAVE_XSAVEC (1U << 1) + #define CPUID_XSAVE_XGETBV1 (1U << 2) + #define CPUID_XSAVE_XSAVES (1U << 3) ++#define CPUID_XSAVE_XFD (1U << 4) + + #define CPUID_6_EAX_ARAT (1U << 2) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index feb9cd7466..f15ed51a32 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -621,6 +621,12 @@ static void tdx_add_supported_cpuid_by_xfam(void) + e->edx |= (tdx_caps->supported_xfam & CPUID_XSTATE_XCR0_MASK) >> 32; + + e = find_in_supported_entry(0xd, 1); ++ /* ++ * Mark XFD always support for TDX, it will be cleared finally in ++ * tdx_adjust_cpuid_features() if XFD is unavailable on the hardware ++ * because in this case the original data has it as 0. ++ */ ++ e->eax |= CPUID_XSAVE_XFD; + e->ecx |= (tdx_caps->supported_xfam & CPUID_XSTATE_XSS_MASK); + e->edx |= (tdx_caps->supported_xfam & CPUID_XSTATE_XSS_MASK) >> 32; + } +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Add-property-sept-ve-disable-for-tdx-guest-.patch b/SOURCES/kvm-i386-tdx-Add-property-sept-ve-disable-for-tdx-guest-.patch new file mode 100644 index 0000000..17a6754 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Add-property-sept-ve-disable-for-tdx-guest-.patch @@ -0,0 +1,113 @@ +From b62e9f0a875520807daa28d0b808e2ba9d96ca79 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 038/115] i386/tdx: Add property sept-ve-disable for tdx-guest + object +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [38/115] 3a21966b474dde36a34aa215e45262ba83a809ac (bonzini/rhel-qemu-kvm) + +Bit 28 of TD attribute, named SEPT_VE_DISABLE. When set to 1, it disables +EPT violation conversion to #VE on guest TD access of PENDING pages. + +Some guest OS (e.g., Linux TD guest) may require this bit as 1. +Otherwise refuse to boot. + +Add sept-ve-disable property for tdx-guest object, for user to configure +this bit. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Acked-by: Markus Armbruster +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-10-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 6016e2972d94c90307b6caf55a8e3aee5424c09b) +Signed-off-by: Paolo Bonzini +--- + qapi/qom.json | 8 +++++++- + target/i386/kvm/tdx.c | 23 +++++++++++++++++++++++ + 2 files changed, 30 insertions(+), 1 deletion(-) + +diff --git a/qapi/qom.json b/qapi/qom.json +index 530efeb7c5..fefb54f90b 100644 +--- a/qapi/qom.json ++++ b/qapi/qom.json +@@ -1016,10 +1016,16 @@ + # @attributes: The 'attributes' of a TD guest that is passed to + # KVM_TDX_INIT_VM + # ++# @sept-ve-disable: toggle bit 28 of TD attributes to control disabling ++# of EPT violation conversion to #VE on guest TD access of PENDING ++# pages. Some guest OS (e.g., Linux TD guest) may require this to ++# be set, otherwise they refuse to boot. ++# + # Since: 10.1 + ## + { 'struct': 'TdxGuestProperties', +- 'data': { '*attributes': 'uint64' } } ++ 'data': { '*attributes': 'uint64', ++ '*sept-ve-disable': 'bool' } } + + ## + # @ThreadContextProperties: +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 8f02c76249..370bd86f2c 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -18,6 +18,8 @@ + #include "kvm_i386.h" + #include "tdx.h" + ++#define TDX_TD_ATTRIBUTES_SEPT_VE_DISABLE BIT_ULL(28) ++ + static TdxGuest *tdx_guest; + + static struct kvm_tdx_capabilities *tdx_caps; +@@ -252,6 +254,24 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) + return 0; + } + ++static bool tdx_guest_get_sept_ve_disable(Object *obj, Error **errp) ++{ ++ TdxGuest *tdx = TDX_GUEST(obj); ++ ++ return !!(tdx->attributes & TDX_TD_ATTRIBUTES_SEPT_VE_DISABLE); ++} ++ ++static void tdx_guest_set_sept_ve_disable(Object *obj, bool value, Error **errp) ++{ ++ TdxGuest *tdx = TDX_GUEST(obj); ++ ++ if (value) { ++ tdx->attributes |= TDX_TD_ATTRIBUTES_SEPT_VE_DISABLE; ++ } else { ++ tdx->attributes &= ~TDX_TD_ATTRIBUTES_SEPT_VE_DISABLE; ++ } ++} ++ + /* tdx guest */ + OBJECT_DEFINE_TYPE_WITH_INTERFACES(TdxGuest, + tdx_guest, +@@ -272,6 +292,9 @@ static void tdx_guest_init(Object *obj) + + object_property_add_uint64_ptr(obj, "attributes", &tdx->attributes, + OBJ_PROP_FLAG_READWRITE); ++ object_property_add_bool(obj, "sept-ve-disable", ++ tdx_guest_get_sept_ve_disable, ++ tdx_guest_set_sept_ve_disable); + } + + static void tdx_guest_finalize(Object *obj) +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Add-supported-CPUID-bits-related-to-TD-Attr.patch b/SOURCES/kvm-i386-tdx-Add-supported-CPUID-bits-related-to-TD-Attr.patch new file mode 100644 index 0000000..c6978ce --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Add-supported-CPUID-bits-related-to-TD-Attr.patch @@ -0,0 +1,144 @@ +From d228f2ad7fb96be3ec1fa3256ee0404cf7e2b094 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 075/115] i386/tdx: Add supported CPUID bits related to TD + Attributes + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [75/115] 095f42329711d2cb7f147856ebf2775a522fd8e3 (bonzini/rhel-qemu-kvm) + +For TDX, some CPUID feature bit is configured via TD attributes. They +are not covered by tdx_caps.cpuid (which only contians the directly +configurable CPUID bits), but they are actually supported when the +related attributre bit is supported. + +Note, LASS and KeyLocker are not supported by KVM for TDX, nor does +QEMU support it (see TDX_SUPPORTED_TD_ATTRS). They are defined in +tdx_attrs_maps[] for the completeness of the existing TD Attribute +bits that are related with CPUID features. + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20250508150002.689633-47-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 31df29c532a9ef473c6efd497950a620099bf1da) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.h | 4 +++ + target/i386/kvm/tdx.c | 60 +++++++++++++++++++++++++++++++++++++++++++ + 2 files changed, 64 insertions(+) + +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index 529f24df00..e02cb75619 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -903,6 +903,8 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w); + #define CPUID_7_0_ECX_LA57 (1U << 16) + /* Read Processor ID */ + #define CPUID_7_0_ECX_RDPID (1U << 22) ++/* KeyLocker */ ++#define CPUID_7_0_ECX_KeyLocker (1U << 23) + /* Bus Lock Debug Exception */ + #define CPUID_7_0_ECX_BUS_LOCK_DETECT (1U << 24) + /* Cache Line Demote Instruction */ +@@ -963,6 +965,8 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w); + #define CPUID_7_1_EAX_AVX_VNNI (1U << 4) + /* AVX512 BFloat16 Instruction */ + #define CPUID_7_1_EAX_AVX512_BF16 (1U << 5) ++/* Linear address space separation */ ++#define CPUID_7_1_EAX_LASS (1U << 6) + /* CMPCCXADD Instructions */ + #define CPUID_7_1_EAX_CMPCCXADD (1U << 7) + /* Fast Zero REP MOVS */ +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 6fa30c3ec4..60dd239c05 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -458,6 +458,34 @@ KvmCpuidInfo tdx_fixed1_bits = { + }, + }; + ++typedef struct TdxAttrsMap { ++ uint32_t attr_index; ++ uint32_t cpuid_leaf; ++ uint32_t cpuid_subleaf; ++ int cpuid_reg; ++ uint32_t feat_mask; ++} TdxAttrsMap; ++ ++static TdxAttrsMap tdx_attrs_maps[] = { ++ {.attr_index = 27, ++ .cpuid_leaf = 7, ++ .cpuid_subleaf = 1, ++ .cpuid_reg = R_EAX, ++ .feat_mask = CPUID_7_1_EAX_LASS,}, ++ ++ {.attr_index = 30, ++ .cpuid_leaf = 7, ++ .cpuid_subleaf = 0, ++ .cpuid_reg = R_ECX, ++ .feat_mask = CPUID_7_0_ECX_PKS,}, ++ ++ {.attr_index = 31, ++ .cpuid_leaf = 7, ++ .cpuid_subleaf = 0, ++ .cpuid_reg = R_ECX, ++ .feat_mask = CPUID_7_0_ECX_KeyLocker,}, ++}; ++ + static struct kvm_cpuid_entry2 *find_in_supported_entry(uint32_t function, + uint32_t index) + { +@@ -494,6 +522,37 @@ static void tdx_add_supported_cpuid_by_fixed1_bits(void) + } + } + ++static void tdx_add_supported_cpuid_by_attrs(void) ++{ ++ struct kvm_cpuid_entry2 *e; ++ TdxAttrsMap *map; ++ int i; ++ ++ for (i = 0; i < ARRAY_SIZE(tdx_attrs_maps); i++) { ++ map = &tdx_attrs_maps[i]; ++ if (!((1ULL << map->attr_index) & tdx_caps->supported_attrs)) { ++ continue; ++ } ++ ++ e = find_in_supported_entry(map->cpuid_leaf, map->cpuid_subleaf); ++ ++ switch(map->cpuid_reg) { ++ case R_EAX: ++ e->eax |= map->feat_mask; ++ break; ++ case R_EBX: ++ e->ebx |= map->feat_mask; ++ break; ++ case R_ECX: ++ e->ecx |= map->feat_mask; ++ break; ++ case R_EDX: ++ e->edx |= map->feat_mask; ++ break; ++ } ++ } ++} ++ + static void tdx_setup_supported_cpuid(void) + { + if (tdx_supported_cpuid) { +@@ -508,6 +567,7 @@ static void tdx_setup_supported_cpuid(void) + tdx_supported_cpuid->nent = tdx_caps->cpuid.nent; + + tdx_add_supported_cpuid_by_fixed1_bits(); ++ tdx_add_supported_cpuid_by_attrs(); + } + + static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Add-supported-CPUID-bits-relates-to-XFAM.patch b/SOURCES/kvm-i386-tdx-Add-supported-CPUID-bits-relates-to-XFAM.patch new file mode 100644 index 0000000..3913487 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Add-supported-CPUID-bits-relates-to-XFAM.patch @@ -0,0 +1,220 @@ +From 714abb122a2cc4819b05a3893dfd2c61a9204c5e Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 076/115] i386/tdx: Add supported CPUID bits relates to XFAM + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [76/115] 459d99074c90bfd8048585dec42749cb18493ee9 (bonzini/rhel-qemu-kvm) + +Some CPUID bits are controlled by XFAM. They are not covered by +tdx_caps.cpuid (which only contians the directly configurable bits), but +they are actually supported when the related XFAM bit is supported. + +Add these XFAM controlled bits to TDX supported CPUID bits based on the +supported_xfam. + +Besides, incorporate the supported_xfam into the supported CPUID leaf of +0xD. + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20250508150002.689633-48-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 8c94c84cb9e0140b48acc9c9d404525ca7ef7457) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 12 ------- + target/i386/cpu.h | 16 ++++++++++ + target/i386/kvm/tdx.c | 73 +++++++++++++++++++++++++++++++++++++++++++ + 3 files changed, 89 insertions(+), 12 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 2da456da64..cd6d9e8c1c 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -1660,15 +1660,6 @@ bool is_feature_word_cpuid(uint32_t feature, uint32_t index, int reg) + return false; + } + +-typedef struct FeatureMask { +- FeatureWord index; +- uint64_t mask; +-} FeatureMask; +- +-typedef struct FeatureDep { +- FeatureMask from, to; +-} FeatureDep; +- + static FeatureDep feature_dependencies[] = { + { + .from = { FEAT_7_0_EDX, CPUID_7_0_EDX_ARCH_CAPABILITIES }, +@@ -1837,9 +1828,6 @@ static const X86RegisterInfo32 x86_reg_info_32[CPU_NB_REGS32] = { + }; + #undef REGISTER + +-/* CPUID feature bits available in XSS */ +-#define CPUID_XSTATE_XSS_MASK (XSTATE_ARCH_LBR_MASK) +- + ExtSaveArea x86_ext_save_areas[XSAVE_STATE_AREA_COUNT] = { + [XSTATE_FP_BIT] = { + /* x87 FP state component is always enabled if XSAVE is supported */ +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index e02cb75619..3a7a409809 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -589,6 +589,7 @@ typedef enum X86Seg { + #define XSTATE_OPMASK_BIT 5 + #define XSTATE_ZMM_Hi256_BIT 6 + #define XSTATE_Hi16_ZMM_BIT 7 ++#define XSTATE_PT_BIT 8 + #define XSTATE_PKRU_BIT 9 + #define XSTATE_ARCH_LBR_BIT 15 + #define XSTATE_XTILE_CFG_BIT 17 +@@ -602,6 +603,7 @@ typedef enum X86Seg { + #define XSTATE_OPMASK_MASK (1ULL << XSTATE_OPMASK_BIT) + #define XSTATE_ZMM_Hi256_MASK (1ULL << XSTATE_ZMM_Hi256_BIT) + #define XSTATE_Hi16_ZMM_MASK (1ULL << XSTATE_Hi16_ZMM_BIT) ++#define XSTATE_PT_MASK (1ULL << XSTATE_PT_BIT) + #define XSTATE_PKRU_MASK (1ULL << XSTATE_PKRU_BIT) + #define XSTATE_ARCH_LBR_MASK (1ULL << XSTATE_ARCH_LBR_BIT) + #define XSTATE_XTILE_CFG_MASK (1ULL << XSTATE_XTILE_CFG_BIT) +@@ -624,6 +626,11 @@ typedef enum X86Seg { + XSTATE_Hi16_ZMM_MASK | XSTATE_PKRU_MASK | \ + XSTATE_XTILE_CFG_MASK | XSTATE_XTILE_DATA_MASK) + ++/* CPUID feature bits available in XSS */ ++#define CPUID_XSTATE_XSS_MASK (XSTATE_ARCH_LBR_MASK) ++ ++#define CPUID_XSTATE_MASK (CPUID_XSTATE_XCR0_MASK | CPUID_XSTATE_XSS_MASK) ++ + /* CPUID feature words */ + typedef enum FeatureWord { + FEAT_1_EDX, /* CPUID[1].EDX */ +@@ -671,6 +678,15 @@ typedef enum FeatureWord { + FEATURE_WORDS, + } FeatureWord; + ++typedef struct FeatureMask { ++ FeatureWord index; ++ uint64_t mask; ++} FeatureMask; ++ ++typedef struct FeatureDep { ++ FeatureMask from, to; ++} FeatureDep; ++ + typedef uint64_t FeatureWordArray[FEATURE_WORDS]; + uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w); + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 60dd239c05..feb9cd7466 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -23,6 +23,8 @@ + + #include + ++#include "cpu.h" ++#include "cpu-internal.h" + #include "hw/i386/e820_memory_layout.h" + #include "hw/i386/tdvf.h" + #include "hw/i386/x86.h" +@@ -486,6 +488,32 @@ static TdxAttrsMap tdx_attrs_maps[] = { + .feat_mask = CPUID_7_0_ECX_KeyLocker,}, + }; + ++typedef struct TdxXFAMDep { ++ int xfam_bit; ++ FeatureMask feat_mask; ++} TdxXFAMDep; ++ ++/* ++ * Note, only the CPUID bits whose virtualization type are "XFAM & Native" are ++ * defiend here. ++ * ++ * For those whose virtualization type are "XFAM & Configured & Native", they ++ * are reported as configurable bits. And they are not supported if not in the ++ * configureable bits list from KVM even if the corresponding XFAM bit is ++ * supported. ++ */ ++TdxXFAMDep tdx_xfam_deps[] = { ++ { XSTATE_YMM_BIT, { FEAT_1_ECX, CPUID_EXT_FMA }}, ++ { XSTATE_YMM_BIT, { FEAT_7_0_EBX, CPUID_7_0_EBX_AVX2 }}, ++ { XSTATE_OPMASK_BIT, { FEAT_7_0_ECX, CPUID_7_0_ECX_AVX512_VBMI}}, ++ { XSTATE_OPMASK_BIT, { FEAT_7_0_EDX, CPUID_7_0_EDX_AVX512_FP16}}, ++ { XSTATE_PT_BIT, { FEAT_7_0_EBX, CPUID_7_0_EBX_INTEL_PT}}, ++ { XSTATE_PKRU_BIT, { FEAT_7_0_ECX, CPUID_7_0_ECX_PKU}}, ++ { XSTATE_XTILE_CFG_BIT, { FEAT_7_0_EDX, CPUID_7_0_EDX_AMX_BF16 }}, ++ { XSTATE_XTILE_CFG_BIT, { FEAT_7_0_EDX, CPUID_7_0_EDX_AMX_TILE }}, ++ { XSTATE_XTILE_CFG_BIT, { FEAT_7_0_EDX, CPUID_7_0_EDX_AMX_INT8 }}, ++}; ++ + static struct kvm_cpuid_entry2 *find_in_supported_entry(uint32_t function, + uint32_t index) + { +@@ -553,6 +581,50 @@ static void tdx_add_supported_cpuid_by_attrs(void) + } + } + ++static void tdx_add_supported_cpuid_by_xfam(void) ++{ ++ struct kvm_cpuid_entry2 *e; ++ int i; ++ ++ const TdxXFAMDep *xfam_dep; ++ const FeatureWordInfo *f; ++ for (i = 0; i < ARRAY_SIZE(tdx_xfam_deps); i++) { ++ xfam_dep = &tdx_xfam_deps[i]; ++ if (!((1ULL << xfam_dep->xfam_bit) & tdx_caps->supported_xfam)) { ++ continue; ++ } ++ ++ f = &feature_word_info[xfam_dep->feat_mask.index]; ++ if (f->type != CPUID_FEATURE_WORD) { ++ continue; ++ } ++ ++ e = find_in_supported_entry(f->cpuid.eax, f->cpuid.ecx); ++ switch(f->cpuid.reg) { ++ case R_EAX: ++ e->eax |= xfam_dep->feat_mask.mask; ++ break; ++ case R_EBX: ++ e->ebx |= xfam_dep->feat_mask.mask; ++ break; ++ case R_ECX: ++ e->ecx |= xfam_dep->feat_mask.mask; ++ break; ++ case R_EDX: ++ e->edx |= xfam_dep->feat_mask.mask; ++ break; ++ } ++ } ++ ++ e = find_in_supported_entry(0xd, 0); ++ e->eax |= (tdx_caps->supported_xfam & CPUID_XSTATE_XCR0_MASK); ++ e->edx |= (tdx_caps->supported_xfam & CPUID_XSTATE_XCR0_MASK) >> 32; ++ ++ e = find_in_supported_entry(0xd, 1); ++ e->ecx |= (tdx_caps->supported_xfam & CPUID_XSTATE_XSS_MASK); ++ e->edx |= (tdx_caps->supported_xfam & CPUID_XSTATE_XSS_MASK) >> 32; ++} ++ + static void tdx_setup_supported_cpuid(void) + { + if (tdx_supported_cpuid) { +@@ -568,6 +640,7 @@ static void tdx_setup_supported_cpuid(void) + + tdx_add_supported_cpuid_by_fixed1_bits(); + tdx_add_supported_cpuid_by_attrs(); ++ tdx_add_supported_cpuid_by_xfam(); + } + + static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Call-KVM_TDX_INIT_VCPU-to-initialize-TDX-vc.patch b/SOURCES/kvm-i386-tdx-Call-KVM_TDX_INIT_VCPU-to-initialize-TDX-vc.patch new file mode 100644 index 0000000..171eb8f --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Call-KVM_TDX_INIT_VCPU-to-initialize-TDX-vc.patch @@ -0,0 +1,66 @@ +From 8d316b8468ec2e87ffc2e75c422698a2acdbcd16 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 054/115] i386/tdx: Call KVM_TDX_INIT_VCPU to initialize TDX + vcpu + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [54/115] 01b560fa024020f7be4e649cabd47f9f6bca920a (bonzini/rhel-qemu-kvm) + +TDX vcpu needs to be initialized by SEAMCALL(TDH.VP.INIT) and KVM +provides vcpu level IOCTL KVM_TDX_INIT_VCPU for it. + +KVM_TDX_INIT_VCPU needs the address of the HOB as input. Invoke it for +each vcpu after HOB list is created. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-26-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 41f7fd22073561a23229c0479d9d708dee9d3a1e) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 14 ++++++++++++++ + 1 file changed, 14 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 8f0826ac11..7980daf8c4 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -259,6 +259,18 @@ static void tdx_init_ram_entries(void) + tdx_guest->nr_ram_entries = j; + } + ++static void tdx_post_init_vcpus(void) ++{ ++ TdxFirmwareEntry *hob; ++ CPUState *cpu; ++ ++ hob = tdx_get_hob_entry(tdx_guest); ++ CPU_FOREACH(cpu) { ++ tdx_vcpu_ioctl(cpu, KVM_TDX_INIT_VCPU, 0, (void *)hob->address, ++ &error_fatal); ++ } ++} ++ + static void tdx_finalize_vm(Notifier *notifier, void *unused) + { + TdxFirmware *tdvf = &tdx_guest->tdvf; +@@ -302,6 +314,8 @@ static void tdx_finalize_vm(Notifier *notifier, void *unused) + + tdvf_hob_create(tdx_guest, tdx_get_hob_entry(tdx_guest)); + ++ tdx_post_init_vcpus(); ++ + for_each_tdx_fw_entry(tdvf, entry) { + struct kvm_tdx_init_mem_region region; + uint32_t flags; +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Clarify-the-error-message-of-mrconfigid-mro.patch b/SOURCES/kvm-i386-tdx-Clarify-the-error-message-of-mrconfigid-mro.patch new file mode 100644 index 0000000..543928e --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Clarify-the-error-message-of-mrconfigid-mro.patch @@ -0,0 +1,75 @@ +From 7dd643f8005449c3e34643f2cb85fcbab8011482 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:49 +0200 +Subject: [PATCH 091/115] i386/tdx: Clarify the error message of + mrconfigid/mrowner/mrownerconfig +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [91/115] 5f8559f8d9f842f79916f49f4806dd4cf2f8c686 (bonzini/rhel-qemu-kvm) + +The error message is misleading - we successfully decoded the data, +the decoded data was simply with the wrong length. + +Change the error message to show it is an length check failure with both +the received and expected values. + +Suggested-by: Daniel P. Berrangé +Signed-off-by: Xiaoyao Li +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Igor Mammedov +Link: https://lore.kernel.org/r/20250603050305.1704586-4-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 41cd354d350d3c64915be9c5decbf20abd84e486) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 12 +++++++++--- + 1 file changed, 9 insertions(+), 3 deletions(-) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index ca3641441c..ed3a55991a 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -1032,7 +1032,9 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) + return -1; + } + if (data_len != QCRYPTO_HASH_DIGEST_LEN_SHA384) { +- error_setg(errp, "TDX: failed to decode mrconfigid"); ++ error_setg(errp, "TDX 'mrconfigid' sha384 digest was %ld bytes, " ++ "expected %d bytes", data_len, ++ QCRYPTO_HASH_DIGEST_LEN_SHA384); + return -1; + } + memcpy(init_vm->mrconfigid, data, data_len); +@@ -1045,7 +1047,9 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) + return -1; + } + if (data_len != QCRYPTO_HASH_DIGEST_LEN_SHA384) { +- error_setg(errp, "TDX: failed to decode mrowner"); ++ error_setg(errp, "TDX 'mrowner' sha384 digest was %ld bytes, " ++ "expected %d bytes", data_len, ++ QCRYPTO_HASH_DIGEST_LEN_SHA384); + return -1; + } + memcpy(init_vm->mrowner, data, data_len); +@@ -1058,7 +1062,9 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) + return -1; + } + if (data_len != QCRYPTO_HASH_DIGEST_LEN_SHA384) { +- error_setg(errp, "TDX: failed to decode mrownerconfig"); ++ error_setg(errp, "TDX 'mrownerconfig' sha384 digest was %ld bytes, " ++ "expected %d bytes", data_len, ++ QCRYPTO_HASH_DIGEST_LEN_SHA384); + return -1; + } + memcpy(init_vm->mrownerconfig, data, data_len); +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Define-supported-KVM-features-for-TDX.patch b/SOURCES/kvm-i386-tdx-Define-supported-KVM-features-for-TDX.patch new file mode 100644 index 0000000..8c8bc4e --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Define-supported-KVM-features-for-TDX.patch @@ -0,0 +1,80 @@ +From 25f3b21b6b1654d1ffde72e231a4635b5929a6a8 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 078/115] i386/tdx: Define supported KVM features for TDX + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [78/115] d0ab8e6f6795d750c9fd07bf19bec65078424075 (bonzini/rhel-qemu-kvm) + +For TDX, only limited KVM PV features are supported. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-50-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 4d6e288a350a977b0fb0613db952087928ccd93e) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 20 ++++++++++++++++++++ + 1 file changed, 20 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index f15ed51a32..32e03caf43 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -32,6 +32,8 @@ + #include "kvm_i386.h" + #include "tdx.h" + ++#include "standard-headers/asm-x86/kvm_para.h" ++ + #define TDX_MIN_TSC_FREQUENCY_KHZ (100 * 1000) + #define TDX_MAX_TSC_FREQUENCY_KHZ (10 * 1000 * 1000) + +@@ -44,6 +46,14 @@ + TDX_TD_ATTRIBUTES_PKS | \ + TDX_TD_ATTRIBUTES_PERFMON) + ++#define TDX_SUPPORTED_KVM_FEATURES ((1U << KVM_FEATURE_NOP_IO_DELAY) | \ ++ (1U << KVM_FEATURE_PV_UNHALT) | \ ++ (1U << KVM_FEATURE_PV_TLB_FLUSH) | \ ++ (1U << KVM_FEATURE_PV_SEND_IPI) | \ ++ (1U << KVM_FEATURE_POLL_CONTROL) | \ ++ (1U << KVM_FEATURE_PV_SCHED_YIELD) | \ ++ (1U << KVM_FEATURE_MSI_EXT_DEST_ID)) ++ + static TdxGuest *tdx_guest; + + static struct kvm_tdx_capabilities *tdx_caps; +@@ -631,6 +641,14 @@ static void tdx_add_supported_cpuid_by_xfam(void) + e->edx |= (tdx_caps->supported_xfam & CPUID_XSTATE_XSS_MASK) >> 32; + } + ++static void tdx_add_supported_kvm_features(void) ++{ ++ struct kvm_cpuid_entry2 *e; ++ ++ e = find_in_supported_entry(0x40000001, 0); ++ e->eax = TDX_SUPPORTED_KVM_FEATURES; ++} ++ + static void tdx_setup_supported_cpuid(void) + { + if (tdx_supported_cpuid) { +@@ -647,6 +665,8 @@ static void tdx_setup_supported_cpuid(void) + tdx_add_supported_cpuid_by_fixed1_bits(); + tdx_add_supported_cpuid_by_attrs(); + tdx_add_supported_cpuid_by_xfam(); ++ ++ tdx_add_supported_kvm_features(); + } + + static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Disable-PIC-for-TDX-VMs.patch b/SOURCES/kvm-i386-tdx-Disable-PIC-for-TDX-VMs.patch new file mode 100644 index 0000000..abe8a91 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Disable-PIC-for-TDX-VMs.patch @@ -0,0 +1,56 @@ +From 52482dfcf0f97cd5db7a210497bdf45390cb7600 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 066/115] i386/tdx: Disable PIC for TDX VMs +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [66/115] 8f3badca2c7cb18d3d10bb1208396f63b7aeb47b (bonzini/rhel-qemu-kvm) + +Legacy PIC (8259) cannot be supported for TDX VMs since TDX module +doesn't allow directly interrupt injection. Using posted interrupts +for the PIC is not a viable option as the guest BIOS/kernel will not +do EOI for PIC IRQs, i.e. will leave the vIRR bit set. + +Hence disable PIC for TDX VMs and error out if user wants PIC. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-38-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit e7ef60892c80a9ce5b8504ceb13a81f4e0d4b3f7) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 7 +++++++ + 1 file changed, 7 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 9bd6843988..4f17e17308 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -381,6 +381,13 @@ static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) + return -EINVAL; + } + ++ if (x86ms->pic == ON_OFF_AUTO_AUTO) { ++ x86ms->pic = ON_OFF_AUTO_OFF; ++ } else if (x86ms->pic == ON_OFF_AUTO_ON) { ++ error_setg(errp, "TDX VM doesn't support PIC"); ++ return -EINVAL; ++ } ++ + if (!tdx_caps) { + r = get_tdx_capabilities(errp); + if (r) { +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Disable-SMM-for-TDX-VMs.patch b/SOURCES/kvm-i386-tdx-Disable-SMM-for-TDX-VMs.patch new file mode 100644 index 0000000..c836e5d --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Disable-SMM-for-TDX-VMs.patch @@ -0,0 +1,61 @@ +From e8e1554e0b626131745170c94a780b0d875aad63 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 065/115] i386/tdx: Disable SMM for TDX VMs +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [65/115] 69f950f26ffbea4f8ad6ff7f0be7ee3a4ce13c2b (bonzini/rhel-qemu-kvm) + +TDX doesn't support SMM and VMM cannot emulate SMM for TDX VMs because +VMM cannot manipulate TDX VM's memory. + +Disable SMM for TDX VMs and error out if user requests to enable SMM. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-37-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 810d4e83d07ca0d072205453a42c324a51d5a5fa) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 9 +++++++++ + 1 file changed, 9 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 7bc36b620e..9bd6843988 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -367,11 +367,20 @@ static Notifier tdx_machine_done_notify = { + + static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) + { ++ MachineState *ms = MACHINE(qdev_get_machine()); ++ X86MachineState *x86ms = X86_MACHINE(ms); + TdxGuest *tdx = TDX_GUEST(cgs); + int r = 0; + + kvm_mark_guest_state_protected(); + ++ if (x86ms->smm == ON_OFF_AUTO_AUTO) { ++ x86ms->smm = ON_OFF_AUTO_OFF; ++ } else if (x86ms->smm == ON_OFF_AUTO_ON) { ++ error_setg(errp, "TDX VM doesn't support SMM"); ++ return -EINVAL; ++ } ++ + if (!tdx_caps) { + r = get_tdx_capabilities(errp); + if (r) { +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Don-t-initialize-pc.rom-for-TDX-VMs.patch b/SOURCES/kvm-i386-tdx-Don-t-initialize-pc.rom-for-TDX-VMs.patch new file mode 100644 index 0000000..71bb3aa --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Don-t-initialize-pc.rom-for-TDX-VMs.patch @@ -0,0 +1,78 @@ +From 43d75b6ffdd152f55c366befb9eabbc8d0a7f6e7 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 048/115] i386/tdx: Don't initialize pc.rom for TDX VMs + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [48/115] 857ab7fccb138b45db15621688e412aef24b5821 (bonzini/rhel-qemu-kvm) + +For TDX, the address below 1MB are entirely general RAM. No need to +initialize pc.rom memory region for TDs. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-20-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 49b1f0f812372129736c1df0421c8f67d86d362b) +Signed-off-by: Paolo Bonzini +--- + hw/i386/pc.c | 29 ++++++++++++++++------------- + 1 file changed, 16 insertions(+), 13 deletions(-) + +diff --git a/hw/i386/pc.c b/hw/i386/pc.c +index 5237538640..057cd1fb86 100644 +--- a/hw/i386/pc.c ++++ b/hw/i386/pc.c +@@ -43,6 +43,7 @@ + #include "sysemu/xen.h" + #include "sysemu/reset.h" + #include "kvm/kvm_i386.h" ++#include "kvm/tdx.h" + #include "hw/xen/xen.h" + #include "qapi/qmp/qlist.h" + #include "qemu/error-report.h" +@@ -1131,21 +1132,23 @@ void pc_memory_init(PCMachineState *pcms, + /* Initialize PC system firmware */ + pc_system_firmware_init(pcms, rom_memory); + +- option_rom_mr = g_malloc(sizeof(*option_rom_mr)); +- if (machine_require_guest_memfd(machine)) { +- memory_region_init_ram_guest_memfd(option_rom_mr, NULL, "pc.rom", +- PC_ROM_SIZE, &error_fatal); +- } else { +- memory_region_init_ram(option_rom_mr, NULL, "pc.rom", PC_ROM_SIZE, +- &error_fatal); +- if (pcmc->pci_enabled) { +- memory_region_set_readonly(option_rom_mr, true); ++ if (!is_tdx_vm()) { ++ option_rom_mr = g_malloc(sizeof(*option_rom_mr)); ++ if (machine_require_guest_memfd(machine)) { ++ memory_region_init_ram_guest_memfd(option_rom_mr, NULL, "pc.rom", ++ PC_ROM_SIZE, &error_fatal); ++ } else { ++ memory_region_init_ram(option_rom_mr, NULL, "pc.rom", PC_ROM_SIZE, ++ &error_fatal); ++ if (pcmc->pci_enabled) { ++ memory_region_set_readonly(option_rom_mr, true); ++ } + } ++ memory_region_add_subregion_overlap(rom_memory, ++ PC_ROM_MIN_VGA, ++ option_rom_mr, ++ 1); + } +- memory_region_add_subregion_overlap(rom_memory, +- PC_ROM_MIN_VGA, +- option_rom_mr, +- 1); + + fw_cfg = fw_cfg_arch_create(machine, + x86ms->boot_cpus, x86ms->apic_id_limit); +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Don-t-mask-off-CPUID_EXT_PDCM.patch b/SOURCES/kvm-i386-tdx-Don-t-mask-off-CPUID_EXT_PDCM.patch new file mode 100644 index 0000000..5f806b7 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Don-t-mask-off-CPUID_EXT_PDCM.patch @@ -0,0 +1,59 @@ +From ec1ff403bb13fbb487a19a16c3da299038fecf2c Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:50 +0200 +Subject: [PATCH 104/115] i386/tdx: Don't mask off CPUID_EXT_PDCM + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [104/115] 46c1e5c86031d5d9ab9082e2d0affb3ddf7d1fb2 (bonzini/rhel-qemu-kvm) + +It gets below warning when booting TDX VMs: + + warning: TDX forcibly sets the feature: CPUID[eax=01h].ECX.pdcm [bit 15] + +Because CPUID_EXT_PDCM is fixed1 for TDX, and MSR_IA32_PERF_CAPABILITIES is +supported for TDX guest unconditioanlly. + +Don't mask off CPUID_EXT_PDCM for TDX. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250625035710.2770679-1-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 7ff24fb657d35c014f735f69aef03810fde607ab) +Signed-off-by: Paolo Bonzini + +Conflict: no call to mark_unavailable_features +--- + target/i386/cpu.c | 4 +++- + 1 file changed, 3 insertions(+), 1 deletion(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 2160754869..d0161f922c 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -27,6 +27,7 @@ + #include "sysemu/hvf.h" + #include "hvf/hvf-i386.h" + #include "kvm/kvm_i386.h" ++#include "kvm/tdx.h" + #include "sev.h" + #include "qapi/error.h" + #include "qemu/error-report.h" +@@ -8011,7 +8012,8 @@ void x86_cpu_expand_features(X86CPU *cpu, Error **errp) + } + } + +- if (!cpu->enable_pmu) { ++ /* PDCM is fixed1 bit for TDX */ ++ if (!cpu->enable_pmu && !is_tdx_vm()) { + env->features[FEAT_1_ECX] &= ~CPUID_EXT_PDCM; + } + +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Don-t-synchronize-guest-tsc-for-TDs.patch b/SOURCES/kvm-i386-tdx-Don-t-synchronize-guest-tsc-for-TDs.patch new file mode 100644 index 0000000..91733a7 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Don-t-synchronize-guest-tsc-for-TDs.patch @@ -0,0 +1,46 @@ +From 6339bd26122c876a663e841810de28c064e09128 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 068/115] i386/tdx: Don't synchronize guest tsc for TDs + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [68/115] 56f6dc49e20933d8539b2413a902a2fbad2751e0 (bonzini/rhel-qemu-kvm) + +TSC of TDs is not accessible and KVM doesn't allow access of +MSR_IA32_TSC for TDs. To avoid the assert() in kvm_get_tsc, make +kvm_synchronize_all_tsc() noop for TDs, + +Signed-off-by: Isaku Yamahata +Reviewed-by: Connor Kuehl +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-40-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 0ed55865b49b703af93e160d48935812a7114e07) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/kvm.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index f4809ee004..0c47eef03c 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -319,7 +319,7 @@ void kvm_synchronize_all_tsc(void) + { + CPUState *cpu; + +- if (kvm_enabled()) { ++ if (kvm_enabled() && !is_tdx_vm()) { + CPU_FOREACH(cpu) { + run_on_cpu(cpu, do_kvm_synchronize_tsc, RUN_ON_CPU_NULL); + } +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Don-t-treat-SYSCALL-as-unavailable.patch b/SOURCES/kvm-i386-tdx-Don-t-treat-SYSCALL-as-unavailable.patch new file mode 100644 index 0000000..418f3b1 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Don-t-treat-SYSCALL-as-unavailable.patch @@ -0,0 +1,60 @@ +From fd2d6e5623acae8b935c2ee5cb03d1a1b1631bb9 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 081/115] i386/tdx: Don't treat SYSCALL as unavailable + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [81/115] f742fd685ae68e7828947fa1013a5bdb0b49750d (bonzini/rhel-qemu-kvm) + +On Intel CPU, the value of CPUID_EXT2_SYSCALL depends on the mode of +the vcpu. It's 0 outside 64-bit mode and 1 in 64-bit mode. + +The initial state of TDX vcpu is 32-bit protected mode. At the time of +calling KVM_TDX_GET_CPUID, vcpu hasn't started running so the value read +is 0. + +In reality, 64-bit mode should always be supported. So mark +CPUID_EXT2_SYSCALL always supported to avoid false warning. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-53-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit deb9db6fb789cfe80527b75983e86137589227a4) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 13 +++++++++++++ + 1 file changed, 13 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 01fff9a27a..3e23010094 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -845,6 +845,19 @@ static int tdx_check_features(X86ConfidentialGuest *cg, CPUState *cs) + continue; + } + ++ /* Fixup for special cases */ ++ switch (w) { ++ case FEAT_8000_0001_EDX: ++ /* ++ * Intel enumerates SYSCALL bit as 1 only when processor in 64-bit ++ * mode and before vcpu running it's not in 64-bit mode. ++ */ ++ actual |= CPUID_EXT2_SYSCALL; ++ break; ++ default: ++ break; ++ } ++ + requested = env->features[w]; + unavailable = requested & ~actual; + mark_unavailable_features(cpu, w, unavailable, unav_prefix); +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Enable-user-exit-on-KVM_HC_MAP_GPA_RANGE.patch b/SOURCES/kvm-i386-tdx-Enable-user-exit-on-KVM_HC_MAP_GPA_RANGE.patch new file mode 100644 index 0000000..f18fc54 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Enable-user-exit-on-KVM_HC_MAP_GPA_RANGE.patch @@ -0,0 +1,55 @@ +From b7e4b3d2a67cf3af83f9226a8f3b7b159d15fba1 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 056/115] i386/tdx: Enable user exit on KVM_HC_MAP_GPA_RANGE + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [56/115] 88e3d0cfdca0c78873aed608518d75ed1703c5fb (bonzini/rhel-qemu-kvm) + +KVM translates TDG.VP.VMCALL to KVM_HC_MAP_GPA_RANGE, and QEMU +needs to enable user exit on KVM_HC_MAP_GPA_RANGE in order to handle the +memory conversion requested by TD guest. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-28-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 1ff5048d74e661943260c33e864c4118acb37ab4) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 7 +++++++ + 1 file changed, 7 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 19ed1038a7..62c83394d0 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -19,6 +19,8 @@ + #include "sysemu/sysemu.h" + #include "exec/ramblock.h" + ++#include ++ + #include "hw/i386/e820_memory_layout.h" + #include "hw/i386/tdvf.h" + #include "hw/i386/x86.h" +@@ -376,6 +378,11 @@ static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) + } + } + ++ /* TDX relies on KVM_HC_MAP_GPA_RANGE to handle TDG.VP.VMCALL */ ++ if (!kvm_enable_hypercall(BIT_ULL(KVM_HC_MAP_GPA_RANGE))) { ++ return -EOPNOTSUPP; ++ } ++ + qemu_add_machine_init_done_notifier(&tdx_machine_done_notify); + + tdx_guest = tdx; +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Error-and-exit-when-named-cpu-model-is-requ.patch b/SOURCES/kvm-i386-tdx-Error-and-exit-when-named-cpu-model-is-requ.patch new file mode 100644 index 0000000..977700b --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Error-and-exit-when-named-cpu-model-is-requ.patch @@ -0,0 +1,59 @@ +From bbfdcd93ce27ea64b6f6854dfdb8635f107de76c Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:49 +0200 +Subject: [PATCH 088/115] i386/tdx: Error and exit when named cpu model is + requested + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [88/115] 7ffb866075f6c8951eee9fbbd6ff3a0362519c4a (bonzini/rhel-qemu-kvm) + +Currently, it gets below error when requesting any named cpu model with +"-cpu" to boot a TDX VM: + + qemu-system-x86_64: KVM_TDX_INIT_VM failed: Invalid argument + +It misleads people to think it's the bug of KVM or QEMU. It is just that +current QEMU doesn't support named cpu model for TDX. + +To support named cpu models for TDX guest, there are opens to be +finalized and needs a mount of additional work. + +For now, explicitly check the case when named cpu model is requested. +Error report a hint and exit. + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20250612133801.2238342-1-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 750560f8a832361cf5cc4cd7bc4f56e1e76206f6) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 6 ++++++ + 1 file changed, 6 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index cca2b11622..3099e40baa 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -739,8 +739,14 @@ static int tdx_kvm_type(X86ConfidentialGuest *cg) + + static void tdx_cpu_instance_init(X86ConfidentialGuest *cg, CPUState *cpu) + { ++ X86CPUClass *xcc = X86_CPU_GET_CLASS(cpu); + X86CPU *x86cpu = X86_CPU(cpu); + ++ if (xcc->model) { ++ error_report("Named cpu model is not supported for TDX yet!"); ++ exit(1); ++ } ++ + object_property_set_bool(OBJECT(cpu), "pmu", false, &error_abort); + + /* invtsc is fixed1 for TD guest */ +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Fetch-and-validate-CPUID-of-TD-guest.patch b/SOURCES/kvm-i386-tdx-Fetch-and-validate-CPUID-of-TD-guest.patch new file mode 100644 index 0000000..17bf4a5 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Fetch-and-validate-CPUID-of-TD-guest.patch @@ -0,0 +1,233 @@ +From 58e6218a7d4b00316cf4ccd6a394190169a4cc61 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 080/115] i386/tdx: Fetch and validate CPUID of TD guest + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [80/115] c2ba91b28a4d0191fb186aea069733b81ab56e0f (bonzini/rhel-qemu-kvm) + +Use KVM_TDX_GET_CPUID to get the CPUIDs that are managed and enfored +by TDX module for TD guest. Check QEMU's configuration against the +fetched data. + +Print wanring message when 1. a feature is not supported but requested +by QEMU or 2. QEMU doesn't want to expose a feature while it is enforced +enabled. + +- If cpu->enforced_cpuid is not set, prints the warning message of both +1) and 2) and tweak QEMU's configuration. + +- If cpu->enforced_cpuid is set, quit if any case of 1) or 2). + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20250508150002.689633-52-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit e3d1a4a6d1d61cf5fbd0e4b389cfb3976093739f) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 33 +++++++++++++- + target/i386/cpu.h | 7 +++ + target/i386/kvm/tdx.c | 101 ++++++++++++++++++++++++++++++++++++++++++ + 3 files changed, 139 insertions(+), 2 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index cd6d9e8c1c..433d0a0418 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -5937,8 +5937,8 @@ static bool x86_cpu_have_filtered_features(X86CPU *cpu) + return false; + } + +-static void mark_unavailable_features(X86CPU *cpu, FeatureWord w, uint64_t mask, +- const char *verbose_prefix) ++void mark_unavailable_features(X86CPU *cpu, FeatureWord w, uint64_t mask, ++ const char *verbose_prefix) + { + CPUX86State *env = &cpu->env; + FeatureWordInfo *f = &feature_word_info[w]; +@@ -5965,6 +5965,35 @@ static void mark_unavailable_features(X86CPU *cpu, FeatureWord w, uint64_t mask, + } + } + ++void mark_forced_on_features(X86CPU *cpu, FeatureWord w, uint64_t mask, ++ const char *verbose_prefix) ++{ ++ CPUX86State *env = &cpu->env; ++ FeatureWordInfo *f = &feature_word_info[w]; ++ int i; ++ ++ if (!cpu->force_features) { ++ env->features[w] |= mask; ++ } ++ ++ cpu->forced_on_features[w] |= mask; ++ ++ if (!verbose_prefix) { ++ return; ++ } ++ ++ for (i = 0; i < 64; ++i) { ++ if ((1ULL << i) & mask) { ++ g_autofree char *feat_word_str = feature_word_description(f); ++ warn_report("%s: %s%s%s [bit %d]", ++ verbose_prefix, ++ feat_word_str, ++ f->feat_names[i] ? "." : "", ++ f->feat_names[i] ? f->feat_names[i] : "", i); ++ } ++ } ++} ++ + static void x86_cpuid_version_get_family(Object *obj, Visitor *v, + const char *name, void *opaque, + Error **errp) +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index 19645eb6f8..2e73945b28 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -2144,6 +2144,9 @@ struct ArchCPU { + /* Features that were filtered out because of missing host capabilities */ + FeatureWordArray filtered_features; + ++ /* Features that are forced enabled by underlying hypervisor, e.g., TDX */ ++ FeatureWordArray forced_on_features; ++ + /* Enable PMU CPUID bits. This can't be enabled by default yet because + * it doesn't have ABI stability guarantees, as it passes all PMU CPUID + * bits returned by GET_SUPPORTED_CPUID (that depend on host CPU and kernel +@@ -2455,6 +2458,10 @@ void host_cpuid(uint32_t function, uint32_t count, + uint32_t *eax, uint32_t *ebx, uint32_t *ecx, uint32_t *edx); + bool cpu_has_x2apic_feature(CPUX86State *env); + bool is_feature_word_cpuid(uint32_t feature, uint32_t index, int reg); ++void mark_unavailable_features(X86CPU *cpu, FeatureWord w, uint64_t mask, ++ const char *verbose_prefix); ++void mark_forced_on_features(X86CPU *cpu, FeatureWord w, uint64_t mask, ++ const char *verbose_prefix); + + static inline bool x86_has_cpuid_0x1f(X86CPU *cpu) + { +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 32e03caf43..01fff9a27a 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -766,6 +766,106 @@ static uint32_t tdx_adjust_cpuid_features(X86ConfidentialGuest *cg, + return value; + } + ++static struct kvm_cpuid2 *tdx_fetch_cpuid(CPUState *cpu, int *ret) ++{ ++ struct kvm_cpuid2 *fetch_cpuid; ++ int size = KVM_MAX_CPUID_ENTRIES; ++ Error *local_err = NULL; ++ int r; ++ ++ do { ++ error_free(local_err); ++ local_err = NULL; ++ ++ fetch_cpuid = g_malloc0(sizeof(*fetch_cpuid) + ++ sizeof(struct kvm_cpuid_entry2) * size); ++ fetch_cpuid->nent = size; ++ r = tdx_vcpu_ioctl(cpu, KVM_TDX_GET_CPUID, 0, fetch_cpuid, &local_err); ++ if (r == -E2BIG) { ++ g_free(fetch_cpuid); ++ size = fetch_cpuid->nent; ++ } ++ } while (r == -E2BIG); ++ ++ if (r < 0) { ++ error_report_err(local_err); ++ *ret = r; ++ return NULL; ++ } ++ ++ return fetch_cpuid; ++} ++ ++static int tdx_check_features(X86ConfidentialGuest *cg, CPUState *cs) ++{ ++ uint64_t actual, requested, unavailable, forced_on; ++ g_autofree struct kvm_cpuid2 *fetch_cpuid; ++ const char *forced_on_prefix = NULL; ++ const char *unav_prefix = NULL; ++ struct kvm_cpuid_entry2 *entry; ++ X86CPU *cpu = X86_CPU(cs); ++ CPUX86State *env = &cpu->env; ++ FeatureWordInfo *wi; ++ FeatureWord w; ++ bool mismatch = false; ++ int r; ++ ++ fetch_cpuid = tdx_fetch_cpuid(cs, &r); ++ if (!fetch_cpuid) { ++ return r; ++ } ++ ++ if (cpu->check_cpuid || cpu->enforce_cpuid) { ++ unav_prefix = "TDX doesn't support requested feature"; ++ forced_on_prefix = "TDX forcibly sets the feature"; ++ } ++ ++ for (w = 0; w < FEATURE_WORDS; w++) { ++ wi = &feature_word_info[w]; ++ actual = 0; ++ ++ switch (wi->type) { ++ case CPUID_FEATURE_WORD: ++ entry = cpuid_find_entry(fetch_cpuid, wi->cpuid.eax, wi->cpuid.ecx); ++ if (!entry) { ++ /* ++ * If KVM doesn't report it means it's totally configurable ++ * by QEMU ++ */ ++ continue; ++ } ++ ++ actual = cpuid_entry_get_reg(entry, wi->cpuid.reg); ++ break; ++ case MSR_FEATURE_WORD: ++ /* ++ * TODO: ++ * validate MSR features when KVM has interface report them. ++ */ ++ continue; ++ } ++ ++ requested = env->features[w]; ++ unavailable = requested & ~actual; ++ mark_unavailable_features(cpu, w, unavailable, unav_prefix); ++ if (unavailable) { ++ mismatch = true; ++ } ++ ++ forced_on = actual & ~requested; ++ mark_forced_on_features(cpu, w, forced_on, forced_on_prefix); ++ if (forced_on) { ++ mismatch = true; ++ } ++ } ++ ++ if (cpu->enforce_cpuid && mismatch) { ++ return -EINVAL; ++ } ++ ++ return 0; ++} ++ + static int tdx_validate_attributes(TdxGuest *tdx, Error **errp) + { + if ((tdx->attributes & ~tdx_caps->supported_attrs)) { +@@ -1161,4 +1261,5 @@ static void tdx_guest_class_init(ObjectClass *oc, void *data) + x86_klass->kvm_type = tdx_kvm_type; + x86_klass->cpu_instance_init = tdx_cpu_instance_init; + x86_klass->adjust_cpuid_features = tdx_adjust_cpuid_features; ++ x86_klass->check_features = tdx_check_features; + } +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Finalize-TDX-VM.patch b/SOURCES/kvm-i386-tdx-Finalize-TDX-VM.patch new file mode 100644 index 0000000..38a3ebd --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Finalize-TDX-VM.patch @@ -0,0 +1,44 @@ +From 8468a1a6a23c02c4a4b09501d7702d7e44095df9 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 055/115] i386/tdx: Finalize TDX VM + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [55/115] 391331e4b8c4582cc4db040188c44ab5ff23987e (bonzini/rhel-qemu-kvm) + +Invoke KVM_TDX_FINALIZE_VM to finalize the TD's measurement and make +the TD vCPUs runnable once machine initialization is complete. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-27-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit ae60ff4e9f9e5790f79abf866ec67270c28ca477) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 3 +++ + 1 file changed, 3 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 7980daf8c4..19ed1038a7 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -353,6 +353,9 @@ static void tdx_finalize_vm(Notifier *notifier, void *unused) + */ + ram_block = tdx_guest->tdvf_mr->ram_block; + ram_block_discard_range(ram_block, 0, ram_block->max_length); ++ ++ tdx_vm_ioctl(KVM_TDX_FINALIZE_VM, 0, NULL, &error_fatal); ++ CONFIDENTIAL_GUEST_SUPPORT(tdx_guest)->ready = true; + } + + static Notifier tdx_machine_done_notify = { +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Fix-build-on-32-bit-host.patch b/SOURCES/kvm-i386-tdx-Fix-build-on-32-bit-host.patch new file mode 100644 index 0000000..5a6f2e6 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Fix-build-on-32-bit-host.patch @@ -0,0 +1,121 @@ +From 537c96692b8d9830e4712a74f17d294cfd43a2bb Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:49 +0200 +Subject: [PATCH 085/115] i386/tdx: Fix build on 32-bit host +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [85/115] e62d030c943c217f64f1488b4ef3f52e8d77cab9 (bonzini/rhel-qemu-kvm) + +Use PRI formats where required and fix pointer cast. + +Cc: Xiaoyao Li +Signed-off-by: Cédric Le Goater +Link: https://lore.kernel.org/r/20250602173101.1052983-2-clg@redhat.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit e7f926eb7f5b81c709313974b476ed181c9c76d5) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 26 +++++++++++++------------- + 1 file changed, 13 insertions(+), 13 deletions(-) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index b9c3ba3725..cca2b11622 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -284,7 +284,7 @@ static void tdx_post_init_vcpus(void) + + hob = tdx_get_hob_entry(tdx_guest); + CPU_FOREACH(cpu) { +- tdx_vcpu_ioctl(cpu, KVM_TDX_INIT_VCPU, 0, (void *)hob->address, ++ tdx_vcpu_ioctl(cpu, KVM_TDX_INIT_VCPU, 0, (void *)(uintptr_t)hob->address, + &error_fatal); + } + } +@@ -339,7 +339,7 @@ static void tdx_finalize_vm(Notifier *notifier, void *unused) + uint32_t flags; + + region = (struct kvm_tdx_init_mem_region) { +- .source_addr = (uint64_t)entry->mem_ptr, ++ .source_addr = (uintptr_t)entry->mem_ptr, + .gpa = entry->address, + .nr_pages = entry->size >> 12, + }; +@@ -893,16 +893,16 @@ static int tdx_check_features(X86ConfidentialGuest *cg, CPUState *cs) + static int tdx_validate_attributes(TdxGuest *tdx, Error **errp) + { + if ((tdx->attributes & ~tdx_caps->supported_attrs)) { +- error_setg(errp, "Invalid attributes 0x%lx for TDX VM " +- "(KVM supported: 0x%llx)", tdx->attributes, +- tdx_caps->supported_attrs); ++ error_setg(errp, "Invalid attributes 0x%"PRIx64" for TDX VM " ++ "(KVM supported: 0x%"PRIx64")", tdx->attributes, ++ (uint64_t)tdx_caps->supported_attrs); + return -1; + } + + if (tdx->attributes & ~TDX_SUPPORTED_TD_ATTRS) { + error_setg(errp, "Some QEMU unsupported TD attribute bits being " +- "requested: 0x%lx (QEMU supported: 0x%llx)", +- tdx->attributes, TDX_SUPPORTED_TD_ATTRS); ++ "requested: 0x%"PRIx64" (QEMU supported: 0x%"PRIx64")", ++ tdx->attributes, (uint64_t)TDX_SUPPORTED_TD_ATTRS); + return -1; + } + +@@ -931,8 +931,8 @@ static int setup_td_xfam(X86CPU *x86cpu, Error **errp) + env->features[FEAT_XSAVE_XSS_HI]; + + if (xfam & ~tdx_caps->supported_xfam) { +- error_setg(errp, "Invalid XFAM 0x%lx for TDX VM (supported: 0x%llx))", +- xfam, tdx_caps->supported_xfam); ++ error_setg(errp, "Invalid XFAM 0x%"PRIx64" for TDX VM (supported: 0x%"PRIx64"))", ++ xfam, (uint64_t)tdx_caps->supported_xfam); + return -1; + } + +@@ -999,14 +999,14 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) + + if (env->tsc_khz && (env->tsc_khz < TDX_MIN_TSC_FREQUENCY_KHZ || + env->tsc_khz > TDX_MAX_TSC_FREQUENCY_KHZ)) { +- error_setg(errp, "Invalid TSC %ld KHz, must specify cpu_frequency " ++ error_setg(errp, "Invalid TSC %"PRId64" KHz, must specify cpu_frequency " + "between [%d, %d] kHz", env->tsc_khz, + TDX_MIN_TSC_FREQUENCY_KHZ, TDX_MAX_TSC_FREQUENCY_KHZ); + return -EINVAL; + } + + if (env->tsc_khz % (25 * 1000)) { +- error_setg(errp, "Invalid TSC %ld KHz, it must be multiple of 25MHz", ++ error_setg(errp, "Invalid TSC %"PRId64" KHz, it must be multiple of 25MHz", + env->tsc_khz); + return -EINVAL; + } +@@ -1014,7 +1014,7 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) + /* it's safe even env->tsc_khz is 0. KVM uses host's tsc_khz in this case */ + r = kvm_vm_ioctl(kvm_state, KVM_SET_TSC_KHZ, env->tsc_khz); + if (r < 0) { +- error_setg_errno(errp, -r, "Unable to set TSC frequency to %ld kHz", ++ error_setg_errno(errp, -r, "Unable to set TSC frequency to %"PRId64" kHz", + env->tsc_khz); + return r; + } +@@ -1139,7 +1139,7 @@ int tdx_handle_report_fatal_error(X86CPU *cpu, struct kvm_run *run) + uint64_t gpa = -1ull; + + if (error_code & 0xffff) { +- error_report("TDX: REPORT_FATAL_ERROR: invalid error code: 0x%lx", ++ error_report("TDX: REPORT_FATAL_ERROR: invalid error code: 0x%"PRIx64, + error_code); + return -1; + } +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Fix-the-report-of-gpa-in-QAPI.patch b/SOURCES/kvm-i386-tdx-Fix-the-report-of-gpa-in-QAPI.patch new file mode 100644 index 0000000..d798c5e --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Fix-the-report-of-gpa-in-QAPI.patch @@ -0,0 +1,78 @@ +From 6c84c60f50287a00c91dd390b8d71f008c704048 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:50 +0200 +Subject: [PATCH 102/115] i386/tdx: Fix the report of gpa in QAPI +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [102/115] ccaf1112d985ac652bd118bce4486853163cbc16 (bonzini/rhel-qemu-kvm) + +Gpa is defined in QAPI but never reported to monitor because has_gpa is +never set to ture. + +Fix it by setting has_gpa to ture when TDX_REPORT_FATAL_ERROR_GPA_VALID +is set in error_code. + +Fixes: 6e250463b08b ("i386/tdx: Wire TDX_REPORT_FATAL_ERROR with GuestPanic facility") +Signed-off-by: Zhenzhong Duan +Reviewed-by: Daniel P. Berrangé +Link: https://lore.kernel.org/r/20250710035538.303136-1-zhenzhong.duan@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit b28f6d5c16f19f8c56926c10929db29f913895ad) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 8 ++++++-- + 1 file changed, 6 insertions(+), 2 deletions(-) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index e56db74f58..20fcd9a4c5 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -1321,7 +1321,8 @@ void tdx_handle_setup_event_notify_interrupt(X86CPU *cpu, struct kvm_run *run) + } + + static void tdx_panicked_on_fatal_error(X86CPU *cpu, uint64_t error_code, +- char *message, uint64_t gpa) ++ char *message, bool has_gpa, ++ uint64_t gpa) + { + GuestPanicInformation *panic_info; + +@@ -1330,6 +1331,7 @@ static void tdx_panicked_on_fatal_error(X86CPU *cpu, uint64_t error_code, + panic_info->u.tdx.error_code = (uint32_t) error_code; + panic_info->u.tdx.message = message; + panic_info->u.tdx.gpa = gpa; ++ panic_info->u.tdx.has_gpa = has_gpa; + + qemu_system_guest_panicked(panic_info); + } +@@ -1349,6 +1351,7 @@ int tdx_handle_report_fatal_error(X86CPU *cpu, struct kvm_run *run) + char *message = NULL; + uint64_t *tmp; + uint64_t gpa = -1ull; ++ bool has_gpa = false; + + if (error_code & 0xffff) { + error_report("TDX: REPORT_FATAL_ERROR: invalid error code: 0x%"PRIx64, +@@ -1381,9 +1384,10 @@ int tdx_handle_report_fatal_error(X86CPU *cpu, struct kvm_run *run) + + if (error_code & TDX_REPORT_FATAL_ERROR_GPA_VALID) { + gpa = run->system_event.data[R_R13]; ++ has_gpa = true; + } + +- tdx_panicked_on_fatal_error(cpu, error_code, message, gpa); ++ tdx_panicked_on_fatal_error(cpu, error_code, message, has_gpa, gpa); + + return -1; + } +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Fix-the-typo-of-the-comment-of-struct-TdxGu.patch b/SOURCES/kvm-i386-tdx-Fix-the-typo-of-the-comment-of-struct-TdxGu.patch new file mode 100644 index 0000000..d790d05 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Fix-the-typo-of-the-comment-of-struct-TdxGu.patch @@ -0,0 +1,50 @@ +From 3d855c5ffd0b235642c334ea3e9680451e2e50a6 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:49 +0200 +Subject: [PATCH 090/115] i386/tdx: Fix the typo of the comment of struct + TdxGuest +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [90/115] 86befce58e11c4eba084141ec177648b901dd165 (bonzini/rhel-qemu-kvm) + +Change sha348 to sha384. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Igor Mammedov +Link: https://lore.kernel.org/r/20250603050305.1704586-3-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit a38da9f4876bb17d7ed9c6e24964b12b61877d38) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.h | 6 +++--- + 1 file changed, 3 insertions(+), 3 deletions(-) + +diff --git a/target/i386/kvm/tdx.h b/target/i386/kvm/tdx.h +index 04b5afe199..8dd66e9014 100644 +--- a/target/i386/kvm/tdx.h ++++ b/target/i386/kvm/tdx.h +@@ -40,9 +40,9 @@ typedef struct TdxGuest { + bool initialized; + uint64_t attributes; /* TD attributes */ + uint64_t xfam; +- char *mrconfigid; /* base64 encoded sha348 digest */ +- char *mrowner; /* base64 encoded sha348 digest */ +- char *mrownerconfig; /* base64 encoded sha348 digest */ ++ char *mrconfigid; /* base64 encoded sha384 digest */ ++ char *mrowner; /* base64 encoded sha384 digest */ ++ char *mrownerconfig; /* base64 encoded sha384 digest */ + + MemoryRegion *tdvf_mr; + TdxFirmware tdvf; +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Force-exposing-CPUID-0x1f.patch b/SOURCES/kvm-i386-tdx-Force-exposing-CPUID-0x1f.patch new file mode 100644 index 0000000..cfc0e54 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Force-exposing-CPUID-0x1f.patch @@ -0,0 +1,45 @@ +From 06700e387cf55d20af5fa85245b189408f35a851 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 063/115] i386/tdx: Force exposing CPUID 0x1f + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [63/115] 45b63d9803eb03a2f9014c9a3eb084c2edf5f11d (bonzini/rhel-qemu-kvm) + +TDX uses CPUID 0x1f to configure TD guest's CPU topology. So set +enable_cpuid_0x1f for TDs. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-35-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 9002494f80b751a7655045c5f46bf90bc1d3bbd0) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 4 ++++ + 1 file changed, 4 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index afd7e62422..410f8a9997 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -400,7 +400,11 @@ static int tdx_kvm_type(X86ConfidentialGuest *cg) + + static void tdx_cpu_instance_init(X86ConfidentialGuest *cg, CPUState *cpu) + { ++ X86CPU *x86cpu = X86_CPU(cpu); ++ + object_property_set_bool(OBJECT(cpu), "pmu", false, &error_abort); ++ ++ x86cpu->enable_cpuid_0x1f = true; + } + + static int tdx_validate_attributes(TdxGuest *tdx, Error **errp) +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Get-tdx_capabilities-via-KVM_TDX_CAPABILITI.patch b/SOURCES/kvm-i386-tdx-Get-tdx_capabilities-via-KVM_TDX_CAPABILITI.patch new file mode 100644 index 0000000..256ddc3 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Get-tdx_capabilities-via-KVM_TDX_CAPABILITI.patch @@ -0,0 +1,193 @@ +From 5c6bd0700ee50d40a791c1a00c475f9042e23343 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:45 +0200 +Subject: [PATCH 034/115] i386/tdx: Get tdx_capabilities via + KVM_TDX_CAPABILITIES + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [34/115] bd4711c0cf17e73ea2f21417b6b031c9118a35c6 (bonzini/rhel-qemu-kvm) + +KVM provides TDX capabilities via sub command KVM_TDX_CAPABILITIES of +IOCTL(KVM_MEMORY_ENCRYPT_OP). Get the capabilities when initializing +TDX context. It will be used to validate user's setting later. + +Since there is no interface reporting how many cpuid configs contains in +KVM_TDX_CAPABILITIES, QEMU chooses to try starting with a known number +and abort when it exceeds KVM_MAX_CPUID_ENTRIES. + +Besides, introduce the interfaces to invoke TDX "ioctls" at VCPU scope +in preparation. + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20250508150002.689633-6-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 8eddedc3701d2190db976a05155a8263c8ec175b) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/kvm.c | 2 - + target/i386/kvm/kvm_i386.h | 2 + + target/i386/kvm/tdx.c | 107 ++++++++++++++++++++++++++++++++++++- + 3 files changed, 108 insertions(+), 3 deletions(-) + +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index 12da34b9c5..bf4493dd84 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -1771,8 +1771,6 @@ static int hyperv_init_vcpu(X86CPU *cpu) + + static Error *invtsc_mig_blocker; + +-#define KVM_MAX_CPUID_ENTRIES 100 +- + static void kvm_init_xsave(CPUX86State *env) + { + if (has_xsave2) { +diff --git a/target/i386/kvm/kvm_i386.h b/target/i386/kvm/kvm_i386.h +index 7edb154a16..499691e8d0 100644 +--- a/target/i386/kvm/kvm_i386.h ++++ b/target/i386/kvm/kvm_i386.h +@@ -13,6 +13,8 @@ + + #include "sysemu/kvm.h" + ++#define KVM_MAX_CPUID_ENTRIES 100 ++ + /* always false if !CONFIG_KVM */ + #define kvm_pit_in_kernel() \ + (kvm_irqchip_in_kernel() && !kvm_irqchip_is_split()) +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 4ff9486081..c67be5e618 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -10,17 +10,122 @@ + */ + + #include "qemu/osdep.h" ++#include "qemu/error-report.h" ++#include "qapi/error.h" + #include "qom/object_interfaces.h" + + #include "hw/i386/x86.h" + #include "kvm_i386.h" + #include "tdx.h" + ++static struct kvm_tdx_capabilities *tdx_caps; ++ ++enum tdx_ioctl_level { ++ TDX_VM_IOCTL, ++ TDX_VCPU_IOCTL, ++}; ++ ++static int tdx_ioctl_internal(enum tdx_ioctl_level level, void *state, ++ int cmd_id, __u32 flags, void *data, ++ Error **errp) ++{ ++ struct kvm_tdx_cmd tdx_cmd = {}; ++ int r; ++ ++ const char *tdx_ioctl_name[] = { ++ [KVM_TDX_CAPABILITIES] = "KVM_TDX_CAPABILITIES", ++ [KVM_TDX_INIT_VM] = "KVM_TDX_INIT_VM", ++ [KVM_TDX_INIT_VCPU] = "KVM_TDX_INIT_VCPU", ++ [KVM_TDX_INIT_MEM_REGION] = "KVM_TDX_INIT_MEM_REGION", ++ [KVM_TDX_FINALIZE_VM] = "KVM_TDX_FINALIZE_VM", ++ [KVM_TDX_GET_CPUID] = "KVM_TDX_GET_CPUID", ++ }; ++ ++ tdx_cmd.id = cmd_id; ++ tdx_cmd.flags = flags; ++ tdx_cmd.data = (__u64)(unsigned long)data; ++ ++ switch (level) { ++ case TDX_VM_IOCTL: ++ r = kvm_vm_ioctl(kvm_state, KVM_MEMORY_ENCRYPT_OP, &tdx_cmd); ++ break; ++ case TDX_VCPU_IOCTL: ++ r = kvm_vcpu_ioctl(state, KVM_MEMORY_ENCRYPT_OP, &tdx_cmd); ++ break; ++ default: ++ error_setg(errp, "Invalid tdx_ioctl_level %d", level); ++ return -EINVAL; ++ } ++ ++ if (r < 0) { ++ error_setg_errno(errp, -r, "TDX ioctl %s failed, hw_errors: 0x%llx", ++ tdx_ioctl_name[cmd_id], tdx_cmd.hw_error); ++ } ++ return r; ++} ++ ++static inline int tdx_vm_ioctl(int cmd_id, __u32 flags, void *data, ++ Error **errp) ++{ ++ return tdx_ioctl_internal(TDX_VM_IOCTL, NULL, cmd_id, flags, data, errp); ++} ++ ++static inline int tdx_vcpu_ioctl(CPUState *cpu, int cmd_id, __u32 flags, ++ void *data, Error **errp) ++{ ++ return tdx_ioctl_internal(TDX_VCPU_IOCTL, cpu, cmd_id, flags, data, errp); ++} ++ ++static int get_tdx_capabilities(Error **errp) ++{ ++ struct kvm_tdx_capabilities *caps; ++ /* 1st generation of TDX reports 6 cpuid configs */ ++ int nr_cpuid_configs = 6; ++ size_t size; ++ int r; ++ ++ do { ++ Error *local_err = NULL; ++ size = sizeof(struct kvm_tdx_capabilities) + ++ nr_cpuid_configs * sizeof(struct kvm_cpuid_entry2); ++ caps = g_malloc0(size); ++ caps->cpuid.nent = nr_cpuid_configs; ++ ++ r = tdx_vm_ioctl(KVM_TDX_CAPABILITIES, 0, caps, &local_err); ++ if (r == -E2BIG) { ++ g_free(caps); ++ nr_cpuid_configs *= 2; ++ if (nr_cpuid_configs > KVM_MAX_CPUID_ENTRIES) { ++ error_report("KVM TDX seems broken that number of CPUID entries" ++ " in kvm_tdx_capabilities exceeds limit: %d", ++ KVM_MAX_CPUID_ENTRIES); ++ error_propagate(errp, local_err); ++ return r; ++ } ++ error_free(local_err); ++ } else if (r < 0) { ++ g_free(caps); ++ error_propagate(errp, local_err); ++ return r; ++ } ++ } while (r == -E2BIG); ++ ++ tdx_caps = caps; ++ ++ return 0; ++} ++ + static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) + { ++ int r = 0; ++ + kvm_mark_guest_state_protected(); + +- return 0; ++ if (!tdx_caps) { ++ r = get_tdx_capabilities(errp); ++ } ++ ++ return r; + } + + static int tdx_kvm_type(X86ConfidentialGuest *cg) +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Handle-KVM_SYSTEM_EVENT_TDX_FATAL.patch b/SOURCES/kvm-i386-tdx-Handle-KVM_SYSTEM_EVENT_TDX_FATAL.patch new file mode 100644 index 0000000..257be66 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Handle-KVM_SYSTEM_EVENT_TDX_FATAL.patch @@ -0,0 +1,146 @@ +From 7c045d8fcd636f8e2e52f303fc668bff20bc6e5d Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 057/115] i386/tdx: Handle KVM_SYSTEM_EVENT_TDX_FATAL + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [57/115] 2b4612a6e7c55849578ab1dc803849190c80692d (bonzini/rhel-qemu-kvm) + +TD guest can use TDG.VP.VMCALL to request +termination. KVM translates such request into KVM_EXIT_SYSTEM_EVENT with +type of KVM_SYSTEM_EVENT_TDX_FATAL. + +Add hanlder for such exit. Parse and print the error message, and +terminate the TD guest in the handler. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-29-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 98dbfd6849f117de02ac6f513f2a1f95563e60ae) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/kvm.c | 10 +++++++++ + target/i386/kvm/tdx-stub.c | 5 +++++ + target/i386/kvm/tdx.c | 46 ++++++++++++++++++++++++++++++++++++++ + target/i386/kvm/tdx.h | 2 ++ + 4 files changed, 63 insertions(+) + +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index fa61221190..4bda4f5525 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -6019,6 +6019,16 @@ int kvm_arch_handle_exit(CPUState *cs, struct kvm_run *run) + case KVM_EXIT_HYPERCALL: + ret = kvm_handle_hypercall(run); + break; ++ case KVM_EXIT_SYSTEM_EVENT: ++ switch (run->system_event.type) { ++ case KVM_SYSTEM_EVENT_TDX_FATAL: ++ ret = tdx_handle_report_fatal_error(cpu, run); ++ break; ++ default: ++ ret = -1; ++ break; ++ } ++ break; + default: + fprintf(stderr, "KVM: unknown exit reason %d\n", run->exit_reason); + ret = -1; +diff --git a/target/i386/kvm/tdx-stub.c b/target/i386/kvm/tdx-stub.c +index 7748b6d0a4..720a4ff046 100644 +--- a/target/i386/kvm/tdx-stub.c ++++ b/target/i386/kvm/tdx-stub.c +@@ -13,3 +13,8 @@ int tdx_parse_tdvf(void *flash_ptr, int size) + { + return -EINVAL; + } ++ ++int tdx_handle_report_fatal_error(X86CPU *cpu, struct kvm_run *run) ++{ ++ return -EINVAL; ++} +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 62c83394d0..679613ab55 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -615,6 +615,52 @@ int tdx_parse_tdvf(void *flash_ptr, int size) + return tdvf_parse_metadata(&tdx_guest->tdvf, flash_ptr, size); + } + ++/* ++ * Only 8 registers can contain valid ASCII byte stream to form the fatal ++ * message, and their sequence is: R14, R15, RBX, RDI, RSI, R8, R9, RDX ++ */ ++#define TDX_FATAL_MESSAGE_MAX 64 ++ ++int tdx_handle_report_fatal_error(X86CPU *cpu, struct kvm_run *run) ++{ ++ uint64_t error_code = run->system_event.data[R_R12]; ++ uint64_t reg_mask = run->system_event.data[R_ECX]; ++ char *message = NULL; ++ uint64_t *tmp; ++ ++ if (error_code & 0xffff) { ++ error_report("TDX: REPORT_FATAL_ERROR: invalid error code: 0x%lx", ++ error_code); ++ return -1; ++ } ++ ++ if (reg_mask) { ++ message = g_malloc0(TDX_FATAL_MESSAGE_MAX + 1); ++ tmp = (uint64_t *)message; ++ ++#define COPY_REG(REG) \ ++ do { \ ++ if (reg_mask & BIT_ULL(REG)) { \ ++ *(tmp++) = run->system_event.data[REG]; \ ++ } \ ++ } while (0) ++ ++ COPY_REG(R_R14); ++ COPY_REG(R_R15); ++ COPY_REG(R_EBX); ++ COPY_REG(R_EDI); ++ COPY_REG(R_ESI); ++ COPY_REG(R_R8); ++ COPY_REG(R_R9); ++ COPY_REG(R_EDX); ++ *((char *)tmp) = '\0'; ++ } ++#undef COPY_REG ++ ++ error_report("TD guest reports fatal error. %s", message ? : ""); ++ return -1; ++} ++ + static bool tdx_guest_get_sept_ve_disable(Object *obj, Error **errp) + { + TdxGuest *tdx = TDX_GUEST(obj); +diff --git a/target/i386/kvm/tdx.h b/target/i386/kvm/tdx.h +index 36a7400e74..04b5afe199 100644 +--- a/target/i386/kvm/tdx.h ++++ b/target/i386/kvm/tdx.h +@@ -8,6 +8,7 @@ + #endif + + #include "confidential-guest.h" ++#include "cpu.h" + #include "hw/i386/tdvf.h" + + #define TYPE_TDX_GUEST "tdx-guest" +@@ -59,5 +60,6 @@ bool is_tdx_vm(void); + int tdx_pre_create_vcpu(CPUState *cpu, Error **errp); + void tdx_set_tdvf_region(MemoryRegion *tdvf_mr); + int tdx_parse_tdvf(void *flash_ptr, int size); ++int tdx_handle_report_fatal_error(X86CPU *cpu, struct kvm_run *run); + + #endif /* QEMU_I386_TDX_H */ +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Implement-adjust_cpuid_features-for-TDX.patch b/SOURCES/kvm-i386-tdx-Implement-adjust_cpuid_features-for-TDX.patch new file mode 100644 index 0000000..2100c04 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Implement-adjust_cpuid_features-for-TDX.patch @@ -0,0 +1,179 @@ +From 8f50b918a55f8a30e782eef028a625a209a19fde Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 073/115] i386/tdx: Implement adjust_cpuid_features() for TDX + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [73/115] 115b0cd2bbc276b2cf915e9ad48fd89c4ee7ce8e (bonzini/rhel-qemu-kvm) + +Maintain a TDX specific supported CPUID set, and use it to mask the +common supported CPUID value of KVM. It can avoid newly added supported +features (reported via KVM_GET_SUPPORTED_CPUID) for common VMs being +falsely reported as supported for TDX. + +As the first step, initialize the TDX supported CPUID set with all the +configurable CPUID bits. It's not complete because there are other CPUID +bits are supported for TDX but not reported as directly configurable. +E.g. the XFAM related bits, attribute related bits and fixed-1 bits. +They will be handled in the future. + +Also, what matters are the CPUID bits related to QEMU's feature word. +Only mask the CPUID leafs which are feature word leaf. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-45-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 75ec6189f5c65cab210dd9f16cf4eef368038d45) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 16 ++++++++++++++++ + target/i386/cpu.h | 1 + + target/i386/kvm/kvm.c | 2 +- + target/i386/kvm/kvm_i386.h | 1 + + target/i386/kvm/tdx.c | 34 ++++++++++++++++++++++++++++++++++ + 5 files changed, 53 insertions(+), 1 deletion(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index ab34626b19..2da456da64 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -1644,6 +1644,22 @@ FeatureWordInfo feature_word_info[FEATURE_WORDS] = { + }, + }; + ++bool is_feature_word_cpuid(uint32_t feature, uint32_t index, int reg) ++{ ++ FeatureWordInfo *wi; ++ FeatureWord w; ++ ++ for (w = 0; w < FEATURE_WORDS; w++) { ++ wi = &feature_word_info[w]; ++ if (wi->type == CPUID_FEATURE_WORD && wi->cpuid.eax == feature && ++ (!wi->cpuid.needs_ecx || wi->cpuid.ecx == index) && ++ wi->cpuid.reg == reg) { ++ return true; ++ } ++ } ++ return false; ++} ++ + typedef struct FeatureMask { + FeatureWord index; + uint64_t mask; +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index 83fa89bf0a..601e828577 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -2431,6 +2431,7 @@ void cpu_set_apic_feature(CPUX86State *env); + void host_cpuid(uint32_t function, uint32_t count, + uint32_t *eax, uint32_t *ebx, uint32_t *ecx, uint32_t *edx); + bool cpu_has_x2apic_feature(CPUX86State *env); ++bool is_feature_word_cpuid(uint32_t feature, uint32_t index, int reg); + + static inline bool x86_has_cpuid_0x1f(X86CPU *cpu) + { +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index 5349ff4db7..76352323e4 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -385,7 +385,7 @@ static bool host_tsx_broken(void) + + /* Returns the value for a specific register on the cpuid entry + */ +-static uint32_t cpuid_entry_get_reg(struct kvm_cpuid_entry2 *entry, int reg) ++uint32_t cpuid_entry_get_reg(struct kvm_cpuid_entry2 *entry, int reg) + { + uint32_t ret = 0; + switch (reg) { +diff --git a/target/i386/kvm/kvm_i386.h b/target/i386/kvm/kvm_i386.h +index f2fbf3f99f..797610496a 100644 +--- a/target/i386/kvm/kvm_i386.h ++++ b/target/i386/kvm/kvm_i386.h +@@ -62,6 +62,7 @@ void kvm_update_msi_routes_all(void *private, bool global, + struct kvm_cpuid_entry2 *cpuid_find_entry(struct kvm_cpuid2 *cpuid, + uint32_t function, + uint32_t index); ++uint32_t cpuid_entry_get_reg(struct kvm_cpuid_entry2 *entry, int reg); + uint32_t kvm_x86_build_cpuid(CPUX86State *env, struct kvm_cpuid_entry2 *entries, + uint32_t cpuid_i); + #endif /* CONFIG_KVM */ +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 58fb68cab0..4949d01f22 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -45,6 +45,7 @@ + static TdxGuest *tdx_guest; + + static struct kvm_tdx_capabilities *tdx_caps; ++static struct kvm_cpuid2 *tdx_supported_cpuid; + + /* Valid after kvm_arch_init()->confidential_guest_kvm_init()->tdx_kvm_init() */ + bool is_tdx_vm(void) +@@ -366,6 +367,20 @@ static Notifier tdx_machine_done_notify = { + .notify = tdx_finalize_vm, + }; + ++static void tdx_setup_supported_cpuid(void) ++{ ++ if (tdx_supported_cpuid) { ++ return; ++ } ++ ++ tdx_supported_cpuid = g_malloc0(sizeof(*tdx_supported_cpuid) + ++ KVM_MAX_CPUID_ENTRIES * sizeof(struct kvm_cpuid_entry2)); ++ ++ memcpy(tdx_supported_cpuid->entries, tdx_caps->cpuid.entries, ++ tdx_caps->cpuid.nent * sizeof(struct kvm_cpuid_entry2)); ++ tdx_supported_cpuid->nent = tdx_caps->cpuid.nent; ++} ++ + static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) + { + MachineState *ms = MACHINE(qdev_get_machine()); +@@ -403,6 +418,8 @@ static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) + } + } + ++ tdx_setup_supported_cpuid(); ++ + /* TDX relies on KVM_HC_MAP_GPA_RANGE to handle TDG.VP.VMCALL */ + if (!kvm_enable_hypercall(BIT_ULL(KVM_HC_MAP_GPA_RANGE))) { + return -EOPNOTSUPP; +@@ -440,6 +457,22 @@ static void tdx_cpu_instance_init(X86ConfidentialGuest *cg, CPUState *cpu) + x86cpu->enable_cpuid_0x1f = true; + } + ++static uint32_t tdx_adjust_cpuid_features(X86ConfidentialGuest *cg, ++ uint32_t feature, uint32_t index, ++ int reg, uint32_t value) ++{ ++ struct kvm_cpuid_entry2 *e; ++ ++ if (is_feature_word_cpuid(feature, index, reg)) { ++ e = cpuid_find_entry(tdx_supported_cpuid, feature, index); ++ if (e) { ++ value &= cpuid_entry_get_reg(e, reg); ++ } ++ } ++ ++ return value; ++} ++ + static int tdx_validate_attributes(TdxGuest *tdx, Error **errp) + { + if ((tdx->attributes & ~tdx_caps->supported_attrs)) { +@@ -834,4 +867,5 @@ static void tdx_guest_class_init(ObjectClass *oc, void *data) + klass->kvm_init = tdx_kvm_init; + x86_klass->kvm_type = tdx_kvm_type; + x86_klass->cpu_instance_init = tdx_cpu_instance_init; ++ x86_klass->adjust_cpuid_features = tdx_adjust_cpuid_features; + } +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Implement-tdx_kvm_init-to-initialize-TDX-VM.patch b/SOURCES/kvm-i386-tdx-Implement-tdx_kvm_init-to-initialize-TDX-VM.patch new file mode 100644 index 0000000..06bfecf --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Implement-tdx_kvm_init-to-initialize-TDX-VM.patch @@ -0,0 +1,92 @@ +From 08b7b3ce573d25b6130c28620c51ac57532bd454 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:45 +0200 +Subject: [PATCH 033/115] i386/tdx: Implement tdx_kvm_init() to initialize TDX + VM context +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [33/115] 1c018f800145aced56f1268aa0f70cbbe1ec4658 (bonzini/rhel-qemu-kvm) + +Implement TDX specific ConfidentialGuestSupportClass::kvm_init() +callback, tdx_kvm_init(). + +Mark guest state is proctected for TDX VM. More TDX specific +initialization will be added later. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-5-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 631a2ac5a4beab740b342367550562cd659b4c4a) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/kvm.c | 11 +---------- + target/i386/kvm/tdx.c | 10 ++++++++++ + 2 files changed, 11 insertions(+), 10 deletions(-) + +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index d7c2dc3c71..12da34b9c5 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -3021,16 +3021,7 @@ int kvm_arch_init(MachineState *ms, KVMState *s) + Error *local_err = NULL; + + /* +- * Initialize SEV context, if required +- * +- * If no memory encryption is requested (ms->cgs == NULL) this is +- * a no-op. +- * +- * It's also a no-op if a non-SEV confidential guest support +- * mechanism is selected. SEV is the only mechanism available to +- * select on x86 at present, so this doesn't arise, but if new +- * mechanisms are supported in future (e.g. TDX), they'll need +- * their own initialization either here or elsewhere. ++ * Initialize confidential guest (SEV/TDX) context, if required + */ + if (ms->cgs) { + ret = confidential_guest_kvm_init(ms->cgs, &local_err); +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index d785c1f6d1..4ff9486081 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -12,9 +12,17 @@ + #include "qemu/osdep.h" + #include "qom/object_interfaces.h" + ++#include "hw/i386/x86.h" + #include "kvm_i386.h" + #include "tdx.h" + ++static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) ++{ ++ kvm_mark_guest_state_protected(); ++ ++ return 0; ++} ++ + static int tdx_kvm_type(X86ConfidentialGuest *cg) + { + /* Do the object check */ +@@ -49,7 +57,9 @@ static void tdx_guest_finalize(Object *obj) + + static void tdx_guest_class_init(ObjectClass *oc, void *data) + { ++ ConfidentialGuestSupportClass *klass = CONFIDENTIAL_GUEST_SUPPORT_CLASS(oc); + X86ConfidentialGuestClass *x86_klass = X86_CONFIDENTIAL_GUEST_CLASS(oc); + ++ klass->kvm_init = tdx_kvm_init; + x86_klass->kvm_type = tdx_kvm_type; + } +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Implement-tdx_kvm_type-for-TDX.patch b/SOURCES/kvm-i386-tdx-Implement-tdx_kvm_type-for-TDX.patch new file mode 100644 index 0000000..5d42fcf --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Implement-tdx_kvm_type-for-TDX.patch @@ -0,0 +1,76 @@ +From c741697e7ae55ca5742e9518b8b5071b66b1eff0 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:45 +0200 +Subject: [PATCH 032/115] i386/tdx: Implement tdx_kvm_type() for TDX +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [32/115] db2604869e78aaf67a57ed2bfc34b15d7e9a830f (bonzini/rhel-qemu-kvm) + +TDX VM requires VM type to be KVM_X86_TDX_VM. Implement tdx_kvm_type() +as X86ConfidentialGuestClass->kvm_type. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-4-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit b455880e5515a9fc2b923bfc6c60bb54519b51d3) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/kvm.c | 1 + + target/i386/kvm/tdx.c | 12 ++++++++++++ + 2 files changed, 13 insertions(+) + +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index fe6b34bb10..d7c2dc3c71 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -183,6 +183,7 @@ static const char *vm_type_name[] = { + [KVM_X86_SEV_VM] = "SEV", + [KVM_X86_SEV_ES_VM] = "SEV-ES", + [KVM_X86_SNP_VM] = "SEV-SNP", ++ [KVM_X86_TDX_VM] = "TDX", + }; + + bool kvm_is_vm_type_supported(int type) +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index ec84ae2947..d785c1f6d1 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -12,8 +12,17 @@ + #include "qemu/osdep.h" + #include "qom/object_interfaces.h" + ++#include "kvm_i386.h" + #include "tdx.h" + ++static int tdx_kvm_type(X86ConfidentialGuest *cg) ++{ ++ /* Do the object check */ ++ TDX_GUEST(cg); ++ ++ return KVM_X86_TDX_VM; ++} ++ + /* tdx guest */ + OBJECT_DEFINE_TYPE_WITH_INTERFACES(TdxGuest, + tdx_guest, +@@ -40,4 +49,7 @@ static void tdx_guest_finalize(Object *obj) + + static void tdx_guest_class_init(ObjectClass *oc, void *data) + { ++ X86ConfidentialGuestClass *x86_klass = X86_CONFIDENTIAL_GUEST_CLASS(oc); ++ ++ x86_klass->kvm_type = tdx_kvm_type; + } +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Implement-user-specified-tsc-frequency.patch b/SOURCES/kvm-i386-tdx-Implement-user-specified-tsc-frequency.patch new file mode 100644 index 0000000..55898ec --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Implement-user-specified-tsc-frequency.patch @@ -0,0 +1,101 @@ +From a71b4e5d1ce8bcb7cf076500a1fe8871a674b03c Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 044/115] i386/tdx: Implement user specified tsc frequency +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [44/115] f12a5f1a60aa8bdd69d79e31b92dcf21380b7659 (bonzini/rhel-qemu-kvm) + +Reuse "-cpu,tsc-frequency=" to get user wanted tsc frequency and call VM +scope VM_SET_TSC_KHZ to set the tsc frequency of TD before KVM_TDX_INIT_VM. + +Besides, sanity check the tsc frequency to be in the legal range and +legal granularity (required by TDX module). + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-16-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 0e73b843616e52882940ab89e1b0e86e22be2162) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/kvm.c | 9 +++++++++ + target/i386/kvm/tdx.c | 25 +++++++++++++++++++++++++ + 2 files changed, 34 insertions(+) + +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index 3b71c182ab..fa61221190 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -861,6 +861,15 @@ static int kvm_arch_set_tsc_khz(CPUState *cs) + int r, cur_freq; + bool set_ioctl = false; + ++ /* ++ * TSC of TD vcpu is immutable, it cannot be set/changed via vcpu scope ++ * VM_SET_TSC_KHZ, but only be initialized via VM scope VM_SET_TSC_KHZ ++ * before ioctl KVM_TDX_INIT_VM in tdx_pre_create_vcpu() ++ */ ++ if (is_tdx_vm()) { ++ return 0; ++ } ++ + if (!env->tsc_khz) { + return 0; + } +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index c96e8eb7b8..56ad5f599d 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -20,6 +20,9 @@ + #include "kvm_i386.h" + #include "tdx.h" + ++#define TDX_MIN_TSC_FREQUENCY_KHZ (100 * 1000) ++#define TDX_MAX_TSC_FREQUENCY_KHZ (10 * 1000 * 1000) ++ + #define TDX_TD_ATTRIBUTES_DEBUG BIT_ULL(0) + #define TDX_TD_ATTRIBUTES_SEPT_VE_DISABLE BIT_ULL(28) + #define TDX_TD_ATTRIBUTES_PKS BIT_ULL(30) +@@ -267,6 +270,28 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) + return r; + } + ++ if (env->tsc_khz && (env->tsc_khz < TDX_MIN_TSC_FREQUENCY_KHZ || ++ env->tsc_khz > TDX_MAX_TSC_FREQUENCY_KHZ)) { ++ error_setg(errp, "Invalid TSC %ld KHz, must specify cpu_frequency " ++ "between [%d, %d] kHz", env->tsc_khz, ++ TDX_MIN_TSC_FREQUENCY_KHZ, TDX_MAX_TSC_FREQUENCY_KHZ); ++ return -EINVAL; ++ } ++ ++ if (env->tsc_khz % (25 * 1000)) { ++ error_setg(errp, "Invalid TSC %ld KHz, it must be multiple of 25MHz", ++ env->tsc_khz); ++ return -EINVAL; ++ } ++ ++ /* it's safe even env->tsc_khz is 0. KVM uses host's tsc_khz in this case */ ++ r = kvm_vm_ioctl(kvm_state, KVM_SET_TSC_KHZ, env->tsc_khz); ++ if (r < 0) { ++ error_setg_errno(errp, -r, "Unable to set TSC frequency to %ld kHz", ++ env->tsc_khz); ++ return r; ++ } ++ + if (tdx_guest->mrconfigid) { + g_autofree uint8_t *data = qbase64_decode(tdx_guest->mrconfigid, + strlen(tdx_guest->mrconfigid), &data_len, errp); +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Initialize-TDX-before-creating-TD-vcpus.patch b/SOURCES/kvm-i386-tdx-Initialize-TDX-before-creating-TD-vcpus.patch new file mode 100644 index 0000000..fb2e89b --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Initialize-TDX-before-creating-TD-vcpus.patch @@ -0,0 +1,283 @@ +From 7f2aa231529a03552d642eccae4c4cc209a3ccdb Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 037/115] i386/tdx: Initialize TDX before creating TD vcpus + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [37/115] 1626ea547590c6f25d7d51b66a23820a701398af (bonzini/rhel-qemu-kvm) + +Invoke KVM_TDX_INIT_VM in kvm_arch_pre_create_vcpu() that +KVM_TDX_INIT_VM configures global TD configurations, e.g. the canonical +CPUID config, and must be executed prior to creating vCPUs. + +Use kvm_x86_arch_cpuid() to setup the CPUID settings for TDX VM. + +Note, this doesn't address the fact that QEMU may change the CPUID +configuration when creating vCPUs, i.e. punts on refactoring QEMU to +provide a stable CPUID config prior to kvm_arch_init(). + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Acked-by: Markus Armbruster +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-9-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit f15898b0f50609d66465326221aa54b6699da674) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/kvm.c | 16 +++--- + target/i386/kvm/kvm_i386.h | 5 ++ + target/i386/kvm/meson.build | 2 +- + target/i386/kvm/tdx-stub.c | 10 ++++ + target/i386/kvm/tdx.c | 105 ++++++++++++++++++++++++++++++++++++ + target/i386/kvm/tdx.h | 6 +++ + 6 files changed, 137 insertions(+), 7 deletions(-) + create mode 100644 target/i386/kvm/tdx-stub.c + +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index 1fddec6b9c..3b71c182ab 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -38,6 +38,7 @@ + #include "kvm_i386.h" + #include "../confidential-guest.h" + #include "sev.h" ++#include "tdx.h" + #include "xen-emu.h" + #include "hyperv.h" + #include "hyperv-proto.h" +@@ -406,9 +407,9 @@ static uint32_t cpuid_entry_get_reg(struct kvm_cpuid_entry2 *entry, int reg) + + /* Find matching entry for function/index on kvm_cpuid2 struct + */ +-static struct kvm_cpuid_entry2 *cpuid_find_entry(struct kvm_cpuid2 *cpuid, +- uint32_t function, +- uint32_t index) ++struct kvm_cpuid_entry2 *cpuid_find_entry(struct kvm_cpuid2 *cpuid, ++ uint32_t function, ++ uint32_t index) + { + int i; + for (i = 0; i < cpuid->nent; ++i) { +@@ -1813,9 +1814,8 @@ static void kvm_init_nested_state(CPUX86State *env) + } + } + +-static uint32_t kvm_x86_build_cpuid(CPUX86State *env, +- struct kvm_cpuid_entry2 *entries, +- uint32_t cpuid_i) ++uint32_t kvm_x86_build_cpuid(CPUX86State *env, struct kvm_cpuid_entry2 *entries, ++ uint32_t cpuid_i) + { + uint32_t limit, i, j; + uint32_t unused; +@@ -2041,6 +2041,10 @@ full: + + int kvm_arch_pre_create_vcpu(CPUState *cpu, Error **errp) + { ++ if (is_tdx_vm()) { ++ return tdx_pre_create_vcpu(cpu, errp); ++ } ++ + return 0; + } + +diff --git a/target/i386/kvm/kvm_i386.h b/target/i386/kvm/kvm_i386.h +index 499691e8d0..f2fbf3f99f 100644 +--- a/target/i386/kvm/kvm_i386.h ++++ b/target/i386/kvm/kvm_i386.h +@@ -59,6 +59,11 @@ uint64_t kvm_swizzle_msi_ext_dest_id(uint64_t address); + void kvm_update_msi_routes_all(void *private, bool global, + uint32_t index, uint32_t mask); + ++struct kvm_cpuid_entry2 *cpuid_find_entry(struct kvm_cpuid2 *cpuid, ++ uint32_t function, ++ uint32_t index); ++uint32_t kvm_x86_build_cpuid(CPUX86State *env, struct kvm_cpuid_entry2 *entries, ++ uint32_t cpuid_i); + #endif /* CONFIG_KVM */ + + void kvm_pc_setup_irq_routing(bool pci_enabled); +diff --git a/target/i386/kvm/meson.build b/target/i386/kvm/meson.build +index 466bccb9cb..3f44cdedb7 100644 +--- a/target/i386/kvm/meson.build ++++ b/target/i386/kvm/meson.build +@@ -8,7 +8,7 @@ i386_kvm_ss.add(files( + + i386_kvm_ss.add(when: 'CONFIG_XEN_EMU', if_true: files('xen-emu.c')) + +-i386_kvm_ss.add(when: 'CONFIG_TDX', if_true: files('tdx.c')) ++i386_kvm_ss.add(when: 'CONFIG_TDX', if_true: files('tdx.c'), if_false: files('tdx-stub.c')) + + i386_system_ss.add(when: 'CONFIG_HYPERV', if_true: files('hyperv.c'), if_false: files('hyperv-stub.c')) + +diff --git a/target/i386/kvm/tdx-stub.c b/target/i386/kvm/tdx-stub.c +new file mode 100644 +index 0000000000..2344433594 +--- /dev/null ++++ b/target/i386/kvm/tdx-stub.c +@@ -0,0 +1,10 @@ ++/* SPDX-License-Identifier: GPL-2.0-or-later */ ++ ++#include "qemu/osdep.h" ++ ++#include "tdx.h" ++ ++int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) ++{ ++ return -EINVAL; ++} +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 16f67e18ae..8f02c76249 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -149,6 +149,109 @@ static int tdx_kvm_type(X86ConfidentialGuest *cg) + return KVM_X86_TDX_VM; + } + ++static int setup_td_xfam(X86CPU *x86cpu, Error **errp) ++{ ++ CPUX86State *env = &x86cpu->env; ++ uint64_t xfam; ++ ++ xfam = env->features[FEAT_XSAVE_XCR0_LO] | ++ env->features[FEAT_XSAVE_XCR0_HI] | ++ env->features[FEAT_XSAVE_XSS_LO] | ++ env->features[FEAT_XSAVE_XSS_HI]; ++ ++ if (xfam & ~tdx_caps->supported_xfam) { ++ error_setg(errp, "Invalid XFAM 0x%lx for TDX VM (supported: 0x%llx))", ++ xfam, tdx_caps->supported_xfam); ++ return -1; ++ } ++ ++ tdx_guest->xfam = xfam; ++ return 0; ++} ++ ++static void tdx_filter_cpuid(struct kvm_cpuid2 *cpuids) ++{ ++ int i, dest_cnt = 0; ++ struct kvm_cpuid_entry2 *src, *dest, *conf; ++ ++ for (i = 0; i < cpuids->nent; i++) { ++ src = cpuids->entries + i; ++ conf = cpuid_find_entry(&tdx_caps->cpuid, src->function, src->index); ++ if (!conf) { ++ continue; ++ } ++ dest = cpuids->entries + dest_cnt; ++ ++ dest->function = src->function; ++ dest->index = src->index; ++ dest->flags = src->flags; ++ dest->eax = src->eax & conf->eax; ++ dest->ebx = src->ebx & conf->ebx; ++ dest->ecx = src->ecx & conf->ecx; ++ dest->edx = src->edx & conf->edx; ++ ++ dest_cnt++; ++ } ++ cpuids->nent = dest_cnt++; ++} ++ ++int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) ++{ ++ X86CPU *x86cpu = X86_CPU(cpu); ++ CPUX86State *env = &x86cpu->env; ++ g_autofree struct kvm_tdx_init_vm *init_vm = NULL; ++ Error *local_err = NULL; ++ int retry = 10000; ++ int r = 0; ++ ++ QEMU_LOCK_GUARD(&tdx_guest->lock); ++ if (tdx_guest->initialized) { ++ return r; ++ } ++ ++ init_vm = g_malloc0(sizeof(struct kvm_tdx_init_vm) + ++ sizeof(struct kvm_cpuid_entry2) * KVM_MAX_CPUID_ENTRIES); ++ ++ r = setup_td_xfam(x86cpu, errp); ++ if (r) { ++ return r; ++ } ++ ++ init_vm->cpuid.nent = kvm_x86_build_cpuid(env, init_vm->cpuid.entries, 0); ++ tdx_filter_cpuid(&init_vm->cpuid); ++ ++ init_vm->attributes = tdx_guest->attributes; ++ init_vm->xfam = tdx_guest->xfam; ++ ++ /* ++ * KVM_TDX_INIT_VM gets -EAGAIN when KVM side SEAMCALL(TDH_MNG_CREATE) ++ * gets TDX_RND_NO_ENTROPY due to Random number generation (e.g., RDRAND or ++ * RDSEED) is busy. ++ * ++ * Retry for the case. ++ */ ++ do { ++ error_free(local_err); ++ local_err = NULL; ++ r = tdx_vm_ioctl(KVM_TDX_INIT_VM, 0, init_vm, &local_err); ++ } while (r == -EAGAIN && --retry); ++ ++ if (r < 0) { ++ if (!retry) { ++ error_append_hint(&local_err, "Hardware RNG (Random Number " ++ "Generator) is busy occupied by someone (via RDRAND/RDSEED) " ++ "maliciously, which leads to KVM_TDX_INIT_VM keeping failure " ++ "due to lack of entropy.\n"); ++ } ++ error_propagate(errp, local_err); ++ return r; ++ } ++ ++ tdx_guest->initialized = true; ++ ++ return 0; ++} ++ + /* tdx guest */ + OBJECT_DEFINE_TYPE_WITH_INTERFACES(TdxGuest, + tdx_guest, +@@ -162,6 +265,8 @@ static void tdx_guest_init(Object *obj) + ConfidentialGuestSupport *cgs = CONFIDENTIAL_GUEST_SUPPORT(obj); + TdxGuest *tdx = TDX_GUEST(obj); + ++ qemu_mutex_init(&tdx->lock); ++ + cgs->require_guest_memfd = true; + tdx->attributes = 0; + +diff --git a/target/i386/kvm/tdx.h b/target/i386/kvm/tdx.h +index de8ae91961..4e2b5c61ff 100644 +--- a/target/i386/kvm/tdx.h ++++ b/target/i386/kvm/tdx.h +@@ -19,7 +19,11 @@ typedef struct TdxGuestClass { + typedef struct TdxGuest { + X86ConfidentialGuest parent_obj; + ++ QemuMutex lock; ++ ++ bool initialized; + uint64_t attributes; /* TD attributes */ ++ uint64_t xfam; + } TdxGuest; + + #ifdef CONFIG_TDX +@@ -28,4 +32,6 @@ bool is_tdx_vm(void); + #define is_tdx_vm() 0 + #endif /* CONFIG_TDX */ + ++int tdx_pre_create_vcpu(CPUState *cpu, Error **errp); ++ + #endif /* QEMU_I386_TDX_H */ +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Introduce-is_tdx_vm-helper-and-cache-tdx_gu.patch b/SOURCES/kvm-i386-tdx-Introduce-is_tdx_vm-helper-and-cache-tdx_gu.patch new file mode 100644 index 0000000..d75bfaa --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Introduce-is_tdx_vm-helper-and-cache-tdx_gu.patch @@ -0,0 +1,104 @@ +From af03f2b600e6d02d86b21cc25bbeeaa35d104cfc Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:45 +0200 +Subject: [PATCH 035/115] i386/tdx: Introduce is_tdx_vm() helper and cache + tdx_guest object + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [35/115] 8e84ea1bc7eaf07ccfd915601b471cd75d01bf90 (bonzini/rhel-qemu-kvm) + +It will need special handling for TDX VMs all around the QEMU. +Introduce is_tdx_vm() helper to query if it's a TDX VM. + +Cache tdx_guest object thus no need to cast from ms->cgs every time. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Isaku Yamahata +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-7-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 1619d0e45be0d1e48a46d80963b4e77dc1b000a2) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 15 ++++++++++++++- + target/i386/kvm/tdx.h | 10 ++++++++++ + 2 files changed, 24 insertions(+), 1 deletion(-) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index c67be5e618..16f67e18ae 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -18,8 +18,16 @@ + #include "kvm_i386.h" + #include "tdx.h" + ++static TdxGuest *tdx_guest; ++ + static struct kvm_tdx_capabilities *tdx_caps; + ++/* Valid after kvm_arch_init()->confidential_guest_kvm_init()->tdx_kvm_init() */ ++bool is_tdx_vm(void) ++{ ++ return !!tdx_guest; ++} ++ + enum tdx_ioctl_level { + TDX_VM_IOCTL, + TDX_VCPU_IOCTL, +@@ -117,15 +125,20 @@ static int get_tdx_capabilities(Error **errp) + + static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) + { ++ TdxGuest *tdx = TDX_GUEST(cgs); + int r = 0; + + kvm_mark_guest_state_protected(); + + if (!tdx_caps) { + r = get_tdx_capabilities(errp); ++ if (r) { ++ return r; ++ } + } + +- return r; ++ tdx_guest = tdx; ++ return 0; + } + + static int tdx_kvm_type(X86ConfidentialGuest *cg) +diff --git a/target/i386/kvm/tdx.h b/target/i386/kvm/tdx.h +index f3b7253361..de8ae91961 100644 +--- a/target/i386/kvm/tdx.h ++++ b/target/i386/kvm/tdx.h +@@ -3,6 +3,10 @@ + #ifndef QEMU_I386_TDX_H + #define QEMU_I386_TDX_H + ++#ifndef CONFIG_USER_ONLY ++#include CONFIG_DEVICES /* CONFIG_TDX */ ++#endif ++ + #include "confidential-guest.h" + + #define TYPE_TDX_GUEST "tdx-guest" +@@ -18,4 +22,10 @@ typedef struct TdxGuest { + uint64_t attributes; /* TD attributes */ + } TdxGuest; + ++#ifdef CONFIG_TDX ++bool is_tdx_vm(void); ++#else ++#define is_tdx_vm() 0 ++#endif /* CONFIG_TDX */ ++ + #endif /* QEMU_I386_TDX_H */ +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Make-invtsc-default-on.patch b/SOURCES/kvm-i386-tdx-Make-invtsc-default-on.patch new file mode 100644 index 0000000..4e47392 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Make-invtsc-default-on.patch @@ -0,0 +1,42 @@ +From 0fe5ae12f6e427821edddf3bb1618aa0576f73cb Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 082/115] i386/tdx: Make invtsc default on + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [82/115] 24e58b9f91fca93c4e17de19592f4e232abee39a (bonzini/rhel-qemu-kvm) + +Because it's fixed1 bit that enforced by TDX module. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-54-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit ea4867b911fc2f6d4c8bd50ec62f0dc0fa190fab) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 3 +++ + 1 file changed, 3 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 3e23010094..3ec31d4872 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -742,6 +742,9 @@ static void tdx_cpu_instance_init(X86ConfidentialGuest *cg, CPUState *cpu) + + object_property_set_bool(OBJECT(cpu), "pmu", false, &error_abort); + ++ /* invtsc is fixed1 for TD guest */ ++ object_property_set_bool(OBJECT(cpu), "invtsc", true, &error_abort); ++ + x86cpu->enable_cpuid_0x1f = true; + } + +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Make-sept_ve_disable-set-by-default.patch b/SOURCES/kvm-i386-tdx-Make-sept_ve_disable-set-by-default.patch new file mode 100644 index 0000000..01c0b5d --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Make-sept_ve_disable-set-by-default.patch @@ -0,0 +1,48 @@ +From c0f849d9cc9b60aff17fd9a1349788856c0e98b9 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 039/115] i386/tdx: Make sept_ve_disable set by default +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [39/115] ccfc985bc607ff9d1c91e65d810003347d410088 (bonzini/rhel-qemu-kvm) + +For TDX KVM use case, Linux guest is the most major one. It requires +sept_ve_disable set. Make it default for the main use case. For other use +case, it can be enabled/disabled via qemu command line. + +Signed-off-by: Isaku Yamahata +Signed-off-by: Xiaoyao Li +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-11-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 714af52276e74a1829674d180ef26ecb6261834c) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 370bd86f2c..2ed40b7614 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -288,7 +288,7 @@ static void tdx_guest_init(Object *obj) + qemu_mutex_init(&tdx->lock); + + cgs->require_guest_memfd = true; +- tdx->attributes = 0; ++ tdx->attributes = TDX_TD_ATTRIBUTES_SEPT_VE_DISABLE; + + object_property_add_uint64_ptr(obj, "attributes", &tdx->attributes, + OBJ_PROP_FLAG_READWRITE); +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Only-configure-MSR_IA32_UCODE_REV-in-kvm_in.patch b/SOURCES/kvm-i386-tdx-Only-configure-MSR_IA32_UCODE_REV-in-kvm_in.patch new file mode 100644 index 0000000..d52c5f5 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Only-configure-MSR_IA32_UCODE_REV-in-kvm_in.patch @@ -0,0 +1,92 @@ +From 44207fd96718352fce68c68515fecf3ea4cee2e9 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 069/115] i386/tdx: Only configure MSR_IA32_UCODE_REV in + kvm_init_msrs() for TDs + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [69/115] eeff1e580ba5e6da1e2eaddeb51784999b994951 (bonzini/rhel-qemu-kvm) + +For TDs, only MSR_IA32_UCODE_REV in kvm_init_msrs() can be configured +by VMM, while the features enumerated/controlled by other MSRs except +MSR_IA32_UCODE_REV in kvm_init_msrs() are not under control of VMM. + +Only configure MSR_IA32_UCODE_REV for TDs. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-41-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit f9aaad3362a5886d78e7d4d50d563ac16c6acdde) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/kvm.c | 40 +++++++++++++++++++++------------------- + 1 file changed, 21 insertions(+), 19 deletions(-) + +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index 0c47eef03c..f3fe553151 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -3761,32 +3761,34 @@ static void kvm_init_msrs(X86CPU *cpu) + CPUX86State *env = &cpu->env; + + kvm_msr_buf_reset(cpu); +- if (has_msr_arch_capabs) { +- kvm_msr_entry_add(cpu, MSR_IA32_ARCH_CAPABILITIES, +- env->features[FEAT_ARCH_CAPABILITIES]); +- } + +- if (has_msr_core_capabs) { +- kvm_msr_entry_add(cpu, MSR_IA32_CORE_CAPABILITY, +- env->features[FEAT_CORE_CAPABILITY]); +- } ++ if (!is_tdx_vm()) { ++ if (has_msr_arch_capabs) { ++ kvm_msr_entry_add(cpu, MSR_IA32_ARCH_CAPABILITIES, ++ env->features[FEAT_ARCH_CAPABILITIES]); ++ } ++ ++ if (has_msr_core_capabs) { ++ kvm_msr_entry_add(cpu, MSR_IA32_CORE_CAPABILITY, ++ env->features[FEAT_CORE_CAPABILITY]); ++ } ++ ++ if (has_msr_perf_capabs && cpu->enable_pmu) { ++ kvm_msr_entry_add_perf(cpu, env->features); ++ } + +- if (has_msr_perf_capabs && cpu->enable_pmu) { +- kvm_msr_entry_add_perf(cpu, env->features); ++ /* ++ * Older kernels do not include VMX MSRs in KVM_GET_MSR_INDEX_LIST, but ++ * all kernels with MSR features should have them. ++ */ ++ if (kvm_feature_msrs && cpu_has_vmx(env)) { ++ kvm_msr_entry_add_vmx(cpu, env->features); ++ } + } + + if (has_msr_ucode_rev) { + kvm_msr_entry_add(cpu, MSR_IA32_UCODE_REV, cpu->ucode_rev); + } +- +- /* +- * Older kernels do not include VMX MSRs in KVM_GET_MSR_INDEX_LIST, but +- * all kernels with MSR features should have them. +- */ +- if (kvm_feature_msrs && cpu_has_vmx(env)) { +- kvm_msr_entry_add_vmx(cpu, env->features); +- } +- + assert(kvm_buf_set_msrs(cpu) == 0); + } + +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Parse-TDVF-metadata-for-TDX-VM.patch b/SOURCES/kvm-i386-tdx-Parse-TDVF-metadata-for-TDX-VM.patch new file mode 100644 index 0000000..0633ae2 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Parse-TDVF-metadata-for-TDX-VM.patch @@ -0,0 +1,116 @@ +From 2a27aea7e254e3828051e68a4c369068eacfb09e Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 047/115] i386/tdx: Parse TDVF metadata for TDX VM +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [47/115] bf2f78d0bbeb0b720e00b1593d02aeeaf46eb13e (bonzini/rhel-qemu-kvm) + +After TDVF is loaded to bios MemoryRegion, it needs parse TDVF metadata. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-19-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit cb5d65a854e58abeb705a2ce14cc3eb28973c606) +Signed-off-by: Paolo Bonzini +--- + hw/i386/pc_sysfw.c | 7 +++++++ + target/i386/kvm/tdx-stub.c | 5 +++++ + target/i386/kvm/tdx.c | 5 +++++ + target/i386/kvm/tdx.h | 3 +++ + 4 files changed, 20 insertions(+) + +diff --git a/hw/i386/pc_sysfw.c b/hw/i386/pc_sysfw.c +index e6271e1020..7436a4f33e 100644 +--- a/hw/i386/pc_sysfw.c ++++ b/hw/i386/pc_sysfw.c +@@ -37,6 +37,7 @@ + #include "hw/block/flash.h" + #include "sysemu/kvm.h" + #include "target/i386/sev.h" ++#include "kvm/tdx.h" + + #define FLASH_SECTOR_SIZE 4096 + +@@ -280,5 +281,11 @@ void x86_firmware_configure(hwaddr gpa, void *ptr, int size) + } + + sev_encrypt_flash(gpa, ptr, size, &error_fatal); ++ } else if (is_tdx_vm()) { ++ ret = tdx_parse_tdvf(ptr, size); ++ if (ret) { ++ error_report("failed to parse TDVF for TDX VM"); ++ exit(1); ++ } + } + } +diff --git a/target/i386/kvm/tdx-stub.c b/target/i386/kvm/tdx-stub.c +index 2344433594..7748b6d0a4 100644 +--- a/target/i386/kvm/tdx-stub.c ++++ b/target/i386/kvm/tdx-stub.c +@@ -8,3 +8,8 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) + { + return -EINVAL; + } ++ ++int tdx_parse_tdvf(void *flash_ptr, int size) ++{ ++ return -EINVAL; ++} +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 2522f2030d..71be3bd28d 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -382,6 +382,11 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) + return 0; + } + ++int tdx_parse_tdvf(void *flash_ptr, int size) ++{ ++ return tdvf_parse_metadata(&tdx_guest->tdvf, flash_ptr, size); ++} ++ + static bool tdx_guest_get_sept_ve_disable(Object *obj, Error **errp) + { + TdxGuest *tdx = TDX_GUEST(obj); +diff --git a/target/i386/kvm/tdx.h b/target/i386/kvm/tdx.h +index b73461b8d8..28a03c2a7b 100644 +--- a/target/i386/kvm/tdx.h ++++ b/target/i386/kvm/tdx.h +@@ -8,6 +8,7 @@ + #endif + + #include "confidential-guest.h" ++#include "hw/i386/tdvf.h" + + #define TYPE_TDX_GUEST "tdx-guest" + #define TDX_GUEST(obj) OBJECT_CHECK(TdxGuest, (obj), TYPE_TDX_GUEST) +@@ -32,6 +33,7 @@ typedef struct TdxGuest { + char *mrownerconfig; /* base64 encoded sha348 digest */ + + MemoryRegion *tdvf_mr; ++ TdxFirmware tdvf; + } TdxGuest; + + #ifdef CONFIG_TDX +@@ -42,5 +44,6 @@ bool is_tdx_vm(void); + + int tdx_pre_create_vcpu(CPUState *cpu, Error **errp); + void tdx_set_tdvf_region(MemoryRegion *tdvf_mr); ++int tdx_parse_tdvf(void *flash_ptr, int size); + + #endif /* QEMU_I386_TDX_H */ +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Remove-enumeration-of-GetQuote-in-tdx_handl.patch b/SOURCES/kvm-i386-tdx-Remove-enumeration-of-GetQuote-in-tdx_handl.patch new file mode 100644 index 0000000..6095a7b --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Remove-enumeration-of-GetQuote-in-tdx_handl.patch @@ -0,0 +1,70 @@ +From 7ead76d9444ad38538033a070836121cb4a7e20f Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:49 +0200 +Subject: [PATCH 099/115] i386/tdx: Remove enumeration of GetQuote in + tdx_handle_get_tdvmcall_info() + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [99/115] 0d402fadbfd0059369166a44d98f51fa46fd7ca3 (bonzini/rhel-qemu-kvm) + +GHCI is finalized with the being one of the base VMCALLs, and +not enuemrated via . + +Adjust tdx_handle_get_tdvmcall_info() to match with GHCI. + +Opportunistically fix the wrong indentation and explicitly set the +ret to TDG_VP_VMCALL_SUCCESS (in case KVM leaves unexpected value). + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20250703024021.3559286-2-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit b57999bb258349fabe497c8ff70277b5e5b281e2) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 6 ++++-- + target/i386/kvm/tdx.h | 2 -- + 2 files changed, 4 insertions(+), 4 deletions(-) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 201da78b06..2ca661cbc4 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -1259,13 +1259,15 @@ out_free: + void tdx_handle_get_tdvmcall_info(X86CPU *cpu, struct kvm_run *run) + { + if (run->tdx.get_tdvmcall_info.leaf != 1) { +- return; ++ return; + } + +- run->tdx.get_tdvmcall_info.r11 = TDG_VP_VMCALL_SUBFUNC_GET_QUOTE; ++ run->tdx.get_tdvmcall_info.r11 = 0; + run->tdx.get_tdvmcall_info.r12 = 0; + run->tdx.get_tdvmcall_info.r13 = 0; + run->tdx.get_tdvmcall_info.r14 = 0; ++ ++ run->tdx.get_tdvmcall_info.ret = TDG_VP_VMCALL_SUCCESS; + } + + static void tdx_panicked_on_fatal_error(X86CPU *cpu, uint64_t error_code, +diff --git a/target/i386/kvm/tdx.h b/target/i386/kvm/tdx.h +index 35a09c19c5..d439078a87 100644 +--- a/target/i386/kvm/tdx.h ++++ b/target/i386/kvm/tdx.h +@@ -32,8 +32,6 @@ typedef struct TdxGuestClass { + #define TDG_VP_VMCALL_GPA_INUSE 0x8000000000000001ULL + #define TDG_VP_VMCALL_ALIGN_ERROR 0x8000000000000002ULL + +-#define TDG_VP_VMCALL_SUBFUNC_GET_QUOTE 0x0000000000000001ULL +- + enum TdxRamType { + TDX_RAM_UNACCEPTED, + TDX_RAM_ADDED, +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Remove-task-watch-only-when-it-s-valid.patch b/SOURCES/kvm-i386-tdx-Remove-task-watch-only-when-it-s-valid.patch new file mode 100644 index 0000000..3c21f99 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Remove-task-watch-only-when-it-s-valid.patch @@ -0,0 +1,50 @@ +From a31b5ed8480b9f7e4d49a9e3ab1077fb1d4eb269 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:50 +0200 +Subject: [PATCH 103/115] i386/tdx: Remove task->watch only when it's valid + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [103/115] a9a4c03cf0f589eab2dc07572c2cb724cca5686c (bonzini/rhel-qemu-kvm) + +In some case (e.g., failed to connect to QGS socket), +tdx_generate_quote_cleanup() is called with task->watch invalid. It +triggers assertion of + + qemu-system-x86_64: GLib: g_source_remove: assertion 'tag > 0' failed + +Fix it by checking task->watch. + +Fixes: 40da501d8989 ("i386/tdx: handle TDG.VP.VMCALL") +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250625035505.2770580-1-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 50fd57418c3f08f13eb964dcb49f065246f2ecbf) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx-quote-generator.c | 4 +++- + 1 file changed, 3 insertions(+), 1 deletion(-) + +diff --git a/target/i386/kvm/tdx-quote-generator.c b/target/i386/kvm/tdx-quote-generator.c +index f59715f617..dee8334b27 100644 +--- a/target/i386/kvm/tdx-quote-generator.c ++++ b/target/i386/kvm/tdx-quote-generator.c +@@ -75,7 +75,9 @@ static void tdx_generate_quote_cleanup(TdxGenerateQuoteTask *task) + { + timer_del(&task->timer); + +- g_source_remove(task->watch); ++ if (task->watch) { ++ g_source_remove(task->watch); ++ } + qio_channel_close(QIO_CHANNEL(task->sioc), NULL); + object_unref(OBJECT(task->sioc)); + +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Remove-the-redundant-qemu_mutex_init-tdx-lo.patch b/SOURCES/kvm-i386-tdx-Remove-the-redundant-qemu_mutex_init-tdx-lo.patch new file mode 100644 index 0000000..fcea188 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Remove-the-redundant-qemu_mutex_init-tdx-lo.patch @@ -0,0 +1,50 @@ +From b01012b874646e7f5ce3d5da8f9c5539ca97ea16 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:50 +0200 +Subject: [PATCH 108/115] i386/tdx: Remove the redundant + qemu_mutex_init(&tdx->lock) +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [108/115] 51fa7fa495c526d62f144a0eb2ff4c87df397c8c (bonzini/rhel-qemu-kvm) + +Commit 40da501d8989 ("i386/tdx: handle TDG.VP.VMCALL") added +redundant qemu_mutex_init(&tdx->lock) in tdx_guest_init by mistake. + +Fix it by removing the redundant one. + +Fixes: 40da501d8989 ("i386/tdx: handle TDG.VP.VMCALL") +Reported-by: Peter Maydell +Signed-off-by: Xiaoyao Li +Reviewed-by: Daniel P. Berrangé +Link: https://lore.kernel.org/r/20250717103707.688929-1-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit f64832033d1262983bfe759669b4f65080f760dc) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 2 -- + 1 file changed, 2 deletions(-) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 08eed19960..2ff5211794 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -1527,8 +1527,6 @@ static void tdx_guest_init(Object *obj) + tdx_guest_set_qgs, + NULL, NULL); + +- qemu_mutex_init(&tdx->lock); +- + tdx->event_notify_vector = -1; + tdx->event_notify_apicid = -1; + } +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Set-APIC-bus-rate-to-match-with-what-TDX-mo.patch b/SOURCES/kvm-i386-tdx-Set-APIC-bus-rate-to-match-with-what-TDX-mo.patch new file mode 100644 index 0000000..e1ffadc --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Set-APIC-bus-rate-to-match-with-what-TDX-mo.patch @@ -0,0 +1,78 @@ +From c97fb38ea2145d0f827cc693e03628d9582433c3 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 043/115] i386/tdx: Set APIC bus rate to match with what TDX + module enforces +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [43/115] 2a25d2d762dba32fd6d0d3f4669c254cf0cb6a65 (bonzini/rhel-qemu-kvm) + +TDX advertises core crystal clock with cpuid[0x15] as 25MHz for TD +guests and it's unchangeable from VMM. As a result, TDX guest reads +the APIC timer at the same frequency, 25MHz. + +While KVM's default emulated frequency for APIC bus is 1GHz, set the +APIC bus rate to match with TDX explicitly to ensure KVM provide correct +emulated APIC timer for TD guest. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-15-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit d529a2ac5ef4620173439942f78ec668f9165fc1) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 13 +++++++++++++ + target/i386/kvm/tdx.h | 3 +++ + 2 files changed, 16 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 39fd964c6b..c96e8eb7b8 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -254,6 +254,19 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) + init_vm = g_malloc0(sizeof(struct kvm_tdx_init_vm) + + sizeof(struct kvm_cpuid_entry2) * KVM_MAX_CPUID_ENTRIES); + ++ if (!kvm_check_extension(kvm_state, KVM_CAP_X86_APIC_BUS_CYCLES_NS)) { ++ error_setg(errp, "KVM doesn't support KVM_CAP_X86_APIC_BUS_CYCLES_NS"); ++ return -EOPNOTSUPP; ++ } ++ ++ r = kvm_vm_enable_cap(kvm_state, KVM_CAP_X86_APIC_BUS_CYCLES_NS, ++ 0, TDX_APIC_BUS_CYCLES_NS); ++ if (r < 0) { ++ error_setg_errno(errp, -r, ++ "Unable to set core crystal clock frequency to 25MHz"); ++ return r; ++ } ++ + if (tdx_guest->mrconfigid) { + g_autofree uint8_t *data = qbase64_decode(tdx_guest->mrconfigid, + strlen(tdx_guest->mrconfigid), &data_len, errp); +diff --git a/target/i386/kvm/tdx.h b/target/i386/kvm/tdx.h +index e472b11fb0..d39e733d9f 100644 +--- a/target/i386/kvm/tdx.h ++++ b/target/i386/kvm/tdx.h +@@ -16,6 +16,9 @@ typedef struct TdxGuestClass { + X86ConfidentialGuestClass parent_class; + } TdxGuestClass; + ++/* TDX requires bus frequency 25MHz */ ++#define TDX_APIC_BUS_CYCLES_NS 40 ++ + typedef struct TdxGuest { + X86ConfidentialGuest parent_obj; + +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Set-and-check-kernel_irqchip-mode-for-TDX.patch b/SOURCES/kvm-i386-tdx-Set-and-check-kernel_irqchip-mode-for-TDX.patch new file mode 100644 index 0000000..7d9280d --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Set-and-check-kernel_irqchip-mode-for-TDX.patch @@ -0,0 +1,64 @@ +From 13e58b76efc0c20373a21d319d43361af80d82f6 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 067/115] i386/tdx: Set and check kernel_irqchip mode for TDX +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [67/115] 6d1c37af7cbaf0e84d1f1e2eb493f2397afaf452 (bonzini/rhel-qemu-kvm) + +KVM mandates kernel_irqchip to be split mode. + +Set it to split mode automatically when users don't provide an explicit +value, otherwise check it to be the split mode. + +Suggested-by: Daniel P. Berrangé +Signed-off-by: Xiaoyao Li +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-39-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit bb45580d842530d78b58179eaf80b6331b15324e) +Signed-off-by: Paolo Bonzini + +Conflicts: system/ -> sysemu/ +--- + target/i386/kvm/tdx.c | 8 ++++++++ + 1 file changed, 8 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 4f17e17308..58fb68cab0 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -16,6 +16,7 @@ + #include "qapi/error.h" + #include "qom/object_interfaces.h" + #include "crypto/hash.h" ++#include "sysemu/kvm_int.h" + #include "sysemu/runstate.h" + #include "sysemu/sysemu.h" + #include "exec/ramblock.h" +@@ -388,6 +389,13 @@ static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) + return -EINVAL; + } + ++ if (kvm_state->kernel_irqchip_split == ON_OFF_AUTO_AUTO) { ++ kvm_state->kernel_irqchip_split = ON_OFF_AUTO_ON; ++ } else if (kvm_state->kernel_irqchip_split != ON_OFF_AUTO_ON) { ++ error_setg(errp, "TDX VM requires kernel_irqchip to be split"); ++ return -EINVAL; ++ } ++ + if (!tdx_caps) { + r = get_tdx_capabilities(errp); + if (r) { +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Set-kvm_readonly_mem_enabled-to-false-for-T.patch b/SOURCES/kvm-i386-tdx-Set-kvm_readonly_mem_enabled-to-false-for-T.patch new file mode 100644 index 0000000..fee6a89 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Set-kvm_readonly_mem_enabled-to-false-for-T.patch @@ -0,0 +1,54 @@ +From 23e6031f307d262cae8a76306acb5341706098d6 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 064/115] i386/tdx: Set kvm_readonly_mem_enabled to false for + TDX VM + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [64/115] ee2c60314038f35359ea6bd00422b39c8573da88 (bonzini/rhel-qemu-kvm) + +TDX only supports readonly for shared memory but not for private memory. + +In the view of QEMU, it has no idea whether a memslot is used as shared +memory of private. Thus just mark kvm_readonly_mem_enabled to false to +TDX VM for simplicity. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-36-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit da6728658bf63d6a3989f1587a33566b3e54bed8) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 9 +++++++++ + 1 file changed, 9 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 410f8a9997..7bc36b620e 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -384,6 +384,15 @@ static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) + return -EOPNOTSUPP; + } + ++ /* ++ * Set kvm_readonly_mem_allowed to false, because TDX only supports readonly ++ * memory for shared memory but not for private memory. Besides, whether a ++ * memslot is private or shared is not determined by QEMU. ++ * ++ * Thus, just mark readonly memory not supported for simplicity. ++ */ ++ kvm_readonly_mem_allowed = false; ++ + qemu_add_machine_init_done_notifier(&tdx_machine_done_notify); + + tdx_guest = tdx; +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Set-value-of-GetTdVmCallInfo-based-on-capab.patch b/SOURCES/kvm-i386-tdx-Set-value-of-GetTdVmCallInfo-based-on-capab.patch new file mode 100644 index 0000000..f83432a --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Set-value-of-GetTdVmCallInfo-based-on-capab.patch @@ -0,0 +1,60 @@ +From 53a6bd5c6e3e4f0cf3fcf1b0a326c13e9defdc50 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:50 +0200 +Subject: [PATCH 100/115] i386/tdx: Set value of based on + capabilities of both KVM and QEMU + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [100/115] ebdd2061181d3ce95404931904e816171ae4f5ac (bonzini/rhel-qemu-kvm) + +KVM reports the supported TDVMCALL sub leafs in TDX capabilities. + +one for kernel-supported + TDVMCALLs (userspace can set those blindly) and one for user-supported + TDVMCALLs (userspace can set those if it knows how to handle them) + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20250703024021.3559286-4-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 55be385b10658a2372f944fa41aaba016e1e8433) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 11 +++++++++-- + 1 file changed, 9 insertions(+), 2 deletions(-) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 2ca661cbc4..a24e15571a 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -1256,14 +1256,21 @@ out_free: + g_free(task); + } + ++#define SUPPORTED_TDVMCALLINFO_1_R11 (0) ++#define SUPPORTED_TDVMCALLINFO_1_R12 (0) ++ + void tdx_handle_get_tdvmcall_info(X86CPU *cpu, struct kvm_run *run) + { + if (run->tdx.get_tdvmcall_info.leaf != 1) { + return; + } + +- run->tdx.get_tdvmcall_info.r11 = 0; +- run->tdx.get_tdvmcall_info.r12 = 0; ++ run->tdx.get_tdvmcall_info.r11 = (tdx_caps->user_tdvmcallinfo_1_r11 & ++ SUPPORTED_TDVMCALLINFO_1_R11) | ++ tdx_caps->kernel_tdvmcallinfo_1_r11; ++ run->tdx.get_tdvmcall_info.r12 = (tdx_caps->user_tdvmcallinfo_1_r12 & ++ SUPPORTED_TDVMCALLINFO_1_R12) | ++ tdx_caps->kernel_tdvmcallinfo_1_r12; + run->tdx.get_tdvmcall_info.r13 = 0; + run->tdx.get_tdvmcall_info.r14 = 0; + +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Setup-the-TD-HOB-list.patch b/SOURCES/kvm-i386-tdx-Setup-the-TD-HOB-list.patch new file mode 100644 index 0000000..dd97485 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Setup-the-TD-HOB-list.patch @@ -0,0 +1,264 @@ +From 2e545a2564bcece4b3f1e2baa18e93d81a94b20d Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 052/115] i386/tdx: Setup the TD HOB list + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [52/115] 789ad8477ccd49c1977e874d30963ebf9ac7f8b3 (bonzini/rhel-qemu-kvm) + +The TD HOB list is used to pass the information from VMM to TDVF. The TD +HOB must include PHIT HOB and Resource Descriptor HOB. More details can +be found in TDVF specification and PI specification. + +Build the TD HOB in TDX's machine_init_done callback. + +Co-developed-by: Isaku Yamahata +Signed-off-by: Isaku Yamahata +Co-developed-by: Sean Christopherson +Signed-off-by: Sean Christopherson +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-24-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit a731425980a4d3f8bb96fc41893b6437672875ee) +Signed-off-by: Paolo Bonzini +--- + hw/i386/meson.build | 2 +- + hw/i386/tdvf-hob.c | 130 ++++++++++++++++++++++++++++++++++++++++++ + hw/i386/tdvf-hob.h | 26 +++++++++ + target/i386/kvm/tdx.c | 16 ++++++ + 4 files changed, 173 insertions(+), 1 deletion(-) + create mode 100644 hw/i386/tdvf-hob.c + create mode 100644 hw/i386/tdvf-hob.h + +diff --git a/hw/i386/meson.build b/hw/i386/meson.build +index d6d8023664..cbac982039 100644 +--- a/hw/i386/meson.build ++++ b/hw/i386/meson.build +@@ -31,7 +31,7 @@ i386_ss.add(when: 'CONFIG_PC', if_true: files( + 'port92.c')) + i386_ss.add(when: 'CONFIG_X86_FW_OVMF', if_true: files('pc_sysfw_ovmf.c'), + if_false: files('pc_sysfw_ovmf-stubs.c')) +-i386_ss.add(when: 'CONFIG_TDX', if_true: files('tdvf.c')) ++i386_ss.add(when: 'CONFIG_TDX', if_true: files('tdvf.c', 'tdvf-hob.c')) + + subdir('kvm') + subdir('xen') +diff --git a/hw/i386/tdvf-hob.c b/hw/i386/tdvf-hob.c +new file mode 100644 +index 0000000000..782b3d1578 +--- /dev/null ++++ b/hw/i386/tdvf-hob.c +@@ -0,0 +1,130 @@ ++/* ++ * Copyright (c) 2025 Intel Corporation ++ * Author: Isaku Yamahata ++ * ++ * Xiaoyao Li ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#include "qemu/osdep.h" ++#include "qemu/error-report.h" ++#include "standard-headers/uefi/uefi.h" ++#include "hw/pci/pcie_host.h" ++#include "tdvf-hob.h" ++ ++typedef struct TdvfHob { ++ hwaddr hob_addr; ++ void *ptr; ++ int size; ++ ++ /* working area */ ++ void *current; ++ void *end; ++} TdvfHob; ++ ++static uint64_t tdvf_current_guest_addr(const TdvfHob *hob) ++{ ++ return hob->hob_addr + (hob->current - hob->ptr); ++} ++ ++static void tdvf_align(TdvfHob *hob, size_t align) ++{ ++ hob->current = QEMU_ALIGN_PTR_UP(hob->current, align); ++} ++ ++static void *tdvf_get_area(TdvfHob *hob, uint64_t size) ++{ ++ void *ret; ++ ++ if (hob->current + size > hob->end) { ++ error_report("TD_HOB overrun, size = 0x%" PRIx64, size); ++ exit(1); ++ } ++ ++ ret = hob->current; ++ hob->current += size; ++ tdvf_align(hob, 8); ++ return ret; ++} ++ ++static void tdvf_hob_add_memory_resources(TdxGuest *tdx, TdvfHob *hob) ++{ ++ EFI_HOB_RESOURCE_DESCRIPTOR *region; ++ EFI_RESOURCE_ATTRIBUTE_TYPE attr; ++ EFI_RESOURCE_TYPE resource_type; ++ ++ TdxRamEntry *e; ++ int i; ++ ++ for (i = 0; i < tdx->nr_ram_entries; i++) { ++ e = &tdx->ram_entries[i]; ++ ++ if (e->type == TDX_RAM_UNACCEPTED) { ++ resource_type = EFI_RESOURCE_MEMORY_UNACCEPTED; ++ attr = EFI_RESOURCE_ATTRIBUTE_TDVF_UNACCEPTED; ++ } else if (e->type == TDX_RAM_ADDED) { ++ resource_type = EFI_RESOURCE_SYSTEM_MEMORY; ++ attr = EFI_RESOURCE_ATTRIBUTE_TDVF_PRIVATE; ++ } else { ++ error_report("unknown TDX_RAM_ENTRY type %d", e->type); ++ exit(1); ++ } ++ ++ region = tdvf_get_area(hob, sizeof(*region)); ++ *region = (EFI_HOB_RESOURCE_DESCRIPTOR) { ++ .Header = { ++ .HobType = EFI_HOB_TYPE_RESOURCE_DESCRIPTOR, ++ .HobLength = cpu_to_le16(sizeof(*region)), ++ .Reserved = cpu_to_le32(0), ++ }, ++ .Owner = EFI_HOB_OWNER_ZERO, ++ .ResourceType = cpu_to_le32(resource_type), ++ .ResourceAttribute = cpu_to_le32(attr), ++ .PhysicalStart = cpu_to_le64(e->address), ++ .ResourceLength = cpu_to_le64(e->length), ++ }; ++ } ++} ++ ++void tdvf_hob_create(TdxGuest *tdx, TdxFirmwareEntry *td_hob) ++{ ++ TdvfHob hob = { ++ .hob_addr = td_hob->address, ++ .size = td_hob->size, ++ .ptr = td_hob->mem_ptr, ++ ++ .current = td_hob->mem_ptr, ++ .end = td_hob->mem_ptr + td_hob->size, ++ }; ++ ++ EFI_HOB_GENERIC_HEADER *last_hob; ++ EFI_HOB_HANDOFF_INFO_TABLE *hit; ++ ++ /* Note, Efi{Free}Memory{Bottom,Top} are ignored, leave 'em zeroed. */ ++ hit = tdvf_get_area(&hob, sizeof(*hit)); ++ *hit = (EFI_HOB_HANDOFF_INFO_TABLE) { ++ .Header = { ++ .HobType = EFI_HOB_TYPE_HANDOFF, ++ .HobLength = cpu_to_le16(sizeof(*hit)), ++ .Reserved = cpu_to_le32(0), ++ }, ++ .Version = cpu_to_le32(EFI_HOB_HANDOFF_TABLE_VERSION), ++ .BootMode = cpu_to_le32(0), ++ .EfiMemoryTop = cpu_to_le64(0), ++ .EfiMemoryBottom = cpu_to_le64(0), ++ .EfiFreeMemoryTop = cpu_to_le64(0), ++ .EfiFreeMemoryBottom = cpu_to_le64(0), ++ .EfiEndOfHobList = cpu_to_le64(0), /* initialized later */ ++ }; ++ ++ tdvf_hob_add_memory_resources(tdx, &hob); ++ ++ last_hob = tdvf_get_area(&hob, sizeof(*last_hob)); ++ *last_hob = (EFI_HOB_GENERIC_HEADER) { ++ .HobType = EFI_HOB_TYPE_END_OF_HOB_LIST, ++ .HobLength = cpu_to_le16(sizeof(*last_hob)), ++ .Reserved = cpu_to_le32(0), ++ }; ++ hit->EfiEndOfHobList = tdvf_current_guest_addr(&hob); ++} +diff --git a/hw/i386/tdvf-hob.h b/hw/i386/tdvf-hob.h +new file mode 100644 +index 0000000000..4fc6a3740a +--- /dev/null ++++ b/hw/i386/tdvf-hob.h +@@ -0,0 +1,26 @@ ++/* SPDX-License-Identifier: GPL-2.0-or-later */ ++ ++#ifndef HW_I386_TD_HOB_H ++#define HW_I386_TD_HOB_H ++ ++#include "hw/i386/tdvf.h" ++#include "target/i386/kvm/tdx.h" ++ ++void tdvf_hob_create(TdxGuest *tdx, TdxFirmwareEntry *td_hob); ++ ++#define EFI_RESOURCE_ATTRIBUTE_TDVF_PRIVATE \ ++ (EFI_RESOURCE_ATTRIBUTE_PRESENT | \ ++ EFI_RESOURCE_ATTRIBUTE_INITIALIZED | \ ++ EFI_RESOURCE_ATTRIBUTE_TESTED) ++ ++#define EFI_RESOURCE_ATTRIBUTE_TDVF_UNACCEPTED \ ++ (EFI_RESOURCE_ATTRIBUTE_PRESENT | \ ++ EFI_RESOURCE_ATTRIBUTE_INITIALIZED | \ ++ EFI_RESOURCE_ATTRIBUTE_TESTED) ++ ++#define EFI_RESOURCE_ATTRIBUTE_TDVF_MMIO \ ++ (EFI_RESOURCE_ATTRIBUTE_PRESENT | \ ++ EFI_RESOURCE_ATTRIBUTE_INITIALIZED | \ ++ EFI_RESOURCE_ATTRIBUTE_UNCACHEABLE) ++ ++#endif +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index d1c7821347..db5d58b600 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -21,6 +21,7 @@ + #include "hw/i386/e820_memory_layout.h" + #include "hw/i386/tdvf.h" + #include "hw/i386/x86.h" ++#include "hw/i386/tdvf-hob.h" + #include "kvm_i386.h" + #include "tdx.h" + +@@ -147,6 +148,19 @@ void tdx_set_tdvf_region(MemoryRegion *tdvf_mr) + tdx_guest->tdvf_mr = tdvf_mr; + } + ++static TdxFirmwareEntry *tdx_get_hob_entry(TdxGuest *tdx) ++{ ++ TdxFirmwareEntry *entry; ++ ++ for_each_tdx_fw_entry(&tdx->tdvf, entry) { ++ if (entry->type == TDVF_SECTION_TYPE_TD_HOB) { ++ return entry; ++ } ++ } ++ error_report("TDVF metadata doesn't specify TD_HOB location."); ++ exit(1); ++} ++ + static void tdx_add_ram_entry(uint64_t address, uint64_t length, + enum TdxRamType type) + { +@@ -281,6 +295,8 @@ static void tdx_finalize_vm(Notifier *notifier, void *unused) + + qsort(tdx_guest->ram_entries, tdx_guest->nr_ram_entries, + sizeof(TdxRamEntry), &tdx_ram_entry_compare); ++ ++ tdvf_hob_create(tdx_guest, tdx_get_hob_entry(tdx_guest)); + } + + static Notifier tdx_machine_done_notify = { +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Support-user-configurable-mrconfigid-mrowne.patch b/SOURCES/kvm-i386-tdx-Support-user-configurable-mrconfigid-mrowne.patch new file mode 100644 index 0000000..af194c4 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Support-user-configurable-mrconfigid-mrowne.patch @@ -0,0 +1,226 @@ +From 9c3c02aab32a45468605b6872c358968615681ef Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 042/115] i386/tdx: Support user configurable + mrconfigid/mrowner/mrownerconfig + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [42/115] 7f7acc5c5153f3486fcad084c036f67dbc483c66 (bonzini/rhel-qemu-kvm) + +Three sha384 hash values, mrconfigid, mrowner and mrownerconfig, of a TD +can be provided for TDX attestation. Detailed meaning of them can be +found: https://lore.kernel.org/qemu-devel/31d6dbc1-f453-4cef-ab08-4813f4e0ff92@intel.com/ + +Allow user to specify those values via property mrconfigid, mrowner and +mrownerconfig. They are all in base64 format. + +example +-object tdx-guest, \ + mrconfigid=ASNFZ4mrze8BI0VniavN7wEjRWeJq83vASNFZ4mrze8BI0VniavN7wEjRWeJq83v,\ + mrowner=ASNFZ4mrze8BI0VniavN7wEjRWeJq83vASNFZ4mrze8BI0VniavN7wEjRWeJq83v,\ + mrownerconfig=ASNFZ4mrze8BI0VniavN7wEjRWeJq83vASNFZ4mrze8BI0VniavN7wEjRWeJq83v + +Signed-off-by: Isaku Yamahata +Co-developed-by: Xiaoyao Li +Signed-off-by: Xiaoyao Li +Acked-by: Markus Armbruster +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-14-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit d05a0858cf876f79b57a622716fbad07f5b2ea08) +Signed-off-by: Paolo Bonzini +--- + qapi/qom.json | 16 +++++++- + target/i386/kvm/tdx.c | 95 +++++++++++++++++++++++++++++++++++++++++++ + target/i386/kvm/tdx.h | 3 ++ + 3 files changed, 113 insertions(+), 1 deletion(-) + +diff --git a/qapi/qom.json b/qapi/qom.json +index fefb54f90b..970ffeee9e 100644 +--- a/qapi/qom.json ++++ b/qapi/qom.json +@@ -1021,11 +1021,25 @@ + # pages. Some guest OS (e.g., Linux TD guest) may require this to + # be set, otherwise they refuse to boot. + # ++# @mrconfigid: ID for non-owner-defined configuration of the guest TD, ++# e.g., run-time or OS configuration (base64 encoded SHA384 digest). ++# Defaults to all zeros. ++# ++# @mrowner: ID for the guest TD’s owner (base64 encoded SHA384 digest). ++# Defaults to all zeros. ++# ++# @mrownerconfig: ID for owner-defined configuration of the guest TD, ++# e.g., specific to the workload rather than the run-time or OS ++# (base64 encoded SHA384 digest). Defaults to all zeros. ++# + # Since: 10.1 + ## + { 'struct': 'TdxGuestProperties', + 'data': { '*attributes': 'uint64', +- '*sept-ve-disable': 'bool' } } ++ '*sept-ve-disable': 'bool', ++ '*mrconfigid': 'str', ++ '*mrowner': 'str', ++ '*mrownerconfig': 'str' } } + + ## + # @ThreadContextProperties: +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 3de3b5fa6a..39fd964c6b 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -11,8 +11,10 @@ + + #include "qemu/osdep.h" + #include "qemu/error-report.h" ++#include "qemu/base64.h" + #include "qapi/error.h" + #include "qom/object_interfaces.h" ++#include "crypto/hash.h" + + #include "hw/i386/x86.h" + #include "kvm_i386.h" +@@ -240,6 +242,7 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) + CPUX86State *env = &x86cpu->env; + g_autofree struct kvm_tdx_init_vm *init_vm = NULL; + Error *local_err = NULL; ++ size_t data_len; + int retry = 10000; + int r = 0; + +@@ -251,6 +254,45 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) + init_vm = g_malloc0(sizeof(struct kvm_tdx_init_vm) + + sizeof(struct kvm_cpuid_entry2) * KVM_MAX_CPUID_ENTRIES); + ++ if (tdx_guest->mrconfigid) { ++ g_autofree uint8_t *data = qbase64_decode(tdx_guest->mrconfigid, ++ strlen(tdx_guest->mrconfigid), &data_len, errp); ++ if (!data) { ++ return -1; ++ } ++ if (data_len != QCRYPTO_HASH_DIGEST_LEN_SHA384) { ++ error_setg(errp, "TDX: failed to decode mrconfigid"); ++ return -1; ++ } ++ memcpy(init_vm->mrconfigid, data, data_len); ++ } ++ ++ if (tdx_guest->mrowner) { ++ g_autofree uint8_t *data = qbase64_decode(tdx_guest->mrowner, ++ strlen(tdx_guest->mrowner), &data_len, errp); ++ if (!data) { ++ return -1; ++ } ++ if (data_len != QCRYPTO_HASH_DIGEST_LEN_SHA384) { ++ error_setg(errp, "TDX: failed to decode mrowner"); ++ return -1; ++ } ++ memcpy(init_vm->mrowner, data, data_len); ++ } ++ ++ if (tdx_guest->mrownerconfig) { ++ g_autofree uint8_t *data = qbase64_decode(tdx_guest->mrownerconfig, ++ strlen(tdx_guest->mrownerconfig), &data_len, errp); ++ if (!data) { ++ return -1; ++ } ++ if (data_len != QCRYPTO_HASH_DIGEST_LEN_SHA384) { ++ error_setg(errp, "TDX: failed to decode mrownerconfig"); ++ return -1; ++ } ++ memcpy(init_vm->mrownerconfig, data, data_len); ++ } ++ + r = setup_td_guest_attributes(x86cpu, errp); + if (r) { + return r; +@@ -314,6 +356,51 @@ static void tdx_guest_set_sept_ve_disable(Object *obj, bool value, Error **errp) + } + } + ++static char *tdx_guest_get_mrconfigid(Object *obj, Error **errp) ++{ ++ TdxGuest *tdx = TDX_GUEST(obj); ++ ++ return g_strdup(tdx->mrconfigid); ++} ++ ++static void tdx_guest_set_mrconfigid(Object *obj, const char *value, Error **errp) ++{ ++ TdxGuest *tdx = TDX_GUEST(obj); ++ ++ g_free(tdx->mrconfigid); ++ tdx->mrconfigid = g_strdup(value); ++} ++ ++static char *tdx_guest_get_mrowner(Object *obj, Error **errp) ++{ ++ TdxGuest *tdx = TDX_GUEST(obj); ++ ++ return g_strdup(tdx->mrowner); ++} ++ ++static void tdx_guest_set_mrowner(Object *obj, const char *value, Error **errp) ++{ ++ TdxGuest *tdx = TDX_GUEST(obj); ++ ++ g_free(tdx->mrowner); ++ tdx->mrowner = g_strdup(value); ++} ++ ++static char *tdx_guest_get_mrownerconfig(Object *obj, Error **errp) ++{ ++ TdxGuest *tdx = TDX_GUEST(obj); ++ ++ return g_strdup(tdx->mrownerconfig); ++} ++ ++static void tdx_guest_set_mrownerconfig(Object *obj, const char *value, Error **errp) ++{ ++ TdxGuest *tdx = TDX_GUEST(obj); ++ ++ g_free(tdx->mrownerconfig); ++ tdx->mrownerconfig = g_strdup(value); ++} ++ + /* tdx guest */ + OBJECT_DEFINE_TYPE_WITH_INTERFACES(TdxGuest, + tdx_guest, +@@ -337,6 +424,14 @@ static void tdx_guest_init(Object *obj) + object_property_add_bool(obj, "sept-ve-disable", + tdx_guest_get_sept_ve_disable, + tdx_guest_set_sept_ve_disable); ++ object_property_add_str(obj, "mrconfigid", ++ tdx_guest_get_mrconfigid, ++ tdx_guest_set_mrconfigid); ++ object_property_add_str(obj, "mrowner", ++ tdx_guest_get_mrowner, tdx_guest_set_mrowner); ++ object_property_add_str(obj, "mrownerconfig", ++ tdx_guest_get_mrownerconfig, ++ tdx_guest_set_mrownerconfig); + } + + static void tdx_guest_finalize(Object *obj) +diff --git a/target/i386/kvm/tdx.h b/target/i386/kvm/tdx.h +index 4e2b5c61ff..e472b11fb0 100644 +--- a/target/i386/kvm/tdx.h ++++ b/target/i386/kvm/tdx.h +@@ -24,6 +24,9 @@ typedef struct TdxGuest { + bool initialized; + uint64_t attributes; /* TD attributes */ + uint64_t xfam; ++ char *mrconfigid; /* base64 encoded sha348 digest */ ++ char *mrowner; /* base64 encoded sha348 digest */ ++ char *mrownerconfig; /* base64 encoded sha348 digest */ + } TdxGuest; + + #ifdef CONFIG_TDX +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Track-RAM-entries-for-TDX-VM.patch b/SOURCES/kvm-i386-tdx-Track-RAM-entries-for-TDX-VM.patch new file mode 100644 index 0000000..eebb3ba --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Track-RAM-entries-for-TDX-VM.patch @@ -0,0 +1,223 @@ +From c4a9204c7d8521c6fdc6b02b48db74ac109bcf09 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 050/115] i386/tdx: Track RAM entries for TDX VM + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [50/115] fc14b8ad21b0be7179f9af62b6b32068122cae61 (bonzini/rhel-qemu-kvm) + +The RAM of TDX VM can be classified into two types: + + - TDX_RAM_UNACCEPTED: default type of TDX memory, which needs to be + accepted by TDX guest before it can be used and will be all-zeros + after being accepted. + + - TDX_RAM_ADDED: the RAM that is ADD'ed to TD guest before running, and + can be used directly. E.g., TD HOB and TEMP MEM that needed by TDVF. + +Maintain TdxRamEntries[] which grabs the initial RAM info from e820 table +and mark each RAM range as default type TDX_RAM_UNACCEPTED. + +Then turn the range of TD HOB and TEMP MEM to TDX_RAM_ADDED since these +ranges will be ADD'ed before TD runs and no need to be accepted runtime. + +The TdxRamEntries[] are later used to setup the memory TD resource HOB +that passes memory info from QEMU to TDVF. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-22-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit f18672e4cf91feed4b91ef85a264a500935a2865) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 109 ++++++++++++++++++++++++++++++++++++++++++ + target/i386/kvm/tdx.h | 14 ++++++ + 2 files changed, 123 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 95687e3a91..d1c7821347 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -18,6 +18,7 @@ + #include "crypto/hash.h" + #include "sysemu/sysemu.h" + ++#include "hw/i386/e820_memory_layout.h" + #include "hw/i386/tdvf.h" + #include "hw/i386/x86.h" + #include "kvm_i386.h" +@@ -146,11 +147,110 @@ void tdx_set_tdvf_region(MemoryRegion *tdvf_mr) + tdx_guest->tdvf_mr = tdvf_mr; + } + ++static void tdx_add_ram_entry(uint64_t address, uint64_t length, ++ enum TdxRamType type) ++{ ++ uint32_t nr_entries = tdx_guest->nr_ram_entries; ++ tdx_guest->ram_entries = g_renew(TdxRamEntry, tdx_guest->ram_entries, ++ nr_entries + 1); ++ ++ tdx_guest->ram_entries[nr_entries].address = address; ++ tdx_guest->ram_entries[nr_entries].length = length; ++ tdx_guest->ram_entries[nr_entries].type = type; ++ tdx_guest->nr_ram_entries++; ++} ++ ++static int tdx_accept_ram_range(uint64_t address, uint64_t length) ++{ ++ uint64_t head_start, tail_start, head_length, tail_length; ++ uint64_t tmp_address, tmp_length; ++ TdxRamEntry *e; ++ int i = 0; ++ ++ do { ++ if (i == tdx_guest->nr_ram_entries) { ++ return -1; ++ } ++ ++ e = &tdx_guest->ram_entries[i++]; ++ } while (address + length <= e->address || address >= e->address + e->length); ++ ++ /* ++ * The to-be-accepted ram range must be fully contained by one ++ * RAM entry. ++ */ ++ if (e->address > address || ++ e->address + e->length < address + length) { ++ return -1; ++ } ++ ++ if (e->type == TDX_RAM_ADDED) { ++ return 0; ++ } ++ ++ tmp_address = e->address; ++ tmp_length = e->length; ++ ++ e->address = address; ++ e->length = length; ++ e->type = TDX_RAM_ADDED; ++ ++ head_length = address - tmp_address; ++ if (head_length > 0) { ++ head_start = tmp_address; ++ tdx_add_ram_entry(head_start, head_length, TDX_RAM_UNACCEPTED); ++ } ++ ++ tail_start = address + length; ++ if (tail_start < tmp_address + tmp_length) { ++ tail_length = tmp_address + tmp_length - tail_start; ++ tdx_add_ram_entry(tail_start, tail_length, TDX_RAM_UNACCEPTED); ++ } ++ ++ return 0; ++} ++ ++static int tdx_ram_entry_compare(const void *lhs_, const void* rhs_) ++{ ++ const TdxRamEntry *lhs = lhs_; ++ const TdxRamEntry *rhs = rhs_; ++ ++ if (lhs->address == rhs->address) { ++ return 0; ++ } ++ if (le64_to_cpu(lhs->address) > le64_to_cpu(rhs->address)) { ++ return 1; ++ } ++ return -1; ++} ++ ++static void tdx_init_ram_entries(void) ++{ ++ unsigned i, j, nr_e820_entries; ++ ++ nr_e820_entries = e820_get_table(NULL); ++ tdx_guest->ram_entries = g_new(TdxRamEntry, nr_e820_entries); ++ ++ for (i = 0, j = 0; i < nr_e820_entries; i++) { ++ uint64_t addr, len; ++ ++ if (e820_get_entry(i, E820_RAM, &addr, &len)) { ++ tdx_guest->ram_entries[j].address = addr; ++ tdx_guest->ram_entries[j].length = len; ++ tdx_guest->ram_entries[j].type = TDX_RAM_UNACCEPTED; ++ j++; ++ } ++ } ++ tdx_guest->nr_ram_entries = j; ++} ++ + static void tdx_finalize_vm(Notifier *notifier, void *unused) + { + TdxFirmware *tdvf = &tdx_guest->tdvf; + TdxFirmwareEntry *entry; + ++ tdx_init_ram_entries(); ++ + for_each_tdx_fw_entry(tdvf, entry) { + switch (entry->type) { + case TDVF_SECTION_TYPE_BFV: +@@ -166,12 +266,21 @@ static void tdx_finalize_vm(Notifier *notifier, void *unused) + entry->type); + exit(1); + } ++ if (tdx_accept_ram_range(entry->address, entry->size)) { ++ error_report("Failed to accept memory for TDVF section %d", ++ entry->type); ++ qemu_ram_munmap(-1, entry->mem_ptr, entry->size); ++ exit(1); ++ } + break; + default: + error_report("Unsupported TDVF section %d", entry->type); + exit(1); + } + } ++ ++ qsort(tdx_guest->ram_entries, tdx_guest->nr_ram_entries, ++ sizeof(TdxRamEntry), &tdx_ram_entry_compare); + } + + static Notifier tdx_machine_done_notify = { +diff --git a/target/i386/kvm/tdx.h b/target/i386/kvm/tdx.h +index 28a03c2a7b..36a7400e74 100644 +--- a/target/i386/kvm/tdx.h ++++ b/target/i386/kvm/tdx.h +@@ -20,6 +20,17 @@ typedef struct TdxGuestClass { + /* TDX requires bus frequency 25MHz */ + #define TDX_APIC_BUS_CYCLES_NS 40 + ++enum TdxRamType { ++ TDX_RAM_UNACCEPTED, ++ TDX_RAM_ADDED, ++}; ++ ++typedef struct TdxRamEntry { ++ uint64_t address; ++ uint64_t length; ++ enum TdxRamType type; ++} TdxRamEntry; ++ + typedef struct TdxGuest { + X86ConfidentialGuest parent_obj; + +@@ -34,6 +45,9 @@ typedef struct TdxGuest { + + MemoryRegion *tdvf_mr; + TdxFirmware tdvf; ++ ++ uint32_t nr_ram_entries; ++ TdxRamEntry *ram_entries; + } TdxGuest; + + #ifdef CONFIG_TDX +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Track-mem_ptr-for-each-firmware-entry-of-TD.patch b/SOURCES/kvm-i386-tdx-Track-mem_ptr-for-each-firmware-entry-of-TD.patch new file mode 100644 index 0000000..bb416b2 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Track-mem_ptr-for-each-firmware-entry-of-TD.patch @@ -0,0 +1,151 @@ +From 95f727555b05af436c6e23a0dde42155a217a6e8 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 049/115] i386/tdx: Track mem_ptr for each firmware entry of + TDVF + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [49/115] 8cee9537e1387c718ec037bbc1a8d0a6b17bbf49 (bonzini/rhel-qemu-kvm) + +For each TDVF sections, QEMU needs to copy the content to guest +private memory via KVM API (KVM_TDX_INIT_MEM_REGION). + +Introduce a field @mem_ptr for TdxFirmwareEntry to track the memory +pointer of each TDVF sections. So that QEMU can add/copy them to guest +private memory later. + +TDVF sections can be classified into two groups: + - Firmware itself, e.g., TDVF BFV and CFV, that located separately from + guest RAM. Its memory pointer is the bios pointer. + + - Sections located at guest RAM, e.g., TEMP_MEM and TD_HOB. + mmap a new memory range for them. + +Register a machine_init_done callback to do the stuff. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-21-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 4420ba0ebbf014acc68f78669e0767e288313ed6) +Signed-off-by: Paolo Bonzini + +Conflicts: system/ -> sysemu/ +--- + hw/i386/tdvf.c | 1 + + include/hw/i386/tdvf.h | 7 +++++++ + target/i386/kvm/tdx.c | 37 +++++++++++++++++++++++++++++++++++++ + 3 files changed, 45 insertions(+) + +diff --git a/hw/i386/tdvf.c b/hw/i386/tdvf.c +index 824a387d42..88453cf3a5 100644 +--- a/hw/i386/tdvf.c ++++ b/hw/i386/tdvf.c +@@ -179,6 +179,7 @@ int tdvf_parse_metadata(TdxFirmware *fw, void *flash_ptr, int size) + } + } + ++ fw->mem_ptr = flash_ptr; + return 0; + + err: +diff --git a/include/hw/i386/tdvf.h b/include/hw/i386/tdvf.h +index 7ebcac42a3..e75c8d1acc 100644 +--- a/include/hw/i386/tdvf.h ++++ b/include/hw/i386/tdvf.h +@@ -26,13 +26,20 @@ typedef struct TdxFirmwareEntry { + uint64_t size; + uint32_t type; + uint32_t attributes; ++ ++ void *mem_ptr; + } TdxFirmwareEntry; + + typedef struct TdxFirmware { ++ void *mem_ptr; ++ + uint32_t nr_entries; + TdxFirmwareEntry *entries; + } TdxFirmware; + ++#define for_each_tdx_fw_entry(fw, e) \ ++ for (e = (fw)->entries; e != (fw)->entries + (fw)->nr_entries; e++) ++ + int tdvf_parse_metadata(TdxFirmware *fw, void *flash_ptr, int size); + + #endif /* HW_I386_TDVF_H */ +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 71be3bd28d..95687e3a91 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -12,10 +12,13 @@ + #include "qemu/osdep.h" + #include "qemu/error-report.h" + #include "qemu/base64.h" ++#include "qemu/mmap-alloc.h" + #include "qapi/error.h" + #include "qom/object_interfaces.h" + #include "crypto/hash.h" ++#include "sysemu/sysemu.h" + ++#include "hw/i386/tdvf.h" + #include "hw/i386/x86.h" + #include "kvm_i386.h" + #include "tdx.h" +@@ -143,6 +146,38 @@ void tdx_set_tdvf_region(MemoryRegion *tdvf_mr) + tdx_guest->tdvf_mr = tdvf_mr; + } + ++static void tdx_finalize_vm(Notifier *notifier, void *unused) ++{ ++ TdxFirmware *tdvf = &tdx_guest->tdvf; ++ TdxFirmwareEntry *entry; ++ ++ for_each_tdx_fw_entry(tdvf, entry) { ++ switch (entry->type) { ++ case TDVF_SECTION_TYPE_BFV: ++ case TDVF_SECTION_TYPE_CFV: ++ entry->mem_ptr = tdvf->mem_ptr + entry->data_offset; ++ break; ++ case TDVF_SECTION_TYPE_TD_HOB: ++ case TDVF_SECTION_TYPE_TEMP_MEM: ++ entry->mem_ptr = qemu_ram_mmap(-1, entry->size, ++ qemu_real_host_page_size(), 0, 0); ++ if (entry->mem_ptr == MAP_FAILED) { ++ error_report("Failed to mmap memory for TDVF section %d", ++ entry->type); ++ exit(1); ++ } ++ break; ++ default: ++ error_report("Unsupported TDVF section %d", entry->type); ++ exit(1); ++ } ++ } ++} ++ ++static Notifier tdx_machine_done_notify = { ++ .notify = tdx_finalize_vm, ++}; ++ + static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) + { + TdxGuest *tdx = TDX_GUEST(cgs); +@@ -157,6 +192,8 @@ static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) + } + } + ++ qemu_add_machine_init_done_notifier(&tdx_machine_done_notify); ++ + tdx_guest = tdx; + return 0; + } +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Validate-TD-attributes.patch b/SOURCES/kvm-i386-tdx-Validate-TD-attributes.patch new file mode 100644 index 0000000..19c2394 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Validate-TD-attributes.patch @@ -0,0 +1,106 @@ +From e0384fc5822eb8fcea9a5e59b89b9430dedadba3 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 041/115] i386/tdx: Validate TD attributes +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [41/115] c95c6476092eb717dd042c64d656c2b1aa70a409 (bonzini/rhel-qemu-kvm) + +Validate TD attributes with tdx_caps that only supported bits are +allowed by KVM. + +Besides, sanity check the attribute bits that have not been supported by +QEMU yet. e.g., debug bit, it will be allowed in the future when debug +TD support lands in QEMU. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Zhao Liu +Reviewed-by: Daniel P. Berrangé +Link: https://lore.kernel.org/r/20250508150002.689633-13-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 53b6f406b4f1a215fb3ec60e56ddba2e019a45ef) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 33 +++++++++++++++++++++++++++++++-- + 1 file changed, 31 insertions(+), 2 deletions(-) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 1ab063f790..3de3b5fa6a 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -18,10 +18,15 @@ + #include "kvm_i386.h" + #include "tdx.h" + ++#define TDX_TD_ATTRIBUTES_DEBUG BIT_ULL(0) + #define TDX_TD_ATTRIBUTES_SEPT_VE_DISABLE BIT_ULL(28) + #define TDX_TD_ATTRIBUTES_PKS BIT_ULL(30) + #define TDX_TD_ATTRIBUTES_PERFMON BIT_ULL(63) + ++#define TDX_SUPPORTED_TD_ATTRS (TDX_TD_ATTRIBUTES_SEPT_VE_DISABLE |\ ++ TDX_TD_ATTRIBUTES_PKS | \ ++ TDX_TD_ATTRIBUTES_PERFMON) ++ + static TdxGuest *tdx_guest; + + static struct kvm_tdx_capabilities *tdx_caps; +@@ -153,13 +158,34 @@ static int tdx_kvm_type(X86ConfidentialGuest *cg) + return KVM_X86_TDX_VM; + } + +-static void setup_td_guest_attributes(X86CPU *x86cpu) ++static int tdx_validate_attributes(TdxGuest *tdx, Error **errp) ++{ ++ if ((tdx->attributes & ~tdx_caps->supported_attrs)) { ++ error_setg(errp, "Invalid attributes 0x%lx for TDX VM " ++ "(KVM supported: 0x%llx)", tdx->attributes, ++ tdx_caps->supported_attrs); ++ return -1; ++ } ++ ++ if (tdx->attributes & ~TDX_SUPPORTED_TD_ATTRS) { ++ error_setg(errp, "Some QEMU unsupported TD attribute bits being " ++ "requested: 0x%lx (QEMU supported: 0x%llx)", ++ tdx->attributes, TDX_SUPPORTED_TD_ATTRS); ++ return -1; ++ } ++ ++ return 0; ++} ++ ++static int setup_td_guest_attributes(X86CPU *x86cpu, Error **errp) + { + CPUX86State *env = &x86cpu->env; + + tdx_guest->attributes |= (env->features[FEAT_7_0_ECX] & CPUID_7_0_ECX_PKS) ? + TDX_TD_ATTRIBUTES_PKS : 0; + tdx_guest->attributes |= x86cpu->enable_pmu ? TDX_TD_ATTRIBUTES_PERFMON : 0; ++ ++ return tdx_validate_attributes(tdx_guest, errp); + } + + static int setup_td_xfam(X86CPU *x86cpu, Error **errp) +@@ -225,7 +251,10 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) + init_vm = g_malloc0(sizeof(struct kvm_tdx_init_vm) + + sizeof(struct kvm_cpuid_entry2) * KVM_MAX_CPUID_ENTRIES); + +- setup_td_guest_attributes(x86cpu); ++ r = setup_td_guest_attributes(x86cpu, errp); ++ if (r) { ++ return r; ++ } + + r = setup_td_xfam(x86cpu, errp); + if (r) { +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Validate-phys_bits-against-host-value.patch b/SOURCES/kvm-i386-tdx-Validate-phys_bits-against-host-value.patch new file mode 100644 index 0000000..b5c4419 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Validate-phys_bits-against-host-value.patch @@ -0,0 +1,88 @@ +From b7e8674a1d3d577a3e88e95d2dab6aac626eca41 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:48 +0200 +Subject: [PATCH 083/115] i386/tdx: Validate phys_bits against host value +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [83/115] 42a17d80e8c176a2dc4e2d3da0ea74f7fdb85577 (bonzini/rhel-qemu-kvm) + +For TDX guest, the phys_bits is not configurable and can only be +host/native value. + +Validate phys_bits inside tdx_check_features(). + +Signed-off-by: Xiaoyao Li +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-55-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 907ee7b67e50a7eea2768c66e3ad67c9aa4ffd3c) +Signed-off-by: Paolo Bonzini +--- + target/i386/host-cpu.c | 2 +- + target/i386/host-cpu.h | 1 + + target/i386/kvm/tdx.c | 8 ++++++++ + 3 files changed, 10 insertions(+), 1 deletion(-) + +diff --git a/target/i386/host-cpu.c b/target/i386/host-cpu.c +index 4a77ecc1fc..4ab536ab80 100644 +--- a/target/i386/host-cpu.c ++++ b/target/i386/host-cpu.c +@@ -15,7 +15,7 @@ + #include "sysemu/sysemu.h" + + /* Note: Only safe for use on x86(-64) hosts */ +-static uint32_t host_cpu_phys_bits(void) ++uint32_t host_cpu_phys_bits(void) + { + uint32_t eax; + uint32_t host_phys_bits; +diff --git a/target/i386/host-cpu.h b/target/i386/host-cpu.h +index 6a9bc918ba..b97ec01c9b 100644 +--- a/target/i386/host-cpu.h ++++ b/target/i386/host-cpu.h +@@ -10,6 +10,7 @@ + #ifndef HOST_CPU_H + #define HOST_CPU_H + ++uint32_t host_cpu_phys_bits(void); + void host_cpu_instance_init(X86CPU *cpu); + void host_cpu_max_instance_init(X86CPU *cpu); + bool host_cpu_realizefn(CPUState *cs, Error **errp); +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 3ec31d4872..b9c3ba3725 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -25,6 +25,7 @@ + + #include "cpu.h" + #include "cpu-internal.h" ++#include "host-cpu.h" + #include "hw/i386/e820_memory_layout.h" + #include "hw/i386/tdvf.h" + #include "hw/i386/x86.h" +@@ -879,6 +880,13 @@ static int tdx_check_features(X86ConfidentialGuest *cg, CPUState *cs) + return -EINVAL; + } + ++ if (cpu->phys_bits != host_cpu_phys_bits()) { ++ error_report("TDX requires guest CPU physical bits (%u) " ++ "to match host CPU physical bits (%u)", ++ cpu->phys_bits, host_cpu_phys_bits()); ++ return -EINVAL; ++ } ++ + return 0; + } + +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Wire-CPU-features-up-with-attributes-of-TD-.patch b/SOURCES/kvm-i386-tdx-Wire-CPU-features-up-with-attributes-of-TD-.patch new file mode 100644 index 0000000..cea4e11 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Wire-CPU-features-up-with-attributes-of-TD-.patch @@ -0,0 +1,78 @@ +From e417afeb76f30fca6c16e57971521c5c9c25681c Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 040/115] i386/tdx: Wire CPU features up with attributes of TD + guest +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [40/115] ce59bfc2281e630283095aec8975d93ef455d599 (bonzini/rhel-qemu-kvm) + +For QEMU VMs, + - PKS is configured via CPUID_7_0_ECX_PKS, e.g., -cpu xxx,+pks and + - PMU is configured by x86cpu->enable_pmu, e.g., -cpu xxx,pmu=on + +While the bit 30 (PKS) and bit 63 (PERFMON) of TD's attributes are also +used to configure the PKS and PERFMON/PMU of TD, reuse the existing +configuration interfaces of 'cpu' for TD's attributes. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-12-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit bb3be394cf80d68251e5b89e823dddc679b6e644) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 13 +++++++++++++ + 1 file changed, 13 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 2ed40b7614..1ab063f790 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -19,6 +19,8 @@ + #include "tdx.h" + + #define TDX_TD_ATTRIBUTES_SEPT_VE_DISABLE BIT_ULL(28) ++#define TDX_TD_ATTRIBUTES_PKS BIT_ULL(30) ++#define TDX_TD_ATTRIBUTES_PERFMON BIT_ULL(63) + + static TdxGuest *tdx_guest; + +@@ -151,6 +153,15 @@ static int tdx_kvm_type(X86ConfidentialGuest *cg) + return KVM_X86_TDX_VM; + } + ++static void setup_td_guest_attributes(X86CPU *x86cpu) ++{ ++ CPUX86State *env = &x86cpu->env; ++ ++ tdx_guest->attributes |= (env->features[FEAT_7_0_ECX] & CPUID_7_0_ECX_PKS) ? ++ TDX_TD_ATTRIBUTES_PKS : 0; ++ tdx_guest->attributes |= x86cpu->enable_pmu ? TDX_TD_ATTRIBUTES_PERFMON : 0; ++} ++ + static int setup_td_xfam(X86CPU *x86cpu, Error **errp) + { + CPUX86State *env = &x86cpu->env; +@@ -214,6 +225,8 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp) + init_vm = g_malloc0(sizeof(struct kvm_tdx_init_vm) + + sizeof(struct kvm_cpuid_entry2) * KVM_MAX_CPUID_ENTRIES); + ++ setup_td_guest_attributes(x86cpu); ++ + r = setup_td_xfam(x86cpu, errp); + if (r) { + return r; +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-Wire-TDX_REPORT_FATAL_ERROR-with-GuestPanic.patch b/SOURCES/kvm-i386-tdx-Wire-TDX_REPORT_FATAL_ERROR-with-GuestPanic.patch new file mode 100644 index 0000000..6114d52 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-Wire-TDX_REPORT_FATAL_ERROR-with-GuestPanic.patch @@ -0,0 +1,240 @@ +From 7f2a5ac9ad2e2d54b09e1bcd1a1d59cbde1443eb Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 058/115] i386/tdx: Wire TDX_REPORT_FATAL_ERROR with GuestPanic + facility + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [58/115] 4cdc786e5ad029e8ed5713b4a27bb038a4b6f935 (bonzini/rhel-qemu-kvm) + +Integrate TDX's TDX_REPORT_FATAL_ERROR into QEMU GuestPanic facility + +Originated-from: Isaku Yamahata +Signed-off-by: Xiaoyao Li +Acked-by: Markus Armbruster +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-30-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 6e250463b08b4028123f201343ee72099ef81e68) +Signed-off-by: Paolo Bonzini + +Conflicts: system/ -> sysemu/ +--- + qapi/run-state.json | 31 +++++++++++++++++++-- + system/runstate.c | 65 +++++++++++++++++++++++++++++++++++++++++++ + target/i386/kvm/tdx.c | 25 ++++++++++++++++- + 3 files changed, 118 insertions(+), 3 deletions(-) + +diff --git a/qapi/run-state.json b/qapi/run-state.json +index ce95cfa46b..ee11adc508 100644 +--- a/qapi/run-state.json ++++ b/qapi/run-state.json +@@ -501,10 +501,12 @@ + # + # @s390: s390 guest panic information type (Since: 2.12) + # ++# @tdx: tdx guest panic information type (Since: 10.1) ++# + # Since: 2.9 + ## + { 'enum': 'GuestPanicInformationType', +- 'data': [ 'hyper-v', 's390' ] } ++ 'data': [ 'hyper-v', 's390', 'tdx' ] } + + ## + # @GuestPanicInformation: +@@ -519,7 +521,8 @@ + 'base': {'type': 'GuestPanicInformationType'}, + 'discriminator': 'type', + 'data': {'hyper-v': 'GuestPanicInformationHyperV', +- 's390': 'GuestPanicInformationS390'}} ++ 's390': 'GuestPanicInformationS390', ++ 'tdx' : 'GuestPanicInformationTdx'}} + + ## + # @GuestPanicInformationHyperV: +@@ -598,6 +601,30 @@ + 'psw-addr': 'uint64', + 'reason': 'S390CrashReason'}} + ++## ++# @GuestPanicInformationTdx: ++# ++# TDX Guest panic information specific to TDX, as specified in the ++# "Guest-Hypervisor Communication Interface (GHCI) Specification", ++# section TDG.VP.VMCALL. ++# ++# @error-code: TD-specific error code ++# ++# @message: Human-readable error message provided by the guest. Not ++# to be trusted. ++# ++# @gpa: guest-physical address of a page that contains more verbose ++# error information, as zero-terminated string. Present when the ++# "GPA valid" bit (bit 63) is set in @error-code. ++# ++# ++# Since: 10.1 ++## ++{'struct': 'GuestPanicInformationTdx', ++ 'data': {'error-code': 'uint32', ++ 'message': 'str', ++ '*gpa': 'uint64'}} ++ + ## + # @MEMORY_FAILURE: + # +diff --git a/system/runstate.c b/system/runstate.c +index c2c9afa905..31970c522e 100644 +--- a/system/runstate.c ++++ b/system/runstate.c +@@ -565,6 +565,58 @@ static void qemu_system_wakeup(void) + } + } + ++static char *tdx_parse_panic_message(char *message) ++{ ++ bool printable = false; ++ char *buf = NULL; ++ int len = 0, i; ++ ++ /* ++ * Although message is defined as a json string, we shouldn't ++ * unconditionally treat it as is because the guest generated it and ++ * it's not necessarily trustable. ++ */ ++ if (message) { ++ /* The caller guarantees the NULL-terminated string. */ ++ len = strlen(message); ++ ++ printable = len > 0; ++ for (i = 0; i < len; i++) { ++ if (!(0x20 <= message[i] && message[i] <= 0x7e)) { ++ printable = false; ++ break; ++ } ++ } ++ } ++ ++ if (len == 0) { ++ buf = g_malloc(1); ++ buf[0] = '\0'; ++ } else { ++ if (!printable) { ++ /* 3 = length of "%02x " */ ++ buf = g_malloc(len * 3); ++ for (i = 0; i < len; i++) { ++ if (message[i] == '\0') { ++ break; ++ } else { ++ sprintf(buf + 3 * i, "%02x ", message[i]); ++ } ++ } ++ if (i > 0) { ++ /* replace the last ' '(space) to NULL */ ++ buf[i * 3 - 1] = '\0'; ++ } else { ++ buf[0] = '\0'; ++ } ++ } else { ++ buf = g_strdup(message); ++ } ++ } ++ ++ return buf; ++} ++ + void qemu_system_guest_panicked(GuestPanicInformation *info) + { + qemu_log_mask(LOG_GUEST_ERROR, "Guest crashed"); +@@ -606,7 +658,20 @@ void qemu_system_guest_panicked(GuestPanicInformation *info) + S390CrashReason_str(info->u.s390.reason), + info->u.s390.psw_mask, + info->u.s390.psw_addr); ++ } else if (info->type == GUEST_PANIC_INFORMATION_TYPE_TDX) { ++ char *message = tdx_parse_panic_message(info->u.tdx.message); ++ qemu_log_mask(LOG_GUEST_ERROR, ++ "\nTDX guest reports fatal error." ++ " error code: 0x%" PRIx32 " error message:\"%s\"\n", ++ info->u.tdx.error_code, message); ++ g_free(message); ++ if (info->u.tdx.gpa != -1ull) { ++ qemu_log_mask(LOG_GUEST_ERROR, "Additional error information " ++ "can be found at gpa page: 0x%" PRIx64 "\n", ++ info->u.tdx.gpa); ++ } + } ++ + qapi_free_GuestPanicInformation(info); + } + } +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 679613ab55..7611f51aae 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -16,6 +16,7 @@ + #include "qapi/error.h" + #include "qom/object_interfaces.h" + #include "crypto/hash.h" ++#include "sysemu/runstate.h" + #include "sysemu/sysemu.h" + #include "exec/ramblock.h" + +@@ -615,18 +616,35 @@ int tdx_parse_tdvf(void *flash_ptr, int size) + return tdvf_parse_metadata(&tdx_guest->tdvf, flash_ptr, size); + } + ++static void tdx_panicked_on_fatal_error(X86CPU *cpu, uint64_t error_code, ++ char *message, uint64_t gpa) ++{ ++ GuestPanicInformation *panic_info; ++ ++ panic_info = g_new0(GuestPanicInformation, 1); ++ panic_info->type = GUEST_PANIC_INFORMATION_TYPE_TDX; ++ panic_info->u.tdx.error_code = (uint32_t) error_code; ++ panic_info->u.tdx.message = message; ++ panic_info->u.tdx.gpa = gpa; ++ ++ qemu_system_guest_panicked(panic_info); ++} ++ + /* + * Only 8 registers can contain valid ASCII byte stream to form the fatal + * message, and their sequence is: R14, R15, RBX, RDI, RSI, R8, R9, RDX + */ + #define TDX_FATAL_MESSAGE_MAX 64 + ++#define TDX_REPORT_FATAL_ERROR_GPA_VALID BIT_ULL(63) ++ + int tdx_handle_report_fatal_error(X86CPU *cpu, struct kvm_run *run) + { + uint64_t error_code = run->system_event.data[R_R12]; + uint64_t reg_mask = run->system_event.data[R_ECX]; + char *message = NULL; + uint64_t *tmp; ++ uint64_t gpa = -1ull; + + if (error_code & 0xffff) { + error_report("TDX: REPORT_FATAL_ERROR: invalid error code: 0x%lx", +@@ -657,7 +675,12 @@ int tdx_handle_report_fatal_error(X86CPU *cpu, struct kvm_run *run) + } + #undef COPY_REG + +- error_report("TD guest reports fatal error. %s", message ? : ""); ++ if (error_code & TDX_REPORT_FATAL_ERROR_GPA_VALID) { ++ gpa = run->system_event.data[R_R13]; ++ } ++ ++ tdx_panicked_on_fatal_error(cpu, error_code, message, gpa); ++ + return -1; + } + +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-handle-TDG.VP.VMCALL-GetQuote.patch b/SOURCES/kvm-i386-tdx-handle-TDG.VP.VMCALL-GetQuote.patch new file mode 100644 index 0000000..0ecf41d --- /dev/null +++ b/SOURCES/kvm-i386-tdx-handle-TDG.VP.VMCALL-GetQuote.patch @@ -0,0 +1,798 @@ +From e100b3595bacf3d18af8a6b8234e8c1abdd1f87c Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:26:18 +0200 +Subject: [PATCH 093/115] i386/tdx: handle TDG.VP.VMCALL + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [93/115] 4e656b0d7f8a022d25573b9583ed7864815cb88b (bonzini/rhel-qemu-kvm) + +Add property "quote-generation-socket" to tdx-guest, which is a property +of type SocketAddress to specify Quote Generation Service(QGS). + +On request of GetQuote, it connects to the QGS socket, read request +data from shared guest memory, send the request data to the QGS, +and store the response into shared guest memory, at last notify +TD guest by interrupt. + +command line example: + qemu-system-x86_64 \ + -object '{"qom-type":"tdx-guest","id":"tdx0","quote-generation-socket":{"type":"unix", "path":"/var/run/tdx-qgs/qgs.socket"}}' \ + -machine confidential-guest-support=tdx0 + +Note, above example uses the unix socket. It can be other types, like vsock, +which depends on the implementation of QGS. + +To avoid no response from QGS server, setup a timer for the transaction. +If timeout, make it an error and interrupt guest. Define the threshold of +time to 30s at present, maybe change to other value if not appropriate. + +Signed-off-by: Isaku Yamahata +Co-developed-by: Chenyi Qiang +Signed-off-by: Chenyi Qiang +Co-developed-by: Xiaoyao Li +Signed-off-by: Xiaoyao Li +Tested-by: Xiaoyao Li +Signed-off-by: Paolo Bonzini +(cherry picked from commit 40da501d8989913935660dc24953ece02c9e98b8) +Signed-off-by: Paolo Bonzini + +Conflicts: system/ -> sysemu/,exec/ +--- + qapi/qom.json | 8 +- + target/i386/kvm/kvm.c | 3 + + target/i386/kvm/meson.build | 2 +- + target/i386/kvm/tdx-quote-generator.c | 300 ++++++++++++++++++++++++++ + target/i386/kvm/tdx-quote-generator.h | 82 +++++++ + target/i386/kvm/tdx-stub.c | 4 + + target/i386/kvm/tdx.c | 176 ++++++++++++++- + target/i386/kvm/tdx.h | 10 + + 8 files changed, 582 insertions(+), 3 deletions(-) + create mode 100644 target/i386/kvm/tdx-quote-generator.c + create mode 100644 target/i386/kvm/tdx-quote-generator.h + +diff --git a/qapi/qom.json b/qapi/qom.json +index 970ffeee9e..72c1605ad7 100644 +--- a/qapi/qom.json ++++ b/qapi/qom.json +@@ -1032,6 +1032,11 @@ + # e.g., specific to the workload rather than the run-time or OS + # (base64 encoded SHA384 digest). Defaults to all zeros. + # ++# @quote-generation-socket: socket address for Quote Generation ++# Service (QGS). QGS is a daemon running on the host. Without ++# it, the guest will not be able to get a TD quote for ++# attestation. ++# + # Since: 10.1 + ## + { 'struct': 'TdxGuestProperties', +@@ -1039,7 +1044,8 @@ + '*sept-ve-disable': 'bool', + '*mrconfigid': 'str', + '*mrowner': 'str', +- '*mrownerconfig': 'str' } } ++ '*mrownerconfig': 'str', ++ '*quote-generation-socket': 'SocketAddress' } } + + ## + # @ThreadContextProperties: +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index 26328a1d3b..fbf11b2122 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -6045,6 +6045,9 @@ int kvm_arch_handle_exit(CPUState *cs, struct kvm_run *run) + * does not handle the TDVMCALL. + */ + switch (run->tdx.nr) { ++ case TDVMCALL_GET_QUOTE: ++ tdx_handle_get_quote(cpu, run); ++ break; + case TDVMCALL_GET_TD_VM_CALL_INFO: + tdx_handle_get_tdvmcall_info(cpu, run); + break; +diff --git a/target/i386/kvm/meson.build b/target/i386/kvm/meson.build +index 3f44cdedb7..2675bf8902 100644 +--- a/target/i386/kvm/meson.build ++++ b/target/i386/kvm/meson.build +@@ -8,7 +8,7 @@ i386_kvm_ss.add(files( + + i386_kvm_ss.add(when: 'CONFIG_XEN_EMU', if_true: files('xen-emu.c')) + +-i386_kvm_ss.add(when: 'CONFIG_TDX', if_true: files('tdx.c'), if_false: files('tdx-stub.c')) ++i386_kvm_ss.add(when: 'CONFIG_TDX', if_true: files('tdx.c', 'tdx-quote-generator.c'), if_false: files('tdx-stub.c')) + + i386_system_ss.add(when: 'CONFIG_HYPERV', if_true: files('hyperv.c'), if_false: files('hyperv-stub.c')) + +diff --git a/target/i386/kvm/tdx-quote-generator.c b/target/i386/kvm/tdx-quote-generator.c +new file mode 100644 +index 0000000000..f59715f617 +--- /dev/null ++++ b/target/i386/kvm/tdx-quote-generator.c +@@ -0,0 +1,300 @@ ++/* ++ * QEMU TDX Quote Generation Support ++ * ++ * Copyright (c) 2025 Intel Corporation ++ * ++ * Author: ++ * Xiaoyao Li ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#include "qemu/osdep.h" ++#include "qemu/error-report.h" ++#include "qapi/error.h" ++#include "qapi/qapi-visit-sockets.h" ++ ++#include "tdx-quote-generator.h" ++ ++#define QGS_MSG_LIB_MAJOR_VER 1 ++#define QGS_MSG_LIB_MINOR_VER 1 ++ ++typedef enum _qgs_msg_type_t { ++ GET_QUOTE_REQ = 0, ++ GET_QUOTE_RESP = 1, ++ GET_COLLATERAL_REQ = 2, ++ GET_COLLATERAL_RESP = 3, ++ GET_PLATFORM_INFO_REQ = 4, ++ GET_PLATFORM_INFO_RESP = 5, ++ QGS_MSG_TYPE_MAX ++} qgs_msg_type_t; ++ ++typedef struct _qgs_msg_header_t { ++ uint16_t major_version; ++ uint16_t minor_version; ++ uint32_t type; ++ uint32_t size; // size of the whole message, include this header, in byte ++ uint32_t error_code; // used in response only ++} qgs_msg_header_t; ++ ++typedef struct _qgs_msg_get_quote_req_t { ++ qgs_msg_header_t header; // header.type = GET_QUOTE_REQ ++ uint32_t report_size; // cannot be 0 ++ uint32_t id_list_size; // length of id_list, in byte, can be 0 ++} qgs_msg_get_quote_req_t; ++ ++typedef struct _qgs_msg_get_quote_resp_s { ++ qgs_msg_header_t header; // header.type = GET_QUOTE_RESP ++ uint32_t selected_id_size; // can be 0 in case only one id is sent in request ++ uint32_t quote_size; // length of quote_data, in byte ++ uint8_t id_quote[]; // selected id followed by quote ++} qgs_msg_get_quote_resp_t; ++ ++#define HEADER_SIZE 4 ++ ++static uint32_t decode_header(const char *buf, size_t len) { ++ if (len < HEADER_SIZE) { ++ return 0; ++ } ++ uint32_t msg_size = 0; ++ for (uint32_t i = 0; i < HEADER_SIZE; ++i) { ++ msg_size = msg_size * 256 + (buf[i] & 0xFF); ++ } ++ return msg_size; ++} ++ ++static void encode_header(char *buf, size_t len, uint32_t size) { ++ assert(len >= HEADER_SIZE); ++ buf[0] = ((size >> 24) & 0xFF); ++ buf[1] = ((size >> 16) & 0xFF); ++ buf[2] = ((size >> 8) & 0xFF); ++ buf[3] = (size & 0xFF); ++} ++ ++static void tdx_generate_quote_cleanup(TdxGenerateQuoteTask *task) ++{ ++ timer_del(&task->timer); ++ ++ g_source_remove(task->watch); ++ qio_channel_close(QIO_CHANNEL(task->sioc), NULL); ++ object_unref(OBJECT(task->sioc)); ++ ++ task->completion(task); ++} ++ ++static gboolean tdx_get_quote_read(QIOChannel *ioc, GIOCondition condition, ++ gpointer opaque) ++{ ++ TdxGenerateQuoteTask *task = opaque; ++ Error *err = NULL; ++ int ret; ++ ++ ret = qio_channel_read(ioc, task->receive_buf + task->receive_buf_received, ++ task->payload_len - task->receive_buf_received, &err); ++ if (ret < 0) { ++ if (ret == QIO_CHANNEL_ERR_BLOCK) { ++ return G_SOURCE_CONTINUE; ++ } else { ++ error_report_err(err); ++ task->status_code = TDX_VP_GET_QUOTE_ERROR; ++ goto end; ++ } ++ } ++ ++ if (ret == 0) { ++ error_report("End of file before reply received"); ++ task->status_code = TDX_VP_GET_QUOTE_ERROR; ++ goto end; ++ } ++ ++ task->receive_buf_received += ret; ++ if (task->receive_buf_received >= HEADER_SIZE) { ++ uint32_t len = decode_header(task->receive_buf, ++ task->receive_buf_received); ++ if (len == 0 || ++ len > (task->payload_len - HEADER_SIZE)) { ++ error_report("Message len %u must be non-zero & less than %zu", ++ len, (task->payload_len - HEADER_SIZE)); ++ task->status_code = TDX_VP_GET_QUOTE_ERROR; ++ goto end; ++ } ++ ++ /* Now we know the size, shrink to fit */ ++ task->payload_len = HEADER_SIZE + len; ++ task->receive_buf = g_renew(char, ++ task->receive_buf, ++ task->payload_len); ++ } ++ ++ if (task->receive_buf_received >= (sizeof(qgs_msg_header_t) + HEADER_SIZE)) { ++ qgs_msg_header_t *hdr = (qgs_msg_header_t *)(task->receive_buf + HEADER_SIZE); ++ if (hdr->major_version != QGS_MSG_LIB_MAJOR_VER || ++ hdr->minor_version != QGS_MSG_LIB_MINOR_VER) { ++ error_report("Invalid QGS message header version %d.%d", ++ hdr->major_version, ++ hdr->minor_version); ++ task->status_code = TDX_VP_GET_QUOTE_ERROR; ++ goto end; ++ } ++ if (hdr->type != GET_QUOTE_RESP) { ++ error_report("Invalid QGS message type %d", ++ hdr->type); ++ task->status_code = TDX_VP_GET_QUOTE_ERROR; ++ goto end; ++ } ++ if (hdr->size > (task->payload_len - HEADER_SIZE)) { ++ error_report("QGS message size %d exceeds payload capacity %zu", ++ hdr->size, task->payload_len); ++ task->status_code = TDX_VP_GET_QUOTE_ERROR; ++ goto end; ++ } ++ if (hdr->error_code != 0) { ++ error_report("QGS message error code %d", ++ hdr->error_code); ++ task->status_code = TDX_VP_GET_QUOTE_ERROR; ++ goto end; ++ } ++ } ++ if (task->receive_buf_received >= (sizeof(qgs_msg_get_quote_resp_t) + HEADER_SIZE)) { ++ qgs_msg_get_quote_resp_t *msg = (qgs_msg_get_quote_resp_t *)(task->receive_buf + HEADER_SIZE); ++ if (msg->selected_id_size != 0) { ++ error_report("QGS message selected ID was %d not 0", ++ msg->selected_id_size); ++ task->status_code = TDX_VP_GET_QUOTE_ERROR; ++ goto end; ++ } ++ ++ if ((task->payload_len - HEADER_SIZE - sizeof(qgs_msg_get_quote_resp_t)) != ++ msg->quote_size) { ++ error_report("QGS quote size %d should be %zu", ++ msg->quote_size, ++ (task->payload_len - sizeof(qgs_msg_get_quote_resp_t))); ++ task->status_code = TDX_VP_GET_QUOTE_ERROR; ++ goto end; ++ } ++ } ++ ++ if (task->receive_buf_received == task->payload_len) { ++ size_t strip = HEADER_SIZE + sizeof(qgs_msg_get_quote_resp_t); ++ memmove(task->receive_buf, ++ task->receive_buf + strip, ++ task->receive_buf_received - strip); ++ task->receive_buf_received -= strip; ++ task->status_code = TDX_VP_GET_QUOTE_SUCCESS; ++ goto end; ++ } ++ ++ return G_SOURCE_CONTINUE; ++ ++end: ++ tdx_generate_quote_cleanup(task); ++ return G_SOURCE_REMOVE; ++} ++ ++static gboolean tdx_send_report(QIOChannel *ioc, GIOCondition condition, ++ gpointer opaque) ++{ ++ TdxGenerateQuoteTask *task = opaque; ++ Error *err = NULL; ++ int ret; ++ ++ ret = qio_channel_write(ioc, task->send_data + task->send_data_sent, ++ task->send_data_size - task->send_data_sent, &err); ++ if (ret < 0) { ++ if (ret == QIO_CHANNEL_ERR_BLOCK) { ++ ret = 0; ++ } else { ++ error_report_err(err); ++ task->status_code = TDX_VP_GET_QUOTE_ERROR; ++ tdx_generate_quote_cleanup(task); ++ goto end; ++ } ++ } ++ task->send_data_sent += ret; ++ ++ if (task->send_data_sent == task->send_data_size) { ++ task->watch = qio_channel_add_watch(QIO_CHANNEL(task->sioc), G_IO_IN, ++ tdx_get_quote_read, task, NULL); ++ goto end; ++ } ++ ++ return G_SOURCE_CONTINUE; ++ ++end: ++ return G_SOURCE_REMOVE; ++} ++ ++static void tdx_quote_generator_connected(QIOTask *qio_task, gpointer opaque) ++{ ++ TdxGenerateQuoteTask *task = opaque; ++ Error *err = NULL; ++ int ret; ++ ++ ret = qio_task_propagate_error(qio_task, &err); ++ if (ret) { ++ error_report_err(err); ++ task->status_code = TDX_VP_GET_QUOTE_QGS_UNAVAILABLE; ++ tdx_generate_quote_cleanup(task); ++ return; ++ } ++ ++ task->watch = qio_channel_add_watch(QIO_CHANNEL(task->sioc), G_IO_OUT, ++ tdx_send_report, task, NULL); ++} ++ ++#define TRANSACTION_TIMEOUT 30000 ++ ++static void getquote_expired(void *opaque) ++{ ++ TdxGenerateQuoteTask *task = opaque; ++ ++ task->status_code = TDX_VP_GET_QUOTE_ERROR; ++ tdx_generate_quote_cleanup(task); ++} ++ ++static void setup_get_quote_timer(TdxGenerateQuoteTask *task) ++{ ++ int64_t time; ++ ++ timer_init_ms(&task->timer, QEMU_CLOCK_VIRTUAL, getquote_expired, task); ++ time = qemu_clock_get_ms(QEMU_CLOCK_VIRTUAL); ++ timer_mod(&task->timer, time + TRANSACTION_TIMEOUT); ++} ++ ++void tdx_generate_quote(TdxGenerateQuoteTask *task, ++ SocketAddress *qg_sock_addr) ++{ ++ QIOChannelSocket *sioc; ++ qgs_msg_get_quote_req_t msg; ++ ++ /* Prepare a QGS message prelude */ ++ msg.header.major_version = QGS_MSG_LIB_MAJOR_VER; ++ msg.header.minor_version = QGS_MSG_LIB_MINOR_VER; ++ msg.header.type = GET_QUOTE_REQ; ++ msg.header.size = sizeof(msg) + task->send_data_size; ++ msg.header.error_code = 0; ++ msg.report_size = task->send_data_size; ++ msg.id_list_size = 0; ++ ++ /* Make room to add the QGS message prelude */ ++ task->send_data = g_renew(char, ++ task->send_data, ++ task->send_data_size + sizeof(msg) + HEADER_SIZE); ++ memmove(task->send_data + sizeof(msg) + HEADER_SIZE, ++ task->send_data, ++ task->send_data_size); ++ memcpy(task->send_data + HEADER_SIZE, ++ &msg, ++ sizeof(msg)); ++ encode_header(task->send_data, HEADER_SIZE, task->send_data_size + sizeof(msg)); ++ task->send_data_size += sizeof(msg) + HEADER_SIZE; ++ ++ sioc = qio_channel_socket_new(); ++ task->sioc = sioc; ++ ++ setup_get_quote_timer(task); ++ ++ qio_channel_socket_connect_async(sioc, qg_sock_addr, ++ tdx_quote_generator_connected, task, ++ NULL, NULL); ++} +diff --git a/target/i386/kvm/tdx-quote-generator.h b/target/i386/kvm/tdx-quote-generator.h +new file mode 100644 +index 0000000000..3bd9b8ef33 +--- /dev/null ++++ b/target/i386/kvm/tdx-quote-generator.h +@@ -0,0 +1,82 @@ ++/* SPDX-License-Identifier: GPL-2.0-or-later */ ++ ++#ifndef QEMU_I386_TDX_QUOTE_GENERATOR_H ++#define QEMU_I386_TDX_QUOTE_GENERATOR_H ++ ++#include "qom/object_interfaces.h" ++#include "io/channel-socket.h" ++#include "exec/hwaddr.h" ++ ++#define TDX_GET_QUOTE_STRUCTURE_VERSION 1ULL ++ ++#define TDX_VP_GET_QUOTE_SUCCESS 0ULL ++#define TDX_VP_GET_QUOTE_IN_FLIGHT (-1ULL) ++#define TDX_VP_GET_QUOTE_ERROR 0x8000000000000000ULL ++#define TDX_VP_GET_QUOTE_QGS_UNAVAILABLE 0x8000000000000001ULL ++ ++/* Limit to avoid resource starvation. */ ++#define TDX_GET_QUOTE_MAX_BUF_LEN (128 * 1024) ++#define TDX_MAX_GET_QUOTE_REQUEST 16 ++ ++#define TDX_GET_QUOTE_HDR_SIZE 24 ++ ++/* Format of pages shared with guest. */ ++struct tdx_get_quote_header { ++ /* Format version: must be 1 in little endian. */ ++ uint64_t structure_version; ++ ++ /* ++ * GetQuote status code in little endian: ++ * Guest must set error_code to 0 to avoid information leak. ++ * Qemu sets this before interrupting guest. ++ */ ++ uint64_t error_code; ++ ++ /* ++ * in-message size in little endian: The message will follow this header. ++ * The in-message will be send to QGS. ++ */ ++ uint32_t in_len; ++ ++ /* ++ * out-message size in little endian: ++ * On request, out_len must be zero to avoid information leak. ++ * On return, message size from QGS. Qemu overwrites this field. ++ * The message will follows this header. The in-message is overwritten. ++ */ ++ uint32_t out_len; ++ ++ /* ++ * Message buffer follows. ++ * Guest sets message that will be send to QGS. If out_len > in_len, guest ++ * should zero remaining buffer to avoid information leak. ++ * Qemu overwrites this buffer with a message returned from QGS. ++ */ ++}; ++ ++typedef struct TdxGenerateQuoteTask { ++ hwaddr buf_gpa; ++ hwaddr payload_gpa; ++ uint64_t payload_len; ++ ++ char *send_data; ++ uint64_t send_data_size; ++ uint64_t send_data_sent; ++ ++ char *receive_buf; ++ uint64_t receive_buf_received; ++ ++ uint64_t status_code; ++ struct tdx_get_quote_header hdr; ++ ++ QIOChannelSocket *sioc; ++ guint watch; ++ QEMUTimer timer; ++ ++ void (*completion)(struct TdxGenerateQuoteTask *task); ++ void *opaque; ++} TdxGenerateQuoteTask; ++ ++void tdx_generate_quote(TdxGenerateQuoteTask *task, SocketAddress *qg_sock_addr); ++ ++#endif /* QEMU_I386_TDX_QUOTE_GENERATOR_H */ +diff --git a/target/i386/kvm/tdx-stub.c b/target/i386/kvm/tdx-stub.c +index 62a12a0677..76fee49eff 100644 +--- a/target/i386/kvm/tdx-stub.c ++++ b/target/i386/kvm/tdx-stub.c +@@ -19,6 +19,10 @@ int tdx_handle_report_fatal_error(X86CPU *cpu, struct kvm_run *run) + return -EINVAL; + } + ++void tdx_handle_get_quote(X86CPU *cpu, struct kvm_run *run) ++{ ++} ++ + void tdx_handle_get_tdvmcall_info(X86CPU *cpu, struct kvm_run *run) + { + } +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index e0197b6582..201da78b06 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -14,12 +14,14 @@ + #include "qemu/base64.h" + #include "qemu/mmap-alloc.h" + #include "qapi/error.h" ++#include "qapi/qapi-visit-sockets.h" + #include "qom/object_interfaces.h" + #include "crypto/hash.h" + #include "sysemu/kvm_int.h" + #include "sysemu/runstate.h" + #include "sysemu/sysemu.h" + #include "exec/ramblock.h" ++#include "exec/address-spaces.h" + + #include + +@@ -32,6 +34,7 @@ + #include "hw/i386/tdvf-hob.h" + #include "kvm_i386.h" + #include "tdx.h" ++#include "tdx-quote-generator.h" + + #include "standard-headers/asm-x86/kvm_para.h" + +@@ -1120,13 +1123,146 @@ int tdx_parse_tdvf(void *flash_ptr, int size) + return tdvf_parse_metadata(&tdx_guest->tdvf, flash_ptr, size); + } + ++static void tdx_get_quote_completion(TdxGenerateQuoteTask *task) ++{ ++ TdxGuest *tdx = task->opaque; ++ int ret; ++ ++ /* Maintain the number of in-flight requests. */ ++ qemu_mutex_lock(&tdx->lock); ++ tdx->num--; ++ qemu_mutex_unlock(&tdx->lock); ++ ++ if (task->status_code == TDX_VP_GET_QUOTE_SUCCESS) { ++ ret = address_space_write(&address_space_memory, task->payload_gpa, ++ MEMTXATTRS_UNSPECIFIED, task->receive_buf, ++ task->receive_buf_received); ++ if (ret != MEMTX_OK) { ++ error_report("TDX: get-quote: failed to write quote data."); ++ } else { ++ task->hdr.out_len = cpu_to_le64(task->receive_buf_received); ++ } ++ } ++ task->hdr.error_code = cpu_to_le64(task->status_code); ++ ++ /* Publish the response contents before marking this request completed. */ ++ smp_wmb(); ++ ret = address_space_write(&address_space_memory, task->buf_gpa, ++ MEMTXATTRS_UNSPECIFIED, &task->hdr, ++ TDX_GET_QUOTE_HDR_SIZE); ++ if (ret != MEMTX_OK) { ++ error_report("TDX: get-quote: failed to update GetQuote header."); ++ } ++ ++ g_free(task->send_data); ++ g_free(task->receive_buf); ++ g_free(task); ++ object_unref(tdx); ++} ++ ++void tdx_handle_get_quote(X86CPU *cpu, struct kvm_run *run) ++{ ++ TdxGenerateQuoteTask *task; ++ struct tdx_get_quote_header hdr; ++ hwaddr buf_gpa = run->tdx.get_quote.gpa; ++ uint64_t buf_len = run->tdx.get_quote.size; ++ ++ QEMU_BUILD_BUG_ON(sizeof(struct tdx_get_quote_header) != TDX_GET_QUOTE_HDR_SIZE); ++ ++ run->tdx.get_quote.ret = TDG_VP_VMCALL_INVALID_OPERAND; ++ ++ if (buf_len == 0) { ++ return; ++ } ++ ++ if (!QEMU_IS_ALIGNED(buf_gpa, 4096) || !QEMU_IS_ALIGNED(buf_len, 4096)) { ++ run->tdx.get_quote.ret = TDG_VP_VMCALL_ALIGN_ERROR; ++ return; ++ } ++ ++ if (address_space_read(&address_space_memory, buf_gpa, MEMTXATTRS_UNSPECIFIED, ++ &hdr, TDX_GET_QUOTE_HDR_SIZE) != MEMTX_OK) { ++ error_report("TDX: get-quote: failed to read GetQuote header."); ++ return; ++ } ++ ++ if (le64_to_cpu(hdr.structure_version) != TDX_GET_QUOTE_STRUCTURE_VERSION) { ++ return; ++ } ++ ++ /* Only safe-guard check to avoid too large buffer size. */ ++ if (buf_len > TDX_GET_QUOTE_MAX_BUF_LEN || ++ le32_to_cpu(hdr.in_len) > buf_len - TDX_GET_QUOTE_HDR_SIZE) { ++ return; ++ } ++ ++ if (!tdx_guest->qg_sock_addr) { ++ hdr.error_code = cpu_to_le64(TDX_VP_GET_QUOTE_QGS_UNAVAILABLE); ++ if (address_space_write(&address_space_memory, buf_gpa, ++ MEMTXATTRS_UNSPECIFIED, ++ &hdr, TDX_GET_QUOTE_HDR_SIZE) != MEMTX_OK) { ++ error_report("TDX: failed to update GetQuote header."); ++ return; ++ } ++ run->tdx.get_quote.ret = TDG_VP_VMCALL_SUCCESS; ++ return; ++ } ++ ++ qemu_mutex_lock(&tdx_guest->lock); ++ if (tdx_guest->num >= TDX_MAX_GET_QUOTE_REQUEST) { ++ qemu_mutex_unlock(&tdx_guest->lock); ++ run->tdx.get_quote.ret = TDG_VP_VMCALL_RETRY; ++ return; ++ } ++ tdx_guest->num++; ++ qemu_mutex_unlock(&tdx_guest->lock); ++ ++ task = g_new(TdxGenerateQuoteTask, 1); ++ task->buf_gpa = buf_gpa; ++ task->payload_gpa = buf_gpa + TDX_GET_QUOTE_HDR_SIZE; ++ task->payload_len = buf_len - TDX_GET_QUOTE_HDR_SIZE; ++ task->hdr = hdr; ++ task->completion = tdx_get_quote_completion; ++ ++ task->send_data_size = le32_to_cpu(hdr.in_len); ++ task->send_data = g_malloc(task->send_data_size); ++ task->send_data_sent = 0; ++ ++ if (address_space_read(&address_space_memory, task->payload_gpa, ++ MEMTXATTRS_UNSPECIFIED, task->send_data, ++ task->send_data_size) != MEMTX_OK) { ++ goto out_free; ++ } ++ ++ /* Mark the buffer in-flight. */ ++ hdr.error_code = cpu_to_le64(TDX_VP_GET_QUOTE_IN_FLIGHT); ++ if (address_space_write(&address_space_memory, buf_gpa, ++ MEMTXATTRS_UNSPECIFIED, ++ &hdr, TDX_GET_QUOTE_HDR_SIZE) != MEMTX_OK) { ++ goto out_free; ++ } ++ ++ task->receive_buf = g_malloc0(task->payload_len); ++ task->receive_buf_received = 0; ++ task->opaque = tdx_guest; ++ ++ object_ref(tdx_guest); ++ tdx_generate_quote(task, tdx_guest->qg_sock_addr); ++ run->tdx.get_quote.ret = TDG_VP_VMCALL_SUCCESS; ++ return; ++ ++out_free: ++ g_free(task->send_data); ++ g_free(task); ++} ++ + void tdx_handle_get_tdvmcall_info(X86CPU *cpu, struct kvm_run *run) + { + if (run->tdx.get_tdvmcall_info.leaf != 1) { + return; + } + +- run->tdx.get_tdvmcall_info.r11 = 0; ++ run->tdx.get_tdvmcall_info.r11 = TDG_VP_VMCALL_SUBFUNC_GET_QUOTE; + run->tdx.get_tdvmcall_info.r12 = 0; + run->tdx.get_tdvmcall_info.r13 = 0; + run->tdx.get_tdvmcall_info.r14 = 0; +@@ -1263,6 +1399,37 @@ static void tdx_guest_set_mrownerconfig(Object *obj, const char *value, Error ** + tdx->mrownerconfig = g_strdup(value); + } + ++static void tdx_guest_get_qgs(Object *obj, Visitor *v, ++ const char *name, void *opaque, ++ Error **errp) ++{ ++ TdxGuest *tdx = TDX_GUEST(obj); ++ ++ if (!tdx->qg_sock_addr) { ++ error_setg(errp, "quote-generation-socket is not set"); ++ return; ++ } ++ visit_type_SocketAddress(v, name, &tdx->qg_sock_addr, errp); ++} ++ ++static void tdx_guest_set_qgs(Object *obj, Visitor *v, ++ const char *name, void *opaque, ++ Error **errp) ++{ ++ TdxGuest *tdx = TDX_GUEST(obj); ++ SocketAddress *sock = NULL; ++ ++ if (!visit_type_SocketAddress(v, name, &sock, errp)) { ++ return; ++ } ++ ++ if (tdx->qg_sock_addr) { ++ qapi_free_SocketAddress(tdx->qg_sock_addr); ++ } ++ ++ tdx->qg_sock_addr = sock; ++} ++ + /* tdx guest */ + OBJECT_DEFINE_TYPE_WITH_INTERFACES(TdxGuest, + tdx_guest, +@@ -1294,6 +1461,13 @@ static void tdx_guest_init(Object *obj) + object_property_add_str(obj, "mrownerconfig", + tdx_guest_get_mrownerconfig, + tdx_guest_set_mrownerconfig); ++ ++ object_property_add(obj, "quote-generation-socket", "SocketAddress", ++ tdx_guest_get_qgs, ++ tdx_guest_set_qgs, ++ NULL, NULL); ++ ++ qemu_mutex_init(&tdx->lock); + } + + static void tdx_guest_finalize(Object *obj) +diff --git a/target/i386/kvm/tdx.h b/target/i386/kvm/tdx.h +index 0dd41d5811..35a09c19c5 100644 +--- a/target/i386/kvm/tdx.h ++++ b/target/i386/kvm/tdx.h +@@ -11,6 +11,8 @@ + #include "cpu.h" + #include "hw/i386/tdvf.h" + ++#include "tdx-quote-generator.h" ++ + #define TYPE_TDX_GUEST "tdx-guest" + #define TDX_GUEST(obj) OBJECT_CHECK(TdxGuest, (obj), TYPE_TDX_GUEST) + +@@ -22,6 +24,7 @@ typedef struct TdxGuestClass { + #define TDX_APIC_BUS_CYCLES_NS 40 + + #define TDVMCALL_GET_TD_VM_CALL_INFO 0x10000 ++#define TDVMCALL_GET_QUOTE 0x10002 + + #define TDG_VP_VMCALL_SUCCESS 0x0000000000000000ULL + #define TDG_VP_VMCALL_RETRY 0x0000000000000001ULL +@@ -29,6 +32,8 @@ typedef struct TdxGuestClass { + #define TDG_VP_VMCALL_GPA_INUSE 0x8000000000000001ULL + #define TDG_VP_VMCALL_ALIGN_ERROR 0x8000000000000002ULL + ++#define TDG_VP_VMCALL_SUBFUNC_GET_QUOTE 0x0000000000000001ULL ++ + enum TdxRamType { + TDX_RAM_UNACCEPTED, + TDX_RAM_ADDED, +@@ -57,6 +62,10 @@ typedef struct TdxGuest { + + uint32_t nr_ram_entries; + TdxRamEntry *ram_entries; ++ ++ /* GetQuote */ ++ SocketAddress *qg_sock_addr; ++ int num; + } TdxGuest; + + #ifdef CONFIG_TDX +@@ -69,6 +78,7 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp); + void tdx_set_tdvf_region(MemoryRegion *tdvf_mr); + int tdx_parse_tdvf(void *flash_ptr, int size); + int tdx_handle_report_fatal_error(X86CPU *cpu, struct kvm_run *run); ++void tdx_handle_get_quote(X86CPU *cpu, struct kvm_run *run); + void tdx_handle_get_tdvmcall_info(X86CPU *cpu, struct kvm_run *run); + + #endif /* QEMU_I386_TDX_H */ +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-handle-TDG.VP.VMCALL-GetTdVmCallInfo.patch b/SOURCES/kvm-i386-tdx-handle-TDG.VP.VMCALL-GetTdVmCallInfo.patch new file mode 100644 index 0000000..9205f08 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-handle-TDG.VP.VMCALL-GetTdVmCallInfo.patch @@ -0,0 +1,111 @@ +From 432fbc1dacdb5de4fb3af42e21749c53cd26a855 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:49 +0200 +Subject: [PATCH 092/115] i386/tdx: handle TDG.VP.VMCALL + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [92/115] 443c167d16e725f756cab6371aad04b7c2b4f15a (bonzini/rhel-qemu-kvm) + +Signed-off-by: Binbin Wu +Signed-off-by: Paolo Bonzini +(cherry picked from commit 427b8cf47a6959cd8b0db12bcf66e9009afa2c07) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/kvm.c | 12 ++++++++++++ + target/i386/kvm/tdx-stub.c | 4 ++++ + target/i386/kvm/tdx.c | 12 ++++++++++++ + target/i386/kvm/tdx.h | 9 +++++++++ + 4 files changed, 37 insertions(+) + +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index b6fddcd543..26328a1d3b 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -6039,6 +6039,18 @@ int kvm_arch_handle_exit(CPUState *cs, struct kvm_run *run) + break; + } + break; ++ case KVM_EXIT_TDX: ++ /* ++ * run->tdx is already set up for the case where userspace ++ * does not handle the TDVMCALL. ++ */ ++ switch (run->tdx.nr) { ++ case TDVMCALL_GET_TD_VM_CALL_INFO: ++ tdx_handle_get_tdvmcall_info(cpu, run); ++ break; ++ } ++ ret = 0; ++ break; + default: + fprintf(stderr, "KVM: unknown exit reason %d\n", run->exit_reason); + ret = -1; +diff --git a/target/i386/kvm/tdx-stub.c b/target/i386/kvm/tdx-stub.c +index 720a4ff046..62a12a0677 100644 +--- a/target/i386/kvm/tdx-stub.c ++++ b/target/i386/kvm/tdx-stub.c +@@ -18,3 +18,7 @@ int tdx_handle_report_fatal_error(X86CPU *cpu, struct kvm_run *run) + { + return -EINVAL; + } ++ ++void tdx_handle_get_tdvmcall_info(X86CPU *cpu, struct kvm_run *run) ++{ ++} +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index ed3a55991a..e0197b6582 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -1120,6 +1120,18 @@ int tdx_parse_tdvf(void *flash_ptr, int size) + return tdvf_parse_metadata(&tdx_guest->tdvf, flash_ptr, size); + } + ++void tdx_handle_get_tdvmcall_info(X86CPU *cpu, struct kvm_run *run) ++{ ++ if (run->tdx.get_tdvmcall_info.leaf != 1) { ++ return; ++ } ++ ++ run->tdx.get_tdvmcall_info.r11 = 0; ++ run->tdx.get_tdvmcall_info.r12 = 0; ++ run->tdx.get_tdvmcall_info.r13 = 0; ++ run->tdx.get_tdvmcall_info.r14 = 0; ++} ++ + static void tdx_panicked_on_fatal_error(X86CPU *cpu, uint64_t error_code, + char *message, uint64_t gpa) + { +diff --git a/target/i386/kvm/tdx.h b/target/i386/kvm/tdx.h +index 8dd66e9014..0dd41d5811 100644 +--- a/target/i386/kvm/tdx.h ++++ b/target/i386/kvm/tdx.h +@@ -21,6 +21,14 @@ typedef struct TdxGuestClass { + /* TDX requires bus frequency 25MHz */ + #define TDX_APIC_BUS_CYCLES_NS 40 + ++#define TDVMCALL_GET_TD_VM_CALL_INFO 0x10000 ++ ++#define TDG_VP_VMCALL_SUCCESS 0x0000000000000000ULL ++#define TDG_VP_VMCALL_RETRY 0x0000000000000001ULL ++#define TDG_VP_VMCALL_INVALID_OPERAND 0x8000000000000000ULL ++#define TDG_VP_VMCALL_GPA_INUSE 0x8000000000000001ULL ++#define TDG_VP_VMCALL_ALIGN_ERROR 0x8000000000000002ULL ++ + enum TdxRamType { + TDX_RAM_UNACCEPTED, + TDX_RAM_ADDED, +@@ -61,5 +69,6 @@ int tdx_pre_create_vcpu(CPUState *cpu, Error **errp); + void tdx_set_tdvf_region(MemoryRegion *tdvf_mr); + int tdx_parse_tdvf(void *flash_ptr, int size); + int tdx_handle_report_fatal_error(X86CPU *cpu, struct kvm_run *run); ++void tdx_handle_get_tdvmcall_info(X86CPU *cpu, struct kvm_run *run); + + #endif /* QEMU_I386_TDX_H */ +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-handle-TDVMCALL_SETUP_EVENT_NOTIFY_INTERRUP.patch b/SOURCES/kvm-i386-tdx-handle-TDVMCALL_SETUP_EVENT_NOTIFY_INTERRUP.patch new file mode 100644 index 0000000..e1bafaa --- /dev/null +++ b/SOURCES/kvm-i386-tdx-handle-TDVMCALL_SETUP_EVENT_NOTIFY_INTERRUP.patch @@ -0,0 +1,197 @@ +From e2157a814aa2a700736e97d2548460e9c6c808f2 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:50 +0200 +Subject: [PATCH 101/115] i386/tdx: handle + TDVMCALL_SETUP_EVENT_NOTIFY_INTERRUPT + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [101/115] 747409544735f7e2da43ce3906c0d0d487b5d78f (bonzini/rhel-qemu-kvm) + +Record the interrupt vector and the apic id of the vcpu that calls +TDVMCALL_SETUP_EVENT_NOTIFY_INTERRUPT. + +Inject the interrupt to TD guest to notify the completion of +when notify interrupt vector is valid. + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20250703024021.3559286-5-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit efa742b23eff2a799c196d756bd506fe74e96fdc) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/kvm.c | 3 +++ + target/i386/kvm/tdx-stub.c | 4 ++++ + target/i386/kvm/tdx.c | 48 +++++++++++++++++++++++++++++++++++++- + target/i386/kvm/tdx.h | 7 ++++++ + 4 files changed, 61 insertions(+), 1 deletion(-) + +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index fbf11b2122..f789965a33 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -6051,6 +6051,9 @@ int kvm_arch_handle_exit(CPUState *cs, struct kvm_run *run) + case TDVMCALL_GET_TD_VM_CALL_INFO: + tdx_handle_get_tdvmcall_info(cpu, run); + break; ++ case TDVMCALL_SETUP_EVENT_NOTIFY_INTERRUPT: ++ tdx_handle_setup_event_notify_interrupt(cpu, run); ++ break; + } + ret = 0; + break; +diff --git a/target/i386/kvm/tdx-stub.c b/target/i386/kvm/tdx-stub.c +index 76fee49eff..1f0e108a69 100644 +--- a/target/i386/kvm/tdx-stub.c ++++ b/target/i386/kvm/tdx-stub.c +@@ -26,3 +26,7 @@ void tdx_handle_get_quote(X86CPU *cpu, struct kvm_run *run) + void tdx_handle_get_tdvmcall_info(X86CPU *cpu, struct kvm_run *run) + { + } ++ ++void tdx_handle_setup_event_notify_interrupt(X86CPU *cpu, struct kvm_run *run) ++{ ++} +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index a24e15571a..e56db74f58 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -28,10 +28,13 @@ + #include "cpu.h" + #include "cpu-internal.h" + #include "host-cpu.h" ++#include "hw/i386/apic_internal.h" ++#include "hw/i386/apic-msidef.h" + #include "hw/i386/e820_memory_layout.h" + #include "hw/i386/tdvf.h" + #include "hw/i386/x86.h" + #include "hw/i386/tdvf-hob.h" ++#include "hw/pci/msi.h" + #include "kvm_i386.h" + #include "tdx.h" + #include "tdx-quote-generator.h" +@@ -1123,6 +1126,28 @@ int tdx_parse_tdvf(void *flash_ptr, int size) + return tdvf_parse_metadata(&tdx_guest->tdvf, flash_ptr, size); + } + ++static void tdx_inject_interrupt(uint32_t apicid, uint32_t vector) ++{ ++ int ret; ++ ++ if (vector < 32 || vector > 255) { ++ return; ++ } ++ ++ MSIMessage msg = { ++ .address = ((apicid & 0xff) << MSI_ADDR_DEST_ID_SHIFT) | ++ (((uint64_t)apicid & 0xffffff00) << 32), ++ .data = vector | (APIC_DM_FIXED << MSI_DATA_DELIVERY_MODE_SHIFT), ++ }; ++ ++ ret = kvm_irqchip_send_msi(kvm_state, msg); ++ if (ret < 0) { ++ /* In this case, no better way to tell it to guest. Log it. */ ++ error_report("TDX: injection interrupt %d failed, interrupt lost (%s).", ++ vector, strerror(-ret)); ++ } ++} ++ + static void tdx_get_quote_completion(TdxGenerateQuoteTask *task) + { + TdxGuest *tdx = task->opaque; +@@ -1154,6 +1179,9 @@ static void tdx_get_quote_completion(TdxGenerateQuoteTask *task) + error_report("TDX: get-quote: failed to update GetQuote header."); + } + ++ tdx_inject_interrupt(tdx_guest->event_notify_apicid, ++ tdx_guest->event_notify_vector); ++ + g_free(task->send_data); + g_free(task->receive_buf); + g_free(task); +@@ -1256,7 +1284,7 @@ out_free: + g_free(task); + } + +-#define SUPPORTED_TDVMCALLINFO_1_R11 (0) ++#define SUPPORTED_TDVMCALLINFO_1_R11 (TDG_VP_VMCALL_SUBFUNC_SET_EVENT_NOTIFY_INTERRUPT) + #define SUPPORTED_TDVMCALLINFO_1_R12 (0) + + void tdx_handle_get_tdvmcall_info(X86CPU *cpu, struct kvm_run *run) +@@ -1277,6 +1305,21 @@ void tdx_handle_get_tdvmcall_info(X86CPU *cpu, struct kvm_run *run) + run->tdx.get_tdvmcall_info.ret = TDG_VP_VMCALL_SUCCESS; + } + ++void tdx_handle_setup_event_notify_interrupt(X86CPU *cpu, struct kvm_run *run) ++{ ++ uint64_t vector = run->tdx.setup_event_notify.vector; ++ ++ if (vector >= 32 && vector < 256) { ++ qemu_mutex_lock(&tdx_guest->lock); ++ tdx_guest->event_notify_vector = vector; ++ tdx_guest->event_notify_apicid = cpu->apic_id; ++ qemu_mutex_unlock(&tdx_guest->lock); ++ run->tdx.setup_event_notify.ret = TDG_VP_VMCALL_SUCCESS; ++ } else { ++ run->tdx.setup_event_notify.ret = TDG_VP_VMCALL_INVALID_OPERAND; ++ } ++} ++ + static void tdx_panicked_on_fatal_error(X86CPU *cpu, uint64_t error_code, + char *message, uint64_t gpa) + { +@@ -1477,6 +1520,9 @@ static void tdx_guest_init(Object *obj) + NULL, NULL); + + qemu_mutex_init(&tdx->lock); ++ ++ tdx->event_notify_vector = -1; ++ tdx->event_notify_apicid = -1; + } + + static void tdx_guest_finalize(Object *obj) +diff --git a/target/i386/kvm/tdx.h b/target/i386/kvm/tdx.h +index d439078a87..1c38faf983 100644 +--- a/target/i386/kvm/tdx.h ++++ b/target/i386/kvm/tdx.h +@@ -25,6 +25,7 @@ typedef struct TdxGuestClass { + + #define TDVMCALL_GET_TD_VM_CALL_INFO 0x10000 + #define TDVMCALL_GET_QUOTE 0x10002 ++#define TDVMCALL_SETUP_EVENT_NOTIFY_INTERRUPT 0x10004 + + #define TDG_VP_VMCALL_SUCCESS 0x0000000000000000ULL + #define TDG_VP_VMCALL_RETRY 0x0000000000000001ULL +@@ -32,6 +33,8 @@ typedef struct TdxGuestClass { + #define TDG_VP_VMCALL_GPA_INUSE 0x8000000000000001ULL + #define TDG_VP_VMCALL_ALIGN_ERROR 0x8000000000000002ULL + ++#define TDG_VP_VMCALL_SUBFUNC_SET_EVENT_NOTIFY_INTERRUPT BIT_ULL(1) ++ + enum TdxRamType { + TDX_RAM_UNACCEPTED, + TDX_RAM_ADDED, +@@ -64,6 +67,9 @@ typedef struct TdxGuest { + /* GetQuote */ + SocketAddress *qg_sock_addr; + int num; ++ ++ uint32_t event_notify_vector; ++ uint32_t event_notify_apicid; + } TdxGuest; + + #ifdef CONFIG_TDX +@@ -78,5 +84,6 @@ int tdx_parse_tdvf(void *flash_ptr, int size); + int tdx_handle_report_fatal_error(X86CPU *cpu, struct kvm_run *run); + void tdx_handle_get_quote(X86CPU *cpu, struct kvm_run *run); + void tdx_handle_get_tdvmcall_info(X86CPU *cpu, struct kvm_run *run); ++void tdx_handle_setup_event_notify_interrupt(X86CPU *cpu, struct kvm_run *run); + + #endif /* QEMU_I386_TDX_H */ +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-implement-tdx_cpu_instance_init.patch b/SOURCES/kvm-i386-tdx-implement-tdx_cpu_instance_init.patch new file mode 100644 index 0000000..5b1fc65 --- /dev/null +++ b/SOURCES/kvm-i386-tdx-implement-tdx_cpu_instance_init.patch @@ -0,0 +1,50 @@ +From 99563f467faeace2726e62292b128faf3fb56fcb Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 061/115] i386/tdx: implement tdx_cpu_instance_init() + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [61/115] 20d53c1ff1d645a043cac27f33f9778732c1469e (bonzini/rhel-qemu-kvm) + +Currently, pmu is not supported for TDX by KVM. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-33-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 7c615242671dbe65e198c20889dcaa9b4b9a1624) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 6 ++++++ + 1 file changed, 6 insertions(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 7611f51aae..afd7e62422 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -398,6 +398,11 @@ static int tdx_kvm_type(X86ConfidentialGuest *cg) + return KVM_X86_TDX_VM; + } + ++static void tdx_cpu_instance_init(X86ConfidentialGuest *cg, CPUState *cpu) ++{ ++ object_property_set_bool(OBJECT(cpu), "pmu", false, &error_abort); ++} ++ + static int tdx_validate_attributes(TdxGuest *tdx, Error **errp) + { + if ((tdx->attributes & ~tdx_caps->supported_attrs)) { +@@ -791,4 +796,5 @@ static void tdx_guest_class_init(ObjectClass *oc, void *data) + + klass->kvm_init = tdx_kvm_init; + x86_klass->kvm_type = tdx_kvm_type; ++ x86_klass->cpu_instance_init = tdx_cpu_instance_init; + } +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-tdx-load-TDVF-for-TD-guest.patch b/SOURCES/kvm-i386-tdx-load-TDVF-for-TD-guest.patch new file mode 100644 index 0000000..0e0b11f --- /dev/null +++ b/SOURCES/kvm-i386-tdx-load-TDVF-for-TD-guest.patch @@ -0,0 +1,105 @@ +From 69a39313aa07e1e4845e53e34a6b866395863c73 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:46 +0200 +Subject: [PATCH 045/115] i386/tdx: load TDVF for TD guest + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [45/115] cdf138cb88831b4c28f78cd9ad72d75465a731cf (bonzini/rhel-qemu-kvm) + +TDVF(OVMF) needs to run at private memory for TD guest. TDX cannot +support pflash device since it doesn't support read-only private memory. +Thus load TDVF(OVMF) with -bios option for TDs. + +Use memory_region_init_ram_guest_memfd() to allocate the MemoryRegion +for TDVF because it needs to be located at private memory. + +Also store the MemoryRegion pointer of TDVF since the shared ramblock of +it can be discared after it gets copied to private ramblock. + +Signed-off-by: Chao Peng +Co-developed-by: Xiaoyao Li +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-17-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 0dd5fe5ebeabefc7b3d7f043991b1edfe6b8eda9) +Signed-off-by: Paolo Bonzini +--- + hw/i386/x86-common.c | 6 +++++- + target/i386/kvm/tdx.c | 6 ++++++ + target/i386/kvm/tdx.h | 3 +++ + 3 files changed, 14 insertions(+), 1 deletion(-) + +diff --git a/hw/i386/x86-common.c b/hw/i386/x86-common.c +index 562215990d..97cd84d618 100644 +--- a/hw/i386/x86-common.c ++++ b/hw/i386/x86-common.c +@@ -44,6 +44,7 @@ + #include "standard-headers/asm-x86/bootparam.h" + #include CONFIG_DEVICES + #include "kvm/kvm_i386.h" ++#include "kvm/tdx.h" + + #ifdef CONFIG_XEN_EMU + #include "hw/xen/xen.h" +@@ -1003,11 +1004,14 @@ void x86_bios_rom_init(X86MachineState *x86ms, const char *default_firmware, + if (machine_require_guest_memfd(MACHINE(x86ms))) { + memory_region_init_ram_guest_memfd(&x86ms->bios, NULL, "pc.bios", + bios_size, &error_fatal); ++ if (is_tdx_vm()) { ++ tdx_set_tdvf_region(&x86ms->bios); ++ } + } else { + memory_region_init_ram(&x86ms->bios, NULL, "pc.bios", + bios_size, &error_fatal); + } +- if (sev_enabled()) { ++ if (sev_enabled() || is_tdx_vm()) { + /* + * The concept of a "reset" simply doesn't exist for + * confidential computing guests, we have to destroy and +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 56ad5f599d..2522f2030d 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -137,6 +137,12 @@ static int get_tdx_capabilities(Error **errp) + return 0; + } + ++void tdx_set_tdvf_region(MemoryRegion *tdvf_mr) ++{ ++ assert(!tdx_guest->tdvf_mr); ++ tdx_guest->tdvf_mr = tdvf_mr; ++} ++ + static int tdx_kvm_init(ConfidentialGuestSupport *cgs, Error **errp) + { + TdxGuest *tdx = TDX_GUEST(cgs); +diff --git a/target/i386/kvm/tdx.h b/target/i386/kvm/tdx.h +index d39e733d9f..b73461b8d8 100644 +--- a/target/i386/kvm/tdx.h ++++ b/target/i386/kvm/tdx.h +@@ -30,6 +30,8 @@ typedef struct TdxGuest { + char *mrconfigid; /* base64 encoded sha348 digest */ + char *mrowner; /* base64 encoded sha348 digest */ + char *mrownerconfig; /* base64 encoded sha348 digest */ ++ ++ MemoryRegion *tdvf_mr; + } TdxGuest; + + #ifdef CONFIG_TDX +@@ -39,5 +41,6 @@ bool is_tdx_vm(void); + #endif /* CONFIG_TDX */ + + int tdx_pre_create_vcpu(CPUState *cpu, Error **errp); ++void tdx_set_tdvf_region(MemoryRegion *tdvf_mr); + + #endif /* QEMU_I386_TDX_H */ +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-topology-Introduce-helpers-for-various-topology.patch b/SOURCES/kvm-i386-topology-Introduce-helpers-for-various-topology.patch new file mode 100644 index 0000000..49a2d4e --- /dev/null +++ b/SOURCES/kvm-i386-topology-Introduce-helpers-for-various-topology.patch @@ -0,0 +1,101 @@ +From 693fa1348914dfa37ce5fa4a654e67d573e4941a Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 011/115] i386/topology: Introduce helpers for various topology + info of different level + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [11/115] 52179ff6718bd4bfd9eac78efd20e7e5337d3b07 (bonzini/rhel-qemu-kvm) + +Introduce various helpers for getting the topology info of different +semantics. Using the helper is more self-explanatory. + +Besides, the semantic of the helper will stay unchanged even when new +topology is added in the future. At that time, updating the +implementation of the helper without affecting the callers. + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20241219110125.1266461-6-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit e60cbeec190d349682bf97cf55446e8ae260b11a) +Signed-off-by: Paolo Bonzini +(cherry picked from commit 12b80c6f6e6edf61c998aa5bac9d0c44a81144a0) +Signed-off-by: Paolo Bonzini +--- + include/hw/i386/topology.h | 25 +++++++++++++++++++++++++ + target/i386/cpu.c | 11 ++++------- + 2 files changed, 29 insertions(+), 7 deletions(-) + +diff --git a/include/hw/i386/topology.h b/include/hw/i386/topology.h +index 1880df621a..e533a117c3 100644 +--- a/include/hw/i386/topology.h ++++ b/include/hw/i386/topology.h +@@ -217,4 +217,29 @@ static inline bool x86_has_extended_topo(unsigned long *topo_bitmap) + test_bit(CPU_TOPO_LEVEL_DIE, topo_bitmap); + } + ++static inline unsigned x86_module_per_pkg(X86CPUTopoInfo *topo_info) ++{ ++ return topo_info->modules_per_die * topo_info->dies_per_pkg; ++} ++ ++static inline unsigned x86_cores_per_pkg(X86CPUTopoInfo *topo_info) ++{ ++ return topo_info->cores_per_module * x86_module_per_pkg(topo_info); ++} ++ ++static inline unsigned x86_threads_per_pkg(X86CPUTopoInfo *topo_info) ++{ ++ return topo_info->threads_per_core * x86_cores_per_pkg(topo_info); ++} ++ ++static inline unsigned x86_threads_per_module(X86CPUTopoInfo *topo_info) ++{ ++ return topo_info->threads_per_core * topo_info->cores_per_module; ++} ++ ++static inline unsigned x86_threads_per_die(X86CPUTopoInfo *topo_info) ++{ ++ return x86_threads_per_module(topo_info) * topo_info->modules_per_die; ++} ++ + #endif /* HW_I386_TOPOLOGY_H */ +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index b639769ef3..1c79eb9a06 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -312,13 +312,11 @@ static uint32_t num_threads_by_topo_level(X86CPUTopoInfo *topo_info, + case CPU_TOPO_LEVEL_CORE: + return topo_info->threads_per_core; + case CPU_TOPO_LEVEL_MODULE: +- return topo_info->threads_per_core * topo_info->cores_per_module; ++ return x86_threads_per_module(topo_info); + case CPU_TOPO_LEVEL_DIE: +- return topo_info->threads_per_core * topo_info->cores_per_module * +- topo_info->modules_per_die; ++ return x86_threads_per_die(topo_info); + case CPU_TOPO_LEVEL_PACKAGE: +- return topo_info->threads_per_core * topo_info->cores_per_module * +- topo_info->modules_per_die * topo_info->dies_per_pkg; ++ return x86_threads_per_pkg(topo_info); + default: + g_assert_not_reached(); + } +@@ -6957,8 +6955,7 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + topo_info.cores_per_module = cs->nr_cores / env->nr_dies / env->nr_modules; + topo_info.threads_per_core = cs->nr_threads; + +- threads_per_pkg = topo_info.threads_per_core * topo_info.cores_per_module * +- topo_info.modules_per_die * topo_info.dies_per_pkg; ++ threads_per_pkg = x86_threads_per_pkg(&topo_info); + + /* Calculate & apply limits for different index ranges */ + if (index >= 0xC0000000) { +-- +2.50.1 + diff --git a/SOURCES/kvm-i386-topology-Update-the-comment-of-x86_apicid_from_.patch b/SOURCES/kvm-i386-topology-Update-the-comment-of-x86_apicid_from_.patch new file mode 100644 index 0000000..090a4c7 --- /dev/null +++ b/SOURCES/kvm-i386-topology-Update-the-comment-of-x86_apicid_from_.patch @@ -0,0 +1,49 @@ +From 544ec72c98bdb325589cf9cddc7356ce5d4ae586 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 010/115] i386/topology: Update the comment of + x86_apicid_from_topo_ids() + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [10/115] 62641dcd4cb51e9d252e4d5b3257ac87cfb165c0 (bonzini/rhel-qemu-kvm) + +Update the comment of x86_apicid_from_topo_ids() to match the current +implementation, + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20241219110125.1266461-5-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 8f78378de70fc79fdc7e1318496bd91ddd22df49) +Signed-off-by: Paolo Bonzini +(cherry picked from commit a4452a3e65f1fdbbafd22e9089864d37d323d7e0) +Signed-off-by: Paolo Bonzini +--- + include/hw/i386/topology.h | 5 +++-- + 1 file changed, 3 insertions(+), 2 deletions(-) + +diff --git a/include/hw/i386/topology.h b/include/hw/i386/topology.h +index dff49fce11..1880df621a 100644 +--- a/include/hw/i386/topology.h ++++ b/include/hw/i386/topology.h +@@ -135,9 +135,10 @@ static inline unsigned apicid_pkg_offset(X86CPUTopoInfo *topo_info) + } + + /* +- * Make APIC ID for the CPU based on Pkg_ID, Core_ID, SMT_ID ++ * Make APIC ID for the CPU based on topology and IDs of each topology level. + * +- * The caller must make sure core_id < nr_cores and smt_id < nr_threads. ++ * The caller must make sure the ID of each level doesn't exceed the width of ++ * the level. + */ + static inline apic_id_t x86_apicid_from_topo_ids(X86CPUTopoInfo *topo_info, + const X86CPUTopoIDs *topo_ids) +-- +2.50.1 + diff --git a/SOURCES/kvm-include-qemu-compiler-add-QEMU_UNINITIALIZED-attribu.patch b/SOURCES/kvm-include-qemu-compiler-add-QEMU_UNINITIALIZED-attribu.patch index b732a76..a196764 100644 --- a/SOURCES/kvm-include-qemu-compiler-add-QEMU_UNINITIALIZED-attribu.patch +++ b/SOURCES/kvm-include-qemu-compiler-add-QEMU_UNINITIALIZED-attribu.patch @@ -1,17 +1,17 @@ -From 73f85b945f09ae118f2c1479110f2e34906e084b Mon Sep 17 00:00:00 2001 +From cf92fd8487195ac45bfbdad15168eaec70f3aaa9 Mon Sep 17 00:00:00 2001 From: Stefan Hajnoczi Date: Tue, 10 Jun 2025 13:36:39 +0100 -Subject: [PATCH 02/31] include/qemu/compiler: add QEMU_UNINITIALIZED attribute +Subject: [PATCH 27/57] include/qemu/compiler: add QEMU_UNINITIALIZED attribute macro MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [1/30] 6b6151625fdf6636cbf352731906f643e2fbfd35 +RH-Commit: [1/30] 43c2412d318b6d8e0dcb0b37340640a9d90c3188 (stefanha/centos-stream-qemu-kvm) The QEMU_UNINITIALIZED macro is to be used to skip the default compiler variable initialization done by -ftrivial-auto-var-init=zero. diff --git a/SOURCES/kvm-io-Fix-partial-struct-copy-in-qio_dns_resolver_looku.patch b/SOURCES/kvm-io-Fix-partial-struct-copy-in-qio_dns_resolver_looku.patch new file mode 100644 index 0000000..23d927a --- /dev/null +++ b/SOURCES/kvm-io-Fix-partial-struct-copy-in-qio_dns_resolver_looku.patch @@ -0,0 +1,73 @@ +From 4545870823aea92b18a7e747b686b666d08006a4 Mon Sep 17 00:00:00 2001 +From: Juraj Marcin +Date: Wed, 21 May 2025 15:52:30 +0200 +Subject: [PATCH 08/57] io: Fix partial struct copy in + qio_dns_resolver_lookup_sync_inet() +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Juraj Marcin +RH-MergeRequest: 369: util/qemu-sockets: Introduce inet socket options controlling TCP keep-alive +RH-Jira: RHEL-67104 +RH-Acked-by: Peter Xu +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/7] 92c8b3e63c22a3ca6e5adc76cac1a9f812034912 (JurajMarcin/centos-src-qemu-kvm) + +Commit aec21d3175 (qapi: Add InetSocketAddress member keep-alive) +introduces the keep-alive flag, but this flag is not copied together +with other options in qio_dns_resolver_lookup_sync_inet(). + +This patch fixes this issue and also prevents future ones by copying the +entire structure first and only then overriding a few attributes that +need to be different. + +Fixes: aec21d31756c (qapi: Add InetSocketAddress member keep-alive) +Signed-off-by: Juraj Marcin +Reviewed-by: Daniel P. Berrangé +Signed-off-by: Daniel P. Berrangé + +(cherry picked from commit 0dc051aa85e1bd68d5c5110fa8af69204e6dbd3d) + +JIRA: https://issues.redhat.com/browse/RHEL-67104 + +Signed-off-by: Juraj Marcin +--- + io/dns-resolver.c | 21 +++++---------------- + 1 file changed, 5 insertions(+), 16 deletions(-) + +diff --git a/io/dns-resolver.c b/io/dns-resolver.c +index 53b0e8407a..3712438f82 100644 +--- a/io/dns-resolver.c ++++ b/io/dns-resolver.c +@@ -111,22 +111,11 @@ static int qio_dns_resolver_lookup_sync_inet(QIODNSResolver *resolver, + uaddr, INET6_ADDRSTRLEN, uport, 32, + NI_NUMERICHOST | NI_NUMERICSERV); + +- newaddr->u.inet = (InetSocketAddress){ +- .host = g_strdup(uaddr), +- .port = g_strdup(uport), +- .has_numeric = true, +- .numeric = true, +- .has_to = iaddr->has_to, +- .to = iaddr->to, +- .has_ipv4 = iaddr->has_ipv4, +- .ipv4 = iaddr->ipv4, +- .has_ipv6 = iaddr->has_ipv6, +- .ipv6 = iaddr->ipv6, +-#ifdef HAVE_IPPROTO_MPTCP +- .has_mptcp = iaddr->has_mptcp, +- .mptcp = iaddr->mptcp, +-#endif +- }; ++ newaddr->u.inet = *iaddr; ++ newaddr->u.inet.host = g_strdup(uaddr), ++ newaddr->u.inet.port = g_strdup(uport), ++ newaddr->u.inet.has_numeric = true, ++ newaddr->u.inet.numeric = true, + + (*addrs)[i] = newaddr; + } +-- +2.39.3 + diff --git a/SOURCES/kvm-io-fix-use-after-free-in-websocket-handshake-code.patch b/SOURCES/kvm-io-fix-use-after-free-in-websocket-handshake-code.patch new file mode 100644 index 0000000..dec3bd5 --- /dev/null +++ b/SOURCES/kvm-io-fix-use-after-free-in-websocket-handshake-code.patch @@ -0,0 +1,189 @@ +From 9f1ce751f7e800f00e4511c6ec874fe38deba7bf Mon Sep 17 00:00:00 2001 +From: Jon Maloy +Date: Tue, 4 Nov 2025 17:28:47 -0500 +Subject: [PATCH 2/2] io: fix use after free in websocket handshake code +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Jon Maloy +RH-MergeRequest: 497: io: fix use after free in websocket handshake code +RH-Jira: RHEL-120125 +RH-Acked-by: Daniel P. Berrangé +RH-Acked-by: Miroslav Rezanina +RH-Commit: [2/2] 68a23cb8e7a580a0d7de79994c71157f25c792ee (redhat/rhel/src/qemu-kvm/jons-qemu-kvm-2) + +JIRA: https://issues.redhat.com/browse/RHEL-120125 +CVE: CVE-2025-11234 + +commit b7a1f2ca45c7865b9e98e02ae605a65fc9458ae9 +Author: Daniel P. Berrangé +Date: Tue Sep 30 12:03:15 2025 +0100 + + io: fix use after free in websocket handshake code + + If the QIOChannelWebsock object is freed while it is waiting to + complete a handshake, a GSource is leaked. This can lead to the + callback firing later on and triggering a use-after-free in the + use of the channel. This was observed in the VNC server with the + following trace from valgrind: + + ==2523108== Invalid read of size 4 + ==2523108== at 0x4054A24: vnc_disconnect_start (vnc.c:1296) + ==2523108== by 0x4054A24: vnc_client_error (vnc.c:1392) + ==2523108== by 0x4068A09: vncws_handshake_done (vnc-ws.c:105) + ==2523108== by 0x44863B4: qio_task_complete (task.c:197) + ==2523108== by 0x448343D: qio_channel_websock_handshake_io (channel-websock.c:588) + ==2523108== by 0x6EDB862: UnknownInlinedFun (gmain.c:3398) + ==2523108== by 0x6EDB862: g_main_context_dispatch_unlocked.lto_priv.0 (gmain.c:4249) + ==2523108== by 0x6EDBAE4: g_main_context_dispatch (gmain.c:4237) + ==2523108== by 0x45EC79F: glib_pollfds_poll (main-loop.c:287) + ==2523108== by 0x45EC79F: os_host_main_loop_wait (main-loop.c:310) + ==2523108== by 0x45EC79F: main_loop_wait (main-loop.c:589) + ==2523108== by 0x423A56D: qemu_main_loop (runstate.c:835) + ==2523108== by 0x454F300: qemu_default_main (main.c:37) + ==2523108== by 0x73D6574: (below main) (libc_start_call_main.h:58) + ==2523108== Address 0x57a6e0dc is 28 bytes inside a block of size 103,608 free'd + ==2523108== at 0x5F2FE43: free (vg_replace_malloc.c:989) + ==2523108== by 0x6EDC444: g_free (gmem.c:208) + ==2523108== by 0x4053F23: vnc_update_client (vnc.c:1153) + ==2523108== by 0x4053F23: vnc_refresh (vnc.c:3225) + ==2523108== by 0x4042881: dpy_refresh (console.c:880) + ==2523108== by 0x4042881: gui_update (console.c:90) + ==2523108== by 0x45EFA1B: timerlist_run_timers.part.0 (qemu-timer.c:562) + ==2523108== by 0x45EFC8F: timerlist_run_timers (qemu-timer.c:495) + ==2523108== by 0x45EFC8F: qemu_clock_run_timers (qemu-timer.c:576) + ==2523108== by 0x45EFC8F: qemu_clock_run_all_timers (qemu-timer.c:663) + ==2523108== by 0x45EC765: main_loop_wait (main-loop.c:600) + ==2523108== by 0x423A56D: qemu_main_loop (runstate.c:835) + ==2523108== by 0x454F300: qemu_default_main (main.c:37) + ==2523108== by 0x73D6574: (below main) (libc_start_call_main.h:58) + ==2523108== Block was alloc'd at + ==2523108== at 0x5F343F3: calloc (vg_replace_malloc.c:1675) + ==2523108== by 0x6EE2F81: g_malloc0 (gmem.c:133) + ==2523108== by 0x4057DA3: vnc_connect (vnc.c:3245) + ==2523108== by 0x448591B: qio_net_listener_channel_func (net-listener.c:54) + ==2523108== by 0x6EDB862: UnknownInlinedFun (gmain.c:3398) + ==2523108== by 0x6EDB862: g_main_context_dispatch_unlocked.lto_priv.0 (gmain.c:4249) + ==2523108== by 0x6EDBAE4: g_main_context_dispatch (gmain.c:4237) + ==2523108== by 0x45EC79F: glib_pollfds_poll (main-loop.c:287) + ==2523108== by 0x45EC79F: os_host_main_loop_wait (main-loop.c:310) + ==2523108== by 0x45EC79F: main_loop_wait (main-loop.c:589) + ==2523108== by 0x423A56D: qemu_main_loop (runstate.c:835) + ==2523108== by 0x454F300: qemu_default_main (main.c:37) + ==2523108== by 0x73D6574: (below main) (libc_start_call_main.h:58) + ==2523108== + + The above can be reproduced by launching QEMU with + + $ qemu-system-x86_64 -vnc localhost:0,websocket=5700 + + and then repeatedly running: + + for i in {1..100}; do + (echo -n "GET / HTTP/1.1" && sleep 0.05) | nc -w 1 localhost 5700 & + done + + CVE-2025-11234 + Reported-by: Grant Millar | Cylo + Reviewed-by: Eric Blake + Signed-off-by: Daniel P. Berrangé + +Signed-off-by: Jon Maloy +--- + include/io/channel-websock.h | 3 ++- + io/channel-websock.c | 22 ++++++++++++++++------ + 2 files changed, 18 insertions(+), 7 deletions(-) + +diff --git a/include/io/channel-websock.h b/include/io/channel-websock.h +index e180827c57..6700cf8946 100644 +--- a/include/io/channel-websock.h ++++ b/include/io/channel-websock.h +@@ -61,7 +61,8 @@ struct QIOChannelWebsock { + size_t payload_remain; + size_t pong_remain; + QIOChannelWebsockMask mask; +- guint io_tag; ++ guint hs_io_tag; /* tracking handshake task */ ++ guint io_tag; /* tracking watch task */ + Error *io_err; + gboolean io_eof; + uint8_t opcode; +diff --git a/io/channel-websock.c b/io/channel-websock.c +index 1aac3c88a8..583ea86187 100644 +--- a/io/channel-websock.c ++++ b/io/channel-websock.c +@@ -545,6 +545,7 @@ static gboolean qio_channel_websock_handshake_send(QIOChannel *ioc, + trace_qio_channel_websock_handshake_fail(ioc, error_get_pretty(err)); + qio_task_set_error(task, err); + qio_task_complete(task); ++ wioc->hs_io_tag = 0; + return FALSE; + } + +@@ -560,6 +561,7 @@ static gboolean qio_channel_websock_handshake_send(QIOChannel *ioc, + trace_qio_channel_websock_handshake_complete(ioc); + qio_task_complete(task); + } ++ wioc->hs_io_tag = 0; + return FALSE; + } + trace_qio_channel_websock_handshake_pending(ioc, G_IO_OUT); +@@ -586,6 +588,7 @@ static gboolean qio_channel_websock_handshake_io(QIOChannel *ioc, + trace_qio_channel_websock_handshake_fail(ioc, error_get_pretty(err)); + qio_task_set_error(task, err); + qio_task_complete(task); ++ wioc->hs_io_tag = 0; + return FALSE; + } + if (ret == 0) { +@@ -597,7 +600,7 @@ static gboolean qio_channel_websock_handshake_io(QIOChannel *ioc, + error_propagate(&wioc->io_err, err); + + trace_qio_channel_websock_handshake_reply(ioc); +- qio_channel_add_watch( ++ wioc->hs_io_tag = qio_channel_add_watch( + wioc->master, + G_IO_OUT, + qio_channel_websock_handshake_send, +@@ -907,11 +910,12 @@ void qio_channel_websock_handshake(QIOChannelWebsock *ioc, + + trace_qio_channel_websock_handshake_start(ioc); + trace_qio_channel_websock_handshake_pending(ioc, G_IO_IN); +- qio_channel_add_watch(ioc->master, +- G_IO_IN, +- qio_channel_websock_handshake_io, +- task, +- NULL); ++ ioc->hs_io_tag = qio_channel_add_watch( ++ ioc->master, ++ G_IO_IN, ++ qio_channel_websock_handshake_io, ++ task, ++ NULL); + } + + +@@ -922,6 +926,9 @@ static void qio_channel_websock_finalize(Object *obj) + buffer_free(&ioc->encinput); + buffer_free(&ioc->encoutput); + buffer_free(&ioc->rawinput); ++ if (ioc->hs_io_tag) { ++ g_source_remove(ioc->hs_io_tag); ++ } + if (ioc->io_tag) { + g_source_remove(ioc->io_tag); + } +@@ -1222,6 +1229,9 @@ static int qio_channel_websock_close(QIOChannel *ioc, + buffer_free(&wioc->encinput); + buffer_free(&wioc->encoutput); + buffer_free(&wioc->rawinput); ++ if (wioc->hs_io_tag) { ++ g_clear_handle_id(&wioc->hs_io_tag, g_source_remove); ++ } + if (wioc->io_tag) { + g_clear_handle_id(&wioc->io_tag, g_source_remove); + } +-- +2.50.1 + diff --git a/SOURCES/kvm-io-move-websock-resource-release-to-close-method.patch b/SOURCES/kvm-io-move-websock-resource-release-to-close-method.patch new file mode 100644 index 0000000..ac68672 --- /dev/null +++ b/SOURCES/kvm-io-move-websock-resource-release-to-close-method.patch @@ -0,0 +1,84 @@ +From e4223fa468bf6887c3053a74a6e83bd0e99af70c Mon Sep 17 00:00:00 2001 +From: Jon Maloy +Date: Tue, 4 Nov 2025 17:23:29 -0500 +Subject: [PATCH 1/2] io: move websock resource release to close method +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Jon Maloy +RH-MergeRequest: 497: io: fix use after free in websocket handshake code +RH-Jira: RHEL-120125 +RH-Acked-by: Daniel P. Berrangé +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/2] 09d2c8d9150af559afb5750b97b1b6985e535580 (redhat/rhel/src/qemu-kvm/jons-qemu-kvm-2) + +JIRA: https://issues.redhat.com/browse/RHEL-120125 +CVE: CVE-2025-11234 + +commit 322c3c4f3abee616a18b3bfe563ec29dd67eae63 +Author: Daniel P. Berrangé +Date: Tue Sep 30 11:58:35 2025 +0100 + + io: move websock resource release to close method + + The QIOChannelWebsock object releases all its resources in the + finalize callback. This is later than desired, as callers expect + to be able to call qio_channel_close() to fully close a channel + and release resources related to I/O. + + The logic in the finalize method is at most a failsafe to handle + cases where a consumer forgets to call qio_channel_close. + + This adds equivalent logic to the close method to release the + resources, using g_clear_handle_id/g_clear_pointer to be robust + against repeated invocations. The finalize method is tweaked + so that the GSource is removed before releasing the underlying + channel. + + Reviewed-by: Eric Blake + Signed-off-by: Daniel P. Berrangé + +Signed-off-by: Jon Maloy +--- + io/channel-websock.c | 11 ++++++++++- + 1 file changed, 10 insertions(+), 1 deletion(-) + +diff --git a/io/channel-websock.c b/io/channel-websock.c +index de39f0d182..1aac3c88a8 100644 +--- a/io/channel-websock.c ++++ b/io/channel-websock.c +@@ -922,13 +922,13 @@ static void qio_channel_websock_finalize(Object *obj) + buffer_free(&ioc->encinput); + buffer_free(&ioc->encoutput); + buffer_free(&ioc->rawinput); +- object_unref(OBJECT(ioc->master)); + if (ioc->io_tag) { + g_source_remove(ioc->io_tag); + } + if (ioc->io_err) { + error_free(ioc->io_err); + } ++ object_unref(OBJECT(ioc->master)); + } + + +@@ -1219,6 +1219,15 @@ static int qio_channel_websock_close(QIOChannel *ioc, + QIOChannelWebsock *wioc = QIO_CHANNEL_WEBSOCK(ioc); + + trace_qio_channel_websock_close(ioc); ++ buffer_free(&wioc->encinput); ++ buffer_free(&wioc->encoutput); ++ buffer_free(&wioc->rawinput); ++ if (wioc->io_tag) { ++ g_clear_handle_id(&wioc->io_tag, g_source_remove); ++ } ++ if (wioc->io_err) { ++ g_clear_pointer(&wioc->io_err, error_free); ++ } + return qio_channel_close(wioc->master, errp); + } + +-- +2.50.1 + diff --git a/SOURCES/kvm-iotests-Improve-iotest-194-to-mirror-data.patch b/SOURCES/kvm-iotests-Improve-iotest-194-to-mirror-data.patch new file mode 100644 index 0000000..f696184 --- /dev/null +++ b/SOURCES/kvm-iotests-Improve-iotest-194-to-mirror-data.patch @@ -0,0 +1,42 @@ +From 8832268a98104ba3065a57dedcd3db43231512ba Mon Sep 17 00:00:00 2001 +From: Eric Blake +Date: Fri, 9 May 2025 15:40:22 -0500 +Subject: [PATCH 07/16] iotests: Improve iotest 194 to mirror data + +RH-Author: Eric Blake +RH-MergeRequest: 365: blockdev-mirror: More efficient handling of sparse mirrors +RH-Jira: RHEL-82906 RHEL-83015 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Jon Maloy +RH-Commit: [5/14] bfbe8eab1035480cef9d69d1974ba66b755b1b60 (ebblake/centos-qemu-kvm) + +Mirroring a completely sparse image to a sparse destination should be +practically instantaneous. It isn't yet, but the test will be more +realistic if it has some non-zero to mirror as well as the holes. + +Signed-off-by: Eric Blake +Reviewed-by: Stefan Hajnoczi +Message-ID: <20250509204341.3553601-20-eblake@redhat.com> +(cherry picked from commit eb89627899bb84148d272394e885725eff456ae9) +Jira: https://issues.redhat.com/browse/RHEL-82906 +Jira: https://issues.redhat.com/browse/RHEL-83015 +Signed-off-by: Eric Blake +--- + tests/qemu-iotests/194 | 1 + + 1 file changed, 1 insertion(+) + +diff --git a/tests/qemu-iotests/194 b/tests/qemu-iotests/194 +index c0ce82dd25..d0b9c084f5 100755 +--- a/tests/qemu-iotests/194 ++++ b/tests/qemu-iotests/194 +@@ -34,6 +34,7 @@ with iotests.FilePath('source.img') as source_img_path, \ + + img_size = '1G' + iotests.qemu_img_create('-f', iotests.imgfmt, source_img_path, img_size) ++ iotests.qemu_io('-f', iotests.imgfmt, '-c', 'write 512M 1M', source_img_path) + iotests.qemu_img_create('-f', iotests.imgfmt, dest_img_path, img_size) + + iotests.log('Launching VMs...') +-- +2.48.1 + diff --git a/SOURCES/kvm-iotests-common.rc-add-disk_usage-function.patch b/SOURCES/kvm-iotests-common.rc-add-disk_usage-function.patch new file mode 100644 index 0000000..14ffba7 --- /dev/null +++ b/SOURCES/kvm-iotests-common.rc-add-disk_usage-function.patch @@ -0,0 +1,68 @@ +From 644f39de9e2466a9570833b1070acf47a53863ea Mon Sep 17 00:00:00 2001 +From: Andrey Drobyshev +Date: Fri, 9 May 2025 15:40:29 -0500 +Subject: [PATCH 14/16] iotests/common.rc: add disk_usage function + +RH-Author: Eric Blake +RH-MergeRequest: 365: blockdev-mirror: More efficient handling of sparse mirrors +RH-Jira: RHEL-82906 RHEL-83015 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Jon Maloy +RH-Commit: [12/14] 0e5d4217f97fe6e952de23eedbc2b8d9c7600665 (ebblake/centos-qemu-kvm) + +Move the definition from iotests/250 to common.rc. This is used to +detect real disk usage of sparse files. In particular, we want to use +it for checking subclusters-based discards. + +Signed-off-by: Andrey Drobyshev +Reviewed-by: Alexander Ivanov +Reviewed-by: Alberto Garcia +Message-ID: <20240913163942.423050-6-andrey.drobyshev@virtuozzo.com> +Signed-off-by: Eric Blake +Reviewed-by: Stefan Hajnoczi +Message-ID: <20250509204341.3553601-27-eblake@redhat.com> +(cherry picked from commit be9bac072ede6e6aa27079f59efcf17b56bd7b26) +Jira: https://issues.redhat.com/browse/RHEL-82906 +Jira: https://issues.redhat.com/browse/RHEL-83015 +Signed-off-by: Eric Blake +--- + tests/qemu-iotests/250 | 5 ----- + tests/qemu-iotests/common.rc | 6 ++++++ + 2 files changed, 6 insertions(+), 5 deletions(-) + +diff --git a/tests/qemu-iotests/250 b/tests/qemu-iotests/250 +index af48f83aba..c0a0dbc0ff 100755 +--- a/tests/qemu-iotests/250 ++++ b/tests/qemu-iotests/250 +@@ -52,11 +52,6 @@ _unsupported_imgopts data_file + # bdrv_co_truncate(bs->file) call in qcow2_co_truncate(), which might succeed + # anyway. + +-disk_usage() +-{ +- du --block-size=1 $1 | awk '{print $1}' +-} +- + size=2100M + + _make_test_img -o "cluster_size=1M,preallocation=metadata" $size +diff --git a/tests/qemu-iotests/common.rc b/tests/qemu-iotests/common.rc +index 95c12577dd..237f746af8 100644 +--- a/tests/qemu-iotests/common.rc ++++ b/tests/qemu-iotests/common.rc +@@ -140,6 +140,12 @@ _optstr_add() + fi + } + ++# report real disk usage for sparse files ++disk_usage() ++{ ++ du --block-size=1 "$1" | awk '{print $1}' ++} ++ + # Set the variables to the empty string to turn Valgrind off + # for specific processes, e.g. + # $ VALGRIND_QEMU_IO= ./check -qcow2 -valgrind 015 +-- +2.48.1 + diff --git a/SOURCES/kvm-kvm-Check-KVM_CAP_MAX_VCPUS-at-vm-level.patch b/SOURCES/kvm-kvm-Check-KVM_CAP_MAX_VCPUS-at-vm-level.patch new file mode 100644 index 0000000..6926857 --- /dev/null +++ b/SOURCES/kvm-kvm-Check-KVM_CAP_MAX_VCPUS-at-vm-level.patch @@ -0,0 +1,45 @@ +From f88758420a256bb2b277d21cb3f3b60426caf51d Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:47 +0200 +Subject: [PATCH 059/115] kvm: Check KVM_CAP_MAX_VCPUS at vm level + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [59/115] aac4b448f0e64cdca79d2a8f3c53f683b9514a8a (bonzini/rhel-qemu-kvm) + +KVM with TDX support starts to report different KVM_CAP_MAX_VCPUS per +different VM types. So switch to check the KVM_CAP_MAX_VCPUS at vm level. + +KVM still returns the global KVM_CAP_MAX_VCPUS when the KVM is old that +doesn't report different value at vm level. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-31-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 77b5403a0298a5460554f768a2098fd21588e555) +Signed-off-by: Paolo Bonzini +--- + accel/kvm/kvm-all.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/accel/kvm/kvm-all.c b/accel/kvm/kvm-all.c +index 65952d7f4e..c1605bc4fa 100644 +--- a/accel/kvm/kvm-all.c ++++ b/accel/kvm/kvm-all.c +@@ -2419,7 +2419,7 @@ static int kvm_recommended_vcpus(KVMState *s) + + static int kvm_max_vcpus(KVMState *s) + { +- int ret = kvm_check_extension(s, KVM_CAP_MAX_VCPUS); ++ int ret = kvm_vm_check_extension(s, KVM_CAP_MAX_VCPUS); + return (ret) ? ret : kvm_recommended_vcpus(s); + } + +-- +2.50.1 + diff --git a/SOURCES/kvm-kvm-Introduce-kvm_arch_pre_create_vcpu.patch b/SOURCES/kvm-kvm-Introduce-kvm_arch_pre_create_vcpu.patch new file mode 100644 index 0000000..a957f4c --- /dev/null +++ b/SOURCES/kvm-kvm-Introduce-kvm_arch_pre_create_vcpu.patch @@ -0,0 +1,187 @@ +From cb4a0164407562ac307b9ccc0031455cf1065bdb Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:45 +0200 +Subject: [PATCH 036/115] kvm: Introduce kvm_arch_pre_create_vcpu() +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [36/115] 1c3f6170a844694bc1d4b5d48860330d5961d26d (bonzini/rhel-qemu-kvm) + +Introduce kvm_arch_pre_create_vcpu(), to perform arch-dependent +work prior to create any vcpu. This is for i386 TDX because it needs +call TDX_INIT_VM before creating any vcpu. + +The specific implementation for i386 will be added in the future patch. + +Signed-off-by: Xiaoyao Li +Acked-by: Gerd Hoffmann +Reviewed-by: Daniel P. Berrangé +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250508150002.689633-8-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit a668268dc08f7f4d30cecd513054bb38ce48c0d6) +Signed-off-by: Paolo Bonzini + +Conflicts: different context in loongarch +--- + accel/kvm/kvm-all.c | 5 +++++ + include/sysemu/kvm.h | 1 + + target/arm/kvm.c | 5 +++++ + target/i386/kvm/kvm.c | 5 +++++ + target/loongarch/kvm/kvm.c | 5 +++++ + target/mips/kvm.c | 5 +++++ + target/ppc/kvm.c | 5 +++++ + target/riscv/kvm/kvm-cpu.c | 5 +++++ + target/s390x/kvm/kvm.c | 5 +++++ + 9 files changed, 41 insertions(+) + +diff --git a/accel/kvm/kvm-all.c b/accel/kvm/kvm-all.c +index 49dedda47e..65952d7f4e 100644 +--- a/accel/kvm/kvm-all.c ++++ b/accel/kvm/kvm-all.c +@@ -530,6 +530,11 @@ int kvm_init_vcpu(CPUState *cpu, Error **errp) + + trace_kvm_init_vcpu(cpu->cpu_index, kvm_arch_vcpu_id(cpu)); + ++ ret = kvm_arch_pre_create_vcpu(cpu, errp); ++ if (ret < 0) { ++ goto err; ++ } ++ + ret = kvm_create_vcpu(cpu); + if (ret < 0) { + error_setg_errno(errp, -ret, +diff --git a/include/sysemu/kvm.h b/include/sysemu/kvm.h +index d9ad723f78..0e6b29b331 100644 +--- a/include/sysemu/kvm.h ++++ b/include/sysemu/kvm.h +@@ -374,6 +374,7 @@ int kvm_arch_get_default_type(MachineState *ms); + + int kvm_arch_init(MachineState *ms, KVMState *s); + ++int kvm_arch_pre_create_vcpu(CPUState *cpu, Error **errp); + int kvm_arch_init_vcpu(CPUState *cpu); + int kvm_arch_destroy_vcpu(CPUState *cpu); + +diff --git a/target/arm/kvm.c b/target/arm/kvm.c +index f1f1b5b375..e0469e7831 100644 +--- a/target/arm/kvm.c ++++ b/target/arm/kvm.c +@@ -1860,6 +1860,11 @@ static int kvm_arm_sve_set_vls(ARMCPU *cpu) + + #define ARM_CPU_ID_MPIDR 3, 0, 0, 0, 5 + ++int kvm_arch_pre_create_vcpu(CPUState *cpu, Error **errp) ++{ ++ return 0; ++} ++ + int kvm_arch_init_vcpu(CPUState *cs) + { + int ret; +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index bf4493dd84..1fddec6b9c 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -2039,6 +2039,11 @@ full: + abort(); + } + ++int kvm_arch_pre_create_vcpu(CPUState *cpu, Error **errp) ++{ ++ return 0; ++} ++ + int kvm_arch_init_vcpu(CPUState *cs) + { + struct { +diff --git a/target/loongarch/kvm/kvm.c b/target/loongarch/kvm/kvm.c +index 9204d4295d..f6e20dbc61 100644 +--- a/target/loongarch/kvm/kvm.c ++++ b/target/loongarch/kvm/kvm.c +@@ -663,6 +663,11 @@ static void kvm_loongarch_vm_stage_change(void *opaque, bool running, + } + } + ++int kvm_arch_pre_create_vcpu(CPUState *cpu, Error **errp) ++{ ++ return 0; ++} ++ + int kvm_arch_init_vcpu(CPUState *cs) + { + uint64_t val; +diff --git a/target/mips/kvm.c b/target/mips/kvm.c +index a98798c669..75058f8938 100644 +--- a/target/mips/kvm.c ++++ b/target/mips/kvm.c +@@ -61,6 +61,11 @@ int kvm_arch_irqchip_create(KVMState *s) + return 0; + } + ++int kvm_arch_pre_create_vcpu(CPUState *cpu, Error **errp) ++{ ++ return 0; ++} ++ + int kvm_arch_init_vcpu(CPUState *cs) + { + CPUMIPSState *env = cpu_env(cs); +diff --git a/target/ppc/kvm.c b/target/ppc/kvm.c +index 7daa097164..7a86d09802 100644 +--- a/target/ppc/kvm.c ++++ b/target/ppc/kvm.c +@@ -479,6 +479,11 @@ static void kvmppc_hw_debug_points_init(CPUPPCState *cenv) + } + } + ++int kvm_arch_pre_create_vcpu(CPUState *cpu, Error **errp) ++{ ++ return 0; ++} ++ + int kvm_arch_init_vcpu(CPUState *cs) + { + PowerPCCPU *cpu = POWERPC_CPU(cs); +diff --git a/target/riscv/kvm/kvm-cpu.c b/target/riscv/kvm/kvm-cpu.c +index 2bfb112be0..18ab470549 100644 +--- a/target/riscv/kvm/kvm-cpu.c ++++ b/target/riscv/kvm/kvm-cpu.c +@@ -1355,6 +1355,11 @@ static int kvm_vcpu_enable_sbi_dbcn(RISCVCPU *cpu, CPUState *cs) + return kvm_set_one_reg(cs, kvm_sbi_dbcn.kvm_reg_id, ®); + } + ++int kvm_arch_pre_create_vcpu(CPUState *cpu, Error **errp) ++{ ++ return 0; ++} ++ + int kvm_arch_init_vcpu(CPUState *cs) + { + int ret = 0; +diff --git a/target/s390x/kvm/kvm.c b/target/s390x/kvm/kvm.c +index afc8d570c9..3273beff18 100644 +--- a/target/s390x/kvm/kvm.c ++++ b/target/s390x/kvm/kvm.c +@@ -408,6 +408,11 @@ unsigned long kvm_arch_vcpu_id(CPUState *cpu) + return cpu->cpu_index; + } + ++int kvm_arch_pre_create_vcpu(CPUState *cpu, Error **errp) ++{ ++ return 0; ++} ++ + int kvm_arch_init_vcpu(CPUState *cs) + { + unsigned int max_cpus = MACHINE(qdev_get_machine())->smp.max_cpus; +-- +2.50.1 + diff --git a/SOURCES/kvm-kvm-i386-make-kvm_filter_msr-and-related-definitions.patch b/SOURCES/kvm-kvm-i386-make-kvm_filter_msr-and-related-definitions.patch new file mode 100644 index 0000000..c65a3f5 --- /dev/null +++ b/SOURCES/kvm-kvm-i386-make-kvm_filter_msr-and-related-definitions.patch @@ -0,0 +1,90 @@ +From e1658c60f353f93b195fb6f569cb44a5fb1f3e99 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:43 +0200 +Subject: [PATCH 003/115] kvm/i386: make kvm_filter_msr() and related + definitions private to kvm module +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [3/115] ef00896ed2cac6bc1ef39f21aad422ae5b571cb6 (bonzini/rhel-qemu-kvm) + +kvm_filer_msr() is only used from i386 kvm module. Make it static so that its +easy for developers to understand that its not used anywhere else. +Same for QEMURDMSRHandler, QEMUWRMSRHandler and KVMMSRHandlers definitions. + +CC: philmd@linaro.org +Reviewed-by: Philippe Mathieu-Daudé +Signed-off-by: Ani Sinha +Link: https://lore.kernel.org/r/20240903140045.41167-1-anisinha@redhat.com +[Make struct unnamed. - Paolo] +Signed-off-by: Paolo Bonzini +(cherry picked from commit ed2880f4e93bf83106ebdc8562a5ee4d93285a3b) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/kvm.c | 12 +++++++++++- + target/i386/kvm/kvm_i386.h | 11 ----------- + 2 files changed, 11 insertions(+), 12 deletions(-) + +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index 94b678e9e3..b02aec915c 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -92,7 +92,17 @@ + * 255 kvm_msr_entry structs */ + #define MSR_BUF_SIZE 4096 + ++typedef bool QEMURDMSRHandler(X86CPU *cpu, uint32_t msr, uint64_t *val); ++typedef bool QEMUWRMSRHandler(X86CPU *cpu, uint32_t msr, uint64_t val); ++typedef struct { ++ uint32_t msr; ++ QEMURDMSRHandler *rdmsr; ++ QEMUWRMSRHandler *wrmsr; ++} KVMMSRHandlers; ++ + static void kvm_init_msrs(X86CPU *cpu); ++static bool kvm_filter_msr(KVMState *s, uint32_t msr, QEMURDMSRHandler *rdmsr, ++ QEMUWRMSRHandler *wrmsr); + + const KVMCapabilityInfo kvm_arch_required_capabilities[] = { + KVM_CAP_INFO(SET_TSS_ADDR), +@@ -5762,7 +5772,7 @@ static bool kvm_install_msr_filters(KVMState *s) + return true; + } + +-bool kvm_filter_msr(KVMState *s, uint32_t msr, QEMURDMSRHandler *rdmsr, ++static bool kvm_filter_msr(KVMState *s, uint32_t msr, QEMURDMSRHandler *rdmsr, + QEMUWRMSRHandler *wrmsr) + { + int i; +diff --git a/target/i386/kvm/kvm_i386.h b/target/i386/kvm/kvm_i386.h +index 34fc60774b..9de9c0d303 100644 +--- a/target/i386/kvm/kvm_i386.h ++++ b/target/i386/kvm/kvm_i386.h +@@ -66,17 +66,6 @@ uint64_t kvm_swizzle_msi_ext_dest_id(uint64_t address); + void kvm_update_msi_routes_all(void *private, bool global, + uint32_t index, uint32_t mask); + +-typedef bool QEMURDMSRHandler(X86CPU *cpu, uint32_t msr, uint64_t *val); +-typedef bool QEMUWRMSRHandler(X86CPU *cpu, uint32_t msr, uint64_t val); +-typedef struct kvm_msr_handlers { +- uint32_t msr; +- QEMURDMSRHandler *rdmsr; +- QEMUWRMSRHandler *wrmsr; +-} KVMMSRHandlers; +- +-bool kvm_filter_msr(KVMState *s, uint32_t msr, QEMURDMSRHandler *rdmsr, +- QEMUWRMSRHandler *wrmsr); +- + #endif /* CONFIG_KVM */ + + void kvm_pc_setup_irq_routing(bool pci_enabled); +-- +2.50.1 + diff --git a/SOURCES/kvm-kvm-remove-unnecessary-ifdef.patch b/SOURCES/kvm-kvm-remove-unnecessary-ifdef.patch new file mode 100644 index 0000000..a474ebe --- /dev/null +++ b/SOURCES/kvm-kvm-remove-unnecessary-ifdef.patch @@ -0,0 +1,52 @@ +From be25ddc3910de08325880177c16d4f0c38feeb73 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:43 +0200 +Subject: [PATCH 004/115] kvm: remove unnecessary #ifdef + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [4/115] c7a7b57246c2c345961d5b5923f8b086c58b82c6 (bonzini/rhel-qemu-kvm) + +Signed-off-by: Paolo Bonzini +(cherry picked from commit feccfa77bed975e2e60c90756f08b8a56df6daf6) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/kvm_i386.h | 11 +---------- + 1 file changed, 1 insertion(+), 10 deletions(-) + +diff --git a/target/i386/kvm/kvm_i386.h b/target/i386/kvm/kvm_i386.h +index 9de9c0d303..7edb154a16 100644 +--- a/target/i386/kvm/kvm_i386.h ++++ b/target/i386/kvm/kvm_i386.h +@@ -13,8 +13,7 @@ + + #include "sysemu/kvm.h" + +-#ifdef CONFIG_KVM +- ++/* always false if !CONFIG_KVM */ + #define kvm_pit_in_kernel() \ + (kvm_irqchip_in_kernel() && !kvm_irqchip_is_split()) + #define kvm_pic_in_kernel() \ +@@ -22,14 +21,6 @@ + #define kvm_ioapic_in_kernel() \ + (kvm_irqchip_in_kernel() && !kvm_irqchip_is_split()) + +-#else +- +-#define kvm_pit_in_kernel() 0 +-#define kvm_pic_in_kernel() 0 +-#define kvm_ioapic_in_kernel() 0 +- +-#endif /* CONFIG_KVM */ +- + bool kvm_has_smm(void); + bool kvm_enable_x2apic(void); + bool kvm_hv_vpindex_settable(void); +-- +2.50.1 + diff --git a/SOURCES/kvm-linux-headers-Update-to-Linux-v6.14-rc3.patch b/SOURCES/kvm-linux-headers-Update-to-Linux-v6.14-rc3.patch new file mode 100644 index 0000000..ccd11d1 --- /dev/null +++ b/SOURCES/kvm-linux-headers-Update-to-Linux-v6.14-rc3.patch @@ -0,0 +1,495 @@ +From d90604bca41cef828b4e37d77b53e472b272dbca Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:45 +0200 +Subject: [PATCH 026/115] linux-headers: Update to Linux v6.14-rc3 + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [26/115] 6674b075f7bd2c7fd4cbd4ccb1689fed0fc3b026 (bonzini/rhel-qemu-kvm) + +Update headers to retrieve the latest KVM caps for RISC-V. + +Signed-off-by: Daniel Henrique Barboza +Message-ID: <20250221153758.652078-2-dbarboza@ventanamicro.com> +Signed-off-by: Alistair Francis +(cherry picked from commit 421ee1ec6f0de0b0fd96b262bda18b97e54263b4) +Signed-off-by: Paolo Bonzini +--- + include/standard-headers/linux/ethtool.h | 4 + + include/standard-headers/linux/fuse.h | 76 ++++++++++++++++++- + .../linux/input-event-codes.h | 1 + + include/standard-headers/linux/pci_regs.h | 16 ++-- + include/standard-headers/linux/virtio_pci.h | 14 ++++ + linux-headers/asm-arm64/kvm.h | 3 - + linux-headers/asm-loongarch/kvm_para.h | 1 + + linux-headers/asm-riscv/kvm.h | 7 +- + linux-headers/asm-x86/kvm.h | 1 + + linux-headers/linux/iommufd.h | 35 ++++++--- + linux-headers/linux/kvm.h | 8 +- + linux-headers/linux/stddef.h | 13 +++- + linux-headers/linux/vduse.h | 2 +- + 13 files changed, 146 insertions(+), 35 deletions(-) + +diff --git a/include/standard-headers/linux/ethtool.h b/include/standard-headers/linux/ethtool.h +index 67c47912e5..e83382531c 100644 +--- a/include/standard-headers/linux/ethtool.h ++++ b/include/standard-headers/linux/ethtool.h +@@ -681,6 +681,8 @@ enum ethtool_link_ext_substate_module { + * @ETH_SS_STATS_ETH_MAC: names of IEEE 802.3 MAC statistics + * @ETH_SS_STATS_ETH_CTRL: names of IEEE 802.3 MAC Control statistics + * @ETH_SS_STATS_RMON: names of RMON statistics ++ * @ETH_SS_STATS_PHY: names of PHY(dev) statistics ++ * @ETH_SS_TS_FLAGS: hardware timestamping flags + * + * @ETH_SS_COUNT: number of defined string sets + */ +@@ -706,6 +708,8 @@ enum ethtool_stringset { + ETH_SS_STATS_ETH_MAC, + ETH_SS_STATS_ETH_CTRL, + ETH_SS_STATS_RMON, ++ ETH_SS_STATS_PHY, ++ ETH_SS_TS_FLAGS, + + /* add new constants above here */ + ETH_SS_COUNT +diff --git a/include/standard-headers/linux/fuse.h b/include/standard-headers/linux/fuse.h +index 889e12ad15..d303effb2a 100644 +--- a/include/standard-headers/linux/fuse.h ++++ b/include/standard-headers/linux/fuse.h +@@ -220,6 +220,15 @@ + * + * 7.41 + * - add FUSE_ALLOW_IDMAP ++ * 7.42 ++ * - Add FUSE_OVER_IO_URING and all other io-uring related flags and data ++ * structures: ++ * - struct fuse_uring_ent_in_out ++ * - struct fuse_uring_req_header ++ * - struct fuse_uring_cmd_req ++ * - FUSE_URING_IN_OUT_HEADER_SZ ++ * - FUSE_URING_OP_IN_OUT_SZ ++ * - enum fuse_uring_cmd + */ + + #ifndef _LINUX_FUSE_H +@@ -251,7 +260,7 @@ + #define FUSE_KERNEL_VERSION 7 + + /** Minor version number of this interface */ +-#define FUSE_KERNEL_MINOR_VERSION 41 ++#define FUSE_KERNEL_MINOR_VERSION 42 + + /** The node ID of the root inode */ + #define FUSE_ROOT_ID 1 +@@ -421,6 +430,7 @@ struct fuse_file_lock { + * FUSE_HAS_RESEND: kernel supports resending pending requests, and the high bit + * of the request ID indicates resend requests + * FUSE_ALLOW_IDMAP: allow creation of idmapped mounts ++ * FUSE_OVER_IO_URING: Indicate that client supports io-uring + */ + #define FUSE_ASYNC_READ (1 << 0) + #define FUSE_POSIX_LOCKS (1 << 1) +@@ -467,6 +477,7 @@ struct fuse_file_lock { + /* Obsolete alias for FUSE_DIRECT_IO_ALLOW_MMAP */ + #define FUSE_DIRECT_IO_RELAX FUSE_DIRECT_IO_ALLOW_MMAP + #define FUSE_ALLOW_IDMAP (1ULL << 40) ++#define FUSE_OVER_IO_URING (1ULL << 41) + + /** + * CUSE INIT request/reply flags +@@ -1202,4 +1213,67 @@ struct fuse_supp_groups { + uint32_t groups[]; + }; + ++/** ++ * Size of the ring buffer header ++ */ ++#define FUSE_URING_IN_OUT_HEADER_SZ 128 ++#define FUSE_URING_OP_IN_OUT_SZ 128 ++ ++/* Used as part of the fuse_uring_req_header */ ++struct fuse_uring_ent_in_out { ++ uint64_t flags; ++ ++ /* ++ * commit ID to be used in a reply to a ring request (see also ++ * struct fuse_uring_cmd_req) ++ */ ++ uint64_t commit_id; ++ ++ /* size of user payload buffer */ ++ uint32_t payload_sz; ++ uint32_t padding; ++ ++ uint64_t reserved; ++}; ++ ++/** ++ * Header for all fuse-io-uring requests ++ */ ++struct fuse_uring_req_header { ++ /* struct fuse_in_header / struct fuse_out_header */ ++ char in_out[FUSE_URING_IN_OUT_HEADER_SZ]; ++ ++ /* per op code header */ ++ char op_in[FUSE_URING_OP_IN_OUT_SZ]; ++ ++ struct fuse_uring_ent_in_out ring_ent_in_out; ++}; ++ ++/** ++ * sqe commands to the kernel ++ */ ++enum fuse_uring_cmd { ++ FUSE_IO_URING_CMD_INVALID = 0, ++ ++ /* register the request buffer and fetch a fuse request */ ++ FUSE_IO_URING_CMD_REGISTER = 1, ++ ++ /* commit fuse request result and fetch next request */ ++ FUSE_IO_URING_CMD_COMMIT_AND_FETCH = 2, ++}; ++ ++/** ++ * In the 80B command area of the SQE. ++ */ ++struct fuse_uring_cmd_req { ++ uint64_t flags; ++ ++ /* entry identifier for commits */ ++ uint64_t commit_id; ++ ++ /* queue the command is for (queue index) */ ++ uint16_t qid; ++ uint8_t padding[6]; ++}; ++ + #endif /* _LINUX_FUSE_H */ +diff --git a/include/standard-headers/linux/input-event-codes.h b/include/standard-headers/linux/input-event-codes.h +index 50b2b7497e..09ba0ad878 100644 +--- a/include/standard-headers/linux/input-event-codes.h ++++ b/include/standard-headers/linux/input-event-codes.h +@@ -519,6 +519,7 @@ + #define KEY_NOTIFICATION_CENTER 0x1bc /* Show/hide the notification center */ + #define KEY_PICKUP_PHONE 0x1bd /* Answer incoming call */ + #define KEY_HANGUP_PHONE 0x1be /* Decline incoming call */ ++#define KEY_LINK_PHONE 0x1bf /* AL Phone Syncing */ + + #define KEY_DEL_EOL 0x1c0 + #define KEY_DEL_EOS 0x1c1 +diff --git a/include/standard-headers/linux/pci_regs.h b/include/standard-headers/linux/pci_regs.h +index 1601c7ed5f..3445c4970e 100644 +--- a/include/standard-headers/linux/pci_regs.h ++++ b/include/standard-headers/linux/pci_regs.h +@@ -533,7 +533,7 @@ + #define PCI_EXP_DEVSTA_TRPND 0x0020 /* Transactions Pending */ + #define PCI_CAP_EXP_RC_ENDPOINT_SIZEOF_V1 12 /* v1 endpoints without link end here */ + #define PCI_EXP_LNKCAP 0x0c /* Link Capabilities */ +-#define PCI_EXP_LNKCAP_SLS 0x0000000f /* Supported Link Speeds */ ++#define PCI_EXP_LNKCAP_SLS 0x0000000f /* Max Link Speed (prior to PCIe r3.0: Supported Link Speeds) */ + #define PCI_EXP_LNKCAP_SLS_2_5GB 0x00000001 /* LNKCAP2 SLS Vector bit 0 */ + #define PCI_EXP_LNKCAP_SLS_5_0GB 0x00000002 /* LNKCAP2 SLS Vector bit 1 */ + #define PCI_EXP_LNKCAP_SLS_8_0GB 0x00000003 /* LNKCAP2 SLS Vector bit 2 */ +@@ -665,6 +665,7 @@ + #define PCI_EXP_DEVCAP2_OBFF_MSG 0x00040000 /* New message signaling */ + #define PCI_EXP_DEVCAP2_OBFF_WAKE 0x00080000 /* Re-use WAKE# for OBFF */ + #define PCI_EXP_DEVCAP2_EE_PREFIX 0x00200000 /* End-End TLP Prefix */ ++#define PCI_EXP_DEVCAP2_EE_PREFIX_MAX 0x00c00000 /* Max End-End TLP Prefixes */ + #define PCI_EXP_DEVCTL2 0x28 /* Device Control 2 */ + #define PCI_EXP_DEVCTL2_COMP_TIMEOUT 0x000f /* Completion Timeout Value */ + #define PCI_EXP_DEVCTL2_COMP_TMOUT_DIS 0x0010 /* Completion Timeout Disable */ +@@ -789,10 +790,11 @@ + /* Same bits as above */ + #define PCI_ERR_CAP 0x18 /* Advanced Error Capabilities & Ctrl*/ + #define PCI_ERR_CAP_FEP(x) ((x) & 0x1f) /* First Error Pointer */ +-#define PCI_ERR_CAP_ECRC_GENC 0x00000020 /* ECRC Generation Capable */ +-#define PCI_ERR_CAP_ECRC_GENE 0x00000040 /* ECRC Generation Enable */ +-#define PCI_ERR_CAP_ECRC_CHKC 0x00000080 /* ECRC Check Capable */ +-#define PCI_ERR_CAP_ECRC_CHKE 0x00000100 /* ECRC Check Enable */ ++#define PCI_ERR_CAP_ECRC_GENC 0x00000020 /* ECRC Generation Capable */ ++#define PCI_ERR_CAP_ECRC_GENE 0x00000040 /* ECRC Generation Enable */ ++#define PCI_ERR_CAP_ECRC_CHKC 0x00000080 /* ECRC Check Capable */ ++#define PCI_ERR_CAP_ECRC_CHKE 0x00000100 /* ECRC Check Enable */ ++#define PCI_ERR_CAP_PREFIX_LOG_PRESENT 0x00000800 /* TLP Prefix Log Present */ + #define PCI_ERR_HEADER_LOG 0x1c /* Header Log Register (16 bytes) */ + #define PCI_ERR_ROOT_COMMAND 0x2c /* Root Error Command */ + #define PCI_ERR_ROOT_CMD_COR_EN 0x00000001 /* Correctable Err Reporting Enable */ +@@ -808,6 +810,7 @@ + #define PCI_ERR_ROOT_FATAL_RCV 0x00000040 /* Fatal Received */ + #define PCI_ERR_ROOT_AER_IRQ 0xf8000000 /* Advanced Error Interrupt Message Number */ + #define PCI_ERR_ROOT_ERR_SRC 0x34 /* Error Source Identification */ ++#define PCI_ERR_PREFIX_LOG 0x38 /* TLP Prefix LOG Register (up to 16 bytes) */ + + /* Virtual Channel */ + #define PCI_VC_PORT_CAP1 0x04 +@@ -1001,9 +1004,6 @@ + #define PCI_ACS_CTRL 0x06 /* ACS Control Register */ + #define PCI_ACS_EGRESS_CTL_V 0x08 /* ACS Egress Control Vector */ + +-#define PCI_VSEC_HDR 4 /* extended cap - vendor-specific */ +-#define PCI_VSEC_HDR_LEN_SHIFT 20 /* shift for length field */ +- + /* SATA capability */ + #define PCI_SATA_REGS 4 /* SATA REGs specifier */ + #define PCI_SATA_REGS_MASK 0xF /* location - BAR#/inline */ +diff --git a/include/standard-headers/linux/virtio_pci.h b/include/standard-headers/linux/virtio_pci.h +index b177ed8972..91fec6f502 100644 +--- a/include/standard-headers/linux/virtio_pci.h ++++ b/include/standard-headers/linux/virtio_pci.h +@@ -116,6 +116,8 @@ + #define VIRTIO_PCI_CAP_PCI_CFG 5 + /* Additional shared memory capability */ + #define VIRTIO_PCI_CAP_SHARED_MEMORY_CFG 8 ++/* PCI vendor data configuration */ ++#define VIRTIO_PCI_CAP_VENDOR_CFG 9 + + /* This is the PCI capability header: */ + struct virtio_pci_cap { +@@ -130,6 +132,18 @@ struct virtio_pci_cap { + uint32_t length; /* Length of the structure, in bytes. */ + }; + ++/* This is the PCI vendor data capability header: */ ++struct virtio_pci_vndr_data { ++ uint8_t cap_vndr; /* Generic PCI field: PCI_CAP_ID_VNDR */ ++ uint8_t cap_next; /* Generic PCI field: next ptr. */ ++ uint8_t cap_len; /* Generic PCI field: capability length */ ++ uint8_t cfg_type; /* Identifies the structure. */ ++ uint16_t vendor_id; /* Identifies the vendor-specific format. */ ++ /* For Vendor Definition */ ++ /* Pads structure to a multiple of 4 bytes */ ++ /* Reads must not have side effects */ ++}; ++ + struct virtio_pci_cap64 { + struct virtio_pci_cap cap; + uint32_t offset_hi; /* Most sig 32 bits of offset */ +diff --git a/linux-headers/asm-arm64/kvm.h b/linux-headers/asm-arm64/kvm.h +index dccd5d965f..ec1e82bdc8 100644 +--- a/linux-headers/asm-arm64/kvm.h ++++ b/linux-headers/asm-arm64/kvm.h +@@ -43,9 +43,6 @@ + #define KVM_COALESCED_MMIO_PAGE_OFFSET 1 + #define KVM_DIRTY_LOG_PAGE_OFFSET 64 + +-#define KVM_REG_SIZE(id) \ +- (1U << (((id) & KVM_REG_SIZE_MASK) >> KVM_REG_SIZE_SHIFT)) +- + struct kvm_regs { + struct user_pt_regs regs; /* sp = sp_el0 */ + +diff --git a/linux-headers/asm-loongarch/kvm_para.h b/linux-headers/asm-loongarch/kvm_para.h +index 4ba4ad8db1..fd7f40713d 100644 +--- a/linux-headers/asm-loongarch/kvm_para.h ++++ b/linux-headers/asm-loongarch/kvm_para.h +@@ -17,5 +17,6 @@ + #define KVM_FEATURE_STEAL_TIME 2 + /* BIT 24 - 31 are features configurable by user space vmm */ + #define KVM_FEATURE_VIRT_EXTIOI 24 ++#define KVM_FEATURE_USER_HCALL 25 + + #endif /* _ASM_KVM_PARA_H */ +diff --git a/linux-headers/asm-riscv/kvm.h b/linux-headers/asm-riscv/kvm.h +index 3482c9a73d..f06bc5efcd 100644 +--- a/linux-headers/asm-riscv/kvm.h ++++ b/linux-headers/asm-riscv/kvm.h +@@ -179,6 +179,9 @@ enum KVM_RISCV_ISA_EXT_ID { + KVM_RISCV_ISA_EXT_SSNPM, + KVM_RISCV_ISA_EXT_SVADE, + KVM_RISCV_ISA_EXT_SVADU, ++ KVM_RISCV_ISA_EXT_SVVPTC, ++ KVM_RISCV_ISA_EXT_ZABHA, ++ KVM_RISCV_ISA_EXT_ZICCRSE, + KVM_RISCV_ISA_EXT_MAX, + }; + +@@ -198,6 +201,7 @@ enum KVM_RISCV_SBI_EXT_ID { + KVM_RISCV_SBI_EXT_VENDOR, + KVM_RISCV_SBI_EXT_DBCN, + KVM_RISCV_SBI_EXT_STA, ++ KVM_RISCV_SBI_EXT_SUSP, + KVM_RISCV_SBI_EXT_MAX, + }; + +@@ -211,9 +215,6 @@ struct kvm_riscv_sbi_sta { + #define KVM_RISCV_TIMER_STATE_OFF 0 + #define KVM_RISCV_TIMER_STATE_ON 1 + +-#define KVM_REG_SIZE(id) \ +- (1U << (((id) & KVM_REG_SIZE_MASK) >> KVM_REG_SIZE_SHIFT)) +- + /* If you need to interpret the index values, here is the key: */ + #define KVM_REG_RISCV_TYPE_MASK 0x00000000FF000000 + #define KVM_REG_RISCV_TYPE_SHIFT 24 +diff --git a/linux-headers/asm-x86/kvm.h b/linux-headers/asm-x86/kvm.h +index 96589490c4..86f2c34e7a 100644 +--- a/linux-headers/asm-x86/kvm.h ++++ b/linux-headers/asm-x86/kvm.h +@@ -923,5 +923,6 @@ struct kvm_hyperv_eventfd { + #define KVM_X86_SEV_VM 2 + #define KVM_X86_SEV_ES_VM 3 + #define KVM_X86_SNP_VM 4 ++#define KVM_X86_TDX_VM 5 + + #endif /* _ASM_X86_KVM_H */ +diff --git a/linux-headers/linux/iommufd.h b/linux-headers/linux/iommufd.h +index 37aae16502..ccbdca5e11 100644 +--- a/linux-headers/linux/iommufd.h ++++ b/linux-headers/linux/iommufd.h +@@ -297,7 +297,7 @@ struct iommu_ioas_unmap { + * ioctl(IOMMU_OPTION_HUGE_PAGES) + * @IOMMU_OPTION_RLIMIT_MODE: + * Change how RLIMIT_MEMLOCK accounting works. The caller must have privilege +- * to invoke this. Value 0 (default) is user based accouting, 1 uses process ++ * to invoke this. Value 0 (default) is user based accounting, 1 uses process + * based accounting. Global option, object_id must be 0 + * @IOMMU_OPTION_HUGE_PAGES: + * Value 1 (default) allows contiguous pages to be combined when generating +@@ -390,7 +390,7 @@ struct iommu_vfio_ioas { + * @IOMMU_HWPT_ALLOC_PASID: Requests a domain that can be used with PASID. The + * domain can be attached to any PASID on the device. + * Any domain attached to the non-PASID part of the +- * device must also be flaged, otherwise attaching a ++ * device must also be flagged, otherwise attaching a + * PASID will blocked. + * If IOMMU does not support PASID it will return + * error (-EOPNOTSUPP). +@@ -558,16 +558,25 @@ struct iommu_hw_info_vtd { + * For the details of @idr, @iidr and @aidr, please refer to the chapters + * from 6.3.1 to 6.3.6 in the SMMUv3 Spec. + * +- * User space should read the underlying ARM SMMUv3 hardware information for +- * the list of supported features. ++ * This reports the raw HW capability, and not all bits are meaningful to be ++ * read by userspace. Only the following fields should be used: + * +- * Note that these values reflect the raw HW capability, without any insight if +- * any required kernel driver support is present. Bits may be set indicating the +- * HW has functionality that is lacking kernel software support, such as BTM. If +- * a VMM is using this information to construct emulated copies of these +- * registers it should only forward bits that it knows it can support. ++ * idr[0]: ST_LEVEL, TERM_MODEL, STALL_MODEL, TTENDIAN , CD2L, ASID16, TTF ++ * idr[1]: SIDSIZE, SSIDSIZE ++ * idr[3]: BBML, RIL ++ * idr[5]: VAX, GRAN64K, GRAN16K, GRAN4K + * +- * In future, presence of required kernel support will be indicated in flags. ++ * - S1P should be assumed to be true if a NESTED HWPT can be created ++ * - VFIO/iommufd only support platforms with COHACC, it should be assumed to be ++ * true. ++ * - ATS is a per-device property. If the VMM describes any devices as ATS ++ * capable in ACPI/DT it should set the corresponding idr. ++ * ++ * This list may expand in future (eg E0PD, AIE, PBHA, D128, DS etc). It is ++ * important that VMMs do not read bits outside the list to allow for ++ * compatibility with future kernels. Several features in the SMMUv3 ++ * architecture are not currently supported by the kernel for nesting: HTTU, ++ * BTM, MPAM and others. + */ + struct iommu_hw_info_arm_smmuv3 { + __u32 flags; +@@ -766,7 +775,7 @@ struct iommu_hwpt_vtd_s1_invalidate { + }; + + /** +- * struct iommu_viommu_arm_smmuv3_invalidate - ARM SMMUv3 cahce invalidation ++ * struct iommu_viommu_arm_smmuv3_invalidate - ARM SMMUv3 cache invalidation + * (IOMMU_VIOMMU_INVALIDATE_DATA_ARM_SMMUV3) + * @cmd: 128-bit cache invalidation command that runs in SMMU CMDQ. + * Must be little-endian. +@@ -859,6 +868,7 @@ enum iommu_hwpt_pgfault_perm { + * @pasid: Process Address Space ID + * @grpid: Page Request Group Index + * @perm: Combination of enum iommu_hwpt_pgfault_perm ++ * @__reserved: Must be 0. + * @addr: Fault address + * @length: a hint of how much data the requestor is expecting to fetch. For + * example, if the PRI initiator knows it is going to do a 10MB +@@ -874,7 +884,8 @@ struct iommu_hwpt_pgfault { + __u32 pasid; + __u32 grpid; + __u32 perm; +- __u64 addr; ++ __u32 __reserved; ++ __aligned_u64 addr; + __u32 length; + __u32 cookie; + }; +diff --git a/linux-headers/linux/kvm.h b/linux-headers/linux/kvm.h +index 3bcd4eabe3..27181b3dd8 100644 +--- a/linux-headers/linux/kvm.h ++++ b/linux-headers/linux/kvm.h +@@ -609,10 +609,6 @@ struct kvm_ioeventfd { + #define KVM_X86_DISABLE_EXITS_HLT (1 << 1) + #define KVM_X86_DISABLE_EXITS_PAUSE (1 << 2) + #define KVM_X86_DISABLE_EXITS_CSTATE (1 << 3) +-#define KVM_X86_DISABLE_VALID_EXITS (KVM_X86_DISABLE_EXITS_MWAIT | \ +- KVM_X86_DISABLE_EXITS_HLT | \ +- KVM_X86_DISABLE_EXITS_PAUSE | \ +- KVM_X86_DISABLE_EXITS_CSTATE) + + /* for KVM_ENABLE_CAP */ + struct kvm_enable_cap { +@@ -1062,6 +1058,10 @@ struct kvm_dirty_tlb { + + #define KVM_REG_SIZE_SHIFT 52 + #define KVM_REG_SIZE_MASK 0x00f0000000000000ULL ++ ++#define KVM_REG_SIZE(id) \ ++ (1U << (((id) & KVM_REG_SIZE_MASK) >> KVM_REG_SIZE_SHIFT)) ++ + #define KVM_REG_SIZE_U8 0x0000000000000000ULL + #define KVM_REG_SIZE_U16 0x0010000000000000ULL + #define KVM_REG_SIZE_U32 0x0020000000000000ULL +diff --git a/linux-headers/linux/stddef.h b/linux-headers/linux/stddef.h +index 96aa341942..e1416f7937 100644 +--- a/linux-headers/linux/stddef.h ++++ b/linux-headers/linux/stddef.h +@@ -8,6 +8,13 @@ + #define __always_inline __inline__ + #endif + ++/* Not all C++ standards support type declarations inside an anonymous union */ ++#ifndef __cplusplus ++#define __struct_group_tag(TAG) TAG ++#else ++#define __struct_group_tag(TAG) ++#endif ++ + /** + * __struct_group() - Create a mirrored named and anonyomous struct + * +@@ -20,13 +27,13 @@ + * and size: one anonymous and one named. The former's members can be used + * normally without sub-struct naming, and the latter can be used to + * reason about the start, end, and size of the group of struct members. +- * The named struct can also be explicitly tagged for layer reuse, as well +- * as both having struct attributes appended. ++ * The named struct can also be explicitly tagged for layer reuse (C only), ++ * as well as both having struct attributes appended. + */ + #define __struct_group(TAG, NAME, ATTRS, MEMBERS...) \ + union { \ + struct { MEMBERS } ATTRS; \ +- struct TAG { MEMBERS } ATTRS NAME; \ ++ struct __struct_group_tag(TAG) { MEMBERS } ATTRS NAME; \ + } ATTRS + + #ifdef __cplusplus +diff --git a/linux-headers/linux/vduse.h b/linux-headers/linux/vduse.h +index 6d2ca064b5..f46269af34 100644 +--- a/linux-headers/linux/vduse.h ++++ b/linux-headers/linux/vduse.h +@@ -1,4 +1,4 @@ +-/* SPDX-License-Identifier: GPL-2.0 WITH Linux-syscall-note */ ++/* SPDX-License-Identifier: ((GPL-2.0 WITH Linux-syscall-note) OR BSD-3-Clause) */ + #ifndef _VDUSE_H_ + #define _VDUSE_H_ + +-- +2.50.1 + diff --git a/SOURCES/kvm-linux-headers-Update-to-Linux-v6.15-rc3.patch b/SOURCES/kvm-linux-headers-Update-to-Linux-v6.15-rc3.patch new file mode 100644 index 0000000..dbfde0e --- /dev/null +++ b/SOURCES/kvm-linux-headers-Update-to-Linux-v6.15-rc3.patch @@ -0,0 +1,949 @@ +From a02b11f5744f20a44c1dc3291ffa20985e3302d3 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:45 +0200 +Subject: [PATCH 027/115] linux-headers: Update to Linux v6.15-rc3 +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [27/115] 019c5b8dd28df2c5a0822468c3c485569983b2c1 (bonzini/rhel-qemu-kvm) + +Update headers to retrieve uapi information for vfio-ap + +Signed-off-by: Rorie Reyes +Reviewed-by: Cédric Le Goater +Link: https://lore.kernel.org/qemu-devel/20250425052401.8287-3-rreyes@linux.ibm.com +Signed-off-by: Cédric Le Goater +(cherry picked from commit 1cab5a02ab8144aad2abd001835e49104e4aae0f) +Signed-off-by: Paolo Bonzini +--- + include/standard-headers/asm-x86/setup_data.h | 4 +- + include/standard-headers/drm/drm_fourcc.h | 41 ++++++ + include/standard-headers/linux/const.h | 2 +- + include/standard-headers/linux/ethtool.h | 22 +++ + include/standard-headers/linux/fuse.h | 12 +- + include/standard-headers/linux/pci_regs.h | 13 +- + include/standard-headers/linux/virtio_net.h | 13 ++ + include/standard-headers/linux/virtio_snd.h | 2 +- + linux-headers/asm-arm64/kvm.h | 11 ++ + linux-headers/asm-arm64/unistd_64.h | 1 + + linux-headers/asm-generic/mman-common.h | 1 + + linux-headers/asm-generic/unistd.h | 4 +- + linux-headers/asm-loongarch/unistd_64.h | 1 + + linux-headers/asm-mips/unistd_n32.h | 1 + + linux-headers/asm-mips/unistd_n64.h | 1 + + linux-headers/asm-mips/unistd_o32.h | 1 + + linux-headers/asm-powerpc/unistd_32.h | 1 + + linux-headers/asm-powerpc/unistd_64.h | 1 + + linux-headers/asm-riscv/kvm.h | 2 + + linux-headers/asm-riscv/unistd_32.h | 1 + + linux-headers/asm-riscv/unistd_64.h | 1 + + linux-headers/asm-s390/unistd_32.h | 1 + + linux-headers/asm-s390/unistd_64.h | 1 + + linux-headers/asm-x86/kvm.h | 3 + + linux-headers/asm-x86/unistd_32.h | 1 + + linux-headers/asm-x86/unistd_64.h | 1 + + linux-headers/asm-x86/unistd_x32.h | 1 + + linux-headers/linux/bits.h | 8 +- + linux-headers/linux/const.h | 2 +- + linux-headers/linux/iommufd.h | 129 +++++++++++++++++- + linux-headers/linux/kvm.h | 1 + + linux-headers/linux/psp-sev.h | 21 ++- + linux-headers/linux/stddef.h | 2 + + linux-headers/linux/vfio.h | 30 ++-- + 34 files changed, 301 insertions(+), 36 deletions(-) + +diff --git a/include/standard-headers/asm-x86/setup_data.h b/include/standard-headers/asm-x86/setup_data.h +index 09355f54c5..a483d72f42 100644 +--- a/include/standard-headers/asm-x86/setup_data.h ++++ b/include/standard-headers/asm-x86/setup_data.h +@@ -18,7 +18,7 @@ + #define SETUP_INDIRECT (1<<31) + #define SETUP_TYPE_MAX (SETUP_ENUM_MAX | SETUP_INDIRECT) + +-#ifndef __ASSEMBLY__ ++#ifndef __ASSEMBLER__ + + #include "standard-headers/linux/types.h" + +@@ -78,6 +78,6 @@ struct ima_setup_data { + uint64_t size; + } QEMU_PACKED; + +-#endif /* __ASSEMBLY__ */ ++#endif /* __ASSEMBLER__ */ + + #endif /* _ASM_X86_SETUP_DATA_H */ +diff --git a/include/standard-headers/drm/drm_fourcc.h b/include/standard-headers/drm/drm_fourcc.h +index 708647776f..a8b759dcbc 100644 +--- a/include/standard-headers/drm/drm_fourcc.h ++++ b/include/standard-headers/drm/drm_fourcc.h +@@ -420,6 +420,7 @@ extern "C" { + #define DRM_FORMAT_MOD_VENDOR_ARM 0x08 + #define DRM_FORMAT_MOD_VENDOR_ALLWINNER 0x09 + #define DRM_FORMAT_MOD_VENDOR_AMLOGIC 0x0a ++#define DRM_FORMAT_MOD_VENDOR_MTK 0x0b + + /* add more to the end as needed */ + +@@ -1452,6 +1453,46 @@ drm_fourcc_canonicalize_nvidia_format_mod(uint64_t modifier) + */ + #define AMLOGIC_FBC_OPTION_MEM_SAVING (1ULL << 0) + ++/* MediaTek modifiers ++ * Bits Parameter Notes ++ * ----- ------------------------ --------------------------------------------- ++ * 7: 0 TILE LAYOUT Values are MTK_FMT_MOD_TILE_* ++ * 15: 8 COMPRESSION Values are MTK_FMT_MOD_COMPRESS_* ++ * 23:16 10 BIT LAYOUT Values are MTK_FMT_MOD_10BIT_LAYOUT_* ++ * ++ */ ++ ++#define DRM_FORMAT_MOD_MTK(__flags) fourcc_mod_code(MTK, __flags) ++ ++/* ++ * MediaTek Tiled Modifier ++ * The lowest 8 bits of the modifier is used to specify the tiling ++ * layout. Only the 16L_32S tiling is used for now, but we define an ++ * "untiled" version and leave room for future expansion. ++ */ ++#define MTK_FMT_MOD_TILE_MASK 0xf ++#define MTK_FMT_MOD_TILE_NONE 0x0 ++#define MTK_FMT_MOD_TILE_16L32S 0x1 ++ ++/* ++ * Bits 8-15 specify compression options ++ */ ++#define MTK_FMT_MOD_COMPRESS_MASK (0xf << 8) ++#define MTK_FMT_MOD_COMPRESS_NONE (0x0 << 8) ++#define MTK_FMT_MOD_COMPRESS_V1 (0x1 << 8) ++ ++/* ++ * Bits 16-23 specify how the bits of 10 bit formats are ++ * stored out in memory ++ */ ++#define MTK_FMT_MOD_10BIT_LAYOUT_MASK (0xf << 16) ++#define MTK_FMT_MOD_10BIT_LAYOUT_PACKED (0x0 << 16) ++#define MTK_FMT_MOD_10BIT_LAYOUT_LSBTILED (0x1 << 16) ++#define MTK_FMT_MOD_10BIT_LAYOUT_LSBRASTER (0x2 << 16) ++ ++/* alias for the most common tiling format */ ++#define DRM_FORMAT_MOD_MTK_16L_32S_TILE DRM_FORMAT_MOD_MTK(MTK_FMT_MOD_TILE_16L32S) ++ + /* + * AMD modifiers + * +diff --git a/include/standard-headers/linux/const.h b/include/standard-headers/linux/const.h +index 2122610de7..95ede23342 100644 +--- a/include/standard-headers/linux/const.h ++++ b/include/standard-headers/linux/const.h +@@ -33,7 +33,7 @@ + * Missing __asm__ support + * + * __BIT128() would not work in the __asm__ code, as it shifts an +- * 'unsigned __init128' data type as direct representation of ++ * 'unsigned __int128' data type as direct representation of + * 128 bit constants is not supported in the gcc compiler, as + * they get silently truncated. + * +diff --git a/include/standard-headers/linux/ethtool.h b/include/standard-headers/linux/ethtool.h +index e83382531c..5d1ad5fdea 100644 +--- a/include/standard-headers/linux/ethtool.h ++++ b/include/standard-headers/linux/ethtool.h +@@ -2059,6 +2059,24 @@ enum ethtool_link_mode_bit_indices { + ETHTOOL_LINK_MODE_10baseT1S_Half_BIT = 100, + ETHTOOL_LINK_MODE_10baseT1S_P2MP_Half_BIT = 101, + ETHTOOL_LINK_MODE_10baseT1BRR_Full_BIT = 102, ++ ETHTOOL_LINK_MODE_200000baseCR_Full_BIT = 103, ++ ETHTOOL_LINK_MODE_200000baseKR_Full_BIT = 104, ++ ETHTOOL_LINK_MODE_200000baseDR_Full_BIT = 105, ++ ETHTOOL_LINK_MODE_200000baseDR_2_Full_BIT = 106, ++ ETHTOOL_LINK_MODE_200000baseSR_Full_BIT = 107, ++ ETHTOOL_LINK_MODE_200000baseVR_Full_BIT = 108, ++ ETHTOOL_LINK_MODE_400000baseCR2_Full_BIT = 109, ++ ETHTOOL_LINK_MODE_400000baseKR2_Full_BIT = 110, ++ ETHTOOL_LINK_MODE_400000baseDR2_Full_BIT = 111, ++ ETHTOOL_LINK_MODE_400000baseDR2_2_Full_BIT = 112, ++ ETHTOOL_LINK_MODE_400000baseSR2_Full_BIT = 113, ++ ETHTOOL_LINK_MODE_400000baseVR2_Full_BIT = 114, ++ ETHTOOL_LINK_MODE_800000baseCR4_Full_BIT = 115, ++ ETHTOOL_LINK_MODE_800000baseKR4_Full_BIT = 116, ++ ETHTOOL_LINK_MODE_800000baseDR4_Full_BIT = 117, ++ ETHTOOL_LINK_MODE_800000baseDR4_2_Full_BIT = 118, ++ ETHTOOL_LINK_MODE_800000baseSR4_Full_BIT = 119, ++ ETHTOOL_LINK_MODE_800000baseVR4_Full_BIT = 120, + + /* must be last entry */ + __ETHTOOL_LINK_MODE_MASK_NBITS +@@ -2271,6 +2289,10 @@ static inline int ethtool_validate_duplex(uint8_t duplex) + * be exploited to reduce the RSS queue spread. + */ + #define RXH_XFRM_SYM_XOR (1 << 0) ++/* Similar to SYM_XOR, except that one copy of the XOR'ed fields is replaced by ++ * an OR of the same fields ++ */ ++#define RXH_XFRM_SYM_OR_XOR (1 << 1) + #define RXH_XFRM_NO_CHANGE 0xff + + /* L2-L4 network traffic flow types */ +diff --git a/include/standard-headers/linux/fuse.h b/include/standard-headers/linux/fuse.h +index d303effb2a..a2b5815d89 100644 +--- a/include/standard-headers/linux/fuse.h ++++ b/include/standard-headers/linux/fuse.h +@@ -229,6 +229,9 @@ + * - FUSE_URING_IN_OUT_HEADER_SZ + * - FUSE_URING_OP_IN_OUT_SZ + * - enum fuse_uring_cmd ++ * ++ * 7.43 ++ * - add FUSE_REQUEST_TIMEOUT + */ + + #ifndef _LINUX_FUSE_H +@@ -260,7 +263,7 @@ + #define FUSE_KERNEL_VERSION 7 + + /** Minor version number of this interface */ +-#define FUSE_KERNEL_MINOR_VERSION 42 ++#define FUSE_KERNEL_MINOR_VERSION 43 + + /** The node ID of the root inode */ + #define FUSE_ROOT_ID 1 +@@ -431,6 +434,8 @@ struct fuse_file_lock { + * of the request ID indicates resend requests + * FUSE_ALLOW_IDMAP: allow creation of idmapped mounts + * FUSE_OVER_IO_URING: Indicate that client supports io-uring ++ * FUSE_REQUEST_TIMEOUT: kernel supports timing out requests. ++ * init_out.request_timeout contains the timeout (in secs) + */ + #define FUSE_ASYNC_READ (1 << 0) + #define FUSE_POSIX_LOCKS (1 << 1) +@@ -473,11 +478,11 @@ struct fuse_file_lock { + #define FUSE_PASSTHROUGH (1ULL << 37) + #define FUSE_NO_EXPORT_SUPPORT (1ULL << 38) + #define FUSE_HAS_RESEND (1ULL << 39) +- + /* Obsolete alias for FUSE_DIRECT_IO_ALLOW_MMAP */ + #define FUSE_DIRECT_IO_RELAX FUSE_DIRECT_IO_ALLOW_MMAP + #define FUSE_ALLOW_IDMAP (1ULL << 40) + #define FUSE_OVER_IO_URING (1ULL << 41) ++#define FUSE_REQUEST_TIMEOUT (1ULL << 42) + + /** + * CUSE INIT request/reply flags +@@ -905,7 +910,8 @@ struct fuse_init_out { + uint16_t map_alignment; + uint32_t flags2; + uint32_t max_stack_depth; +- uint32_t unused[6]; ++ uint16_t request_timeout; ++ uint16_t unused[11]; + }; + + #define CUSE_INIT_INFO_MAX 4096 +diff --git a/include/standard-headers/linux/pci_regs.h b/include/standard-headers/linux/pci_regs.h +index 3445c4970e..ba326710f9 100644 +--- a/include/standard-headers/linux/pci_regs.h ++++ b/include/standard-headers/linux/pci_regs.h +@@ -486,6 +486,7 @@ + #define PCI_EXP_TYPE_RC_EC 0xa /* Root Complex Event Collector */ + #define PCI_EXP_FLAGS_SLOT 0x0100 /* Slot implemented */ + #define PCI_EXP_FLAGS_IRQ 0x3e00 /* Interrupt message number */ ++#define PCI_EXP_FLAGS_FLIT 0x8000 /* Flit Mode Supported */ + #define PCI_EXP_DEVCAP 0x04 /* Device capabilities */ + #define PCI_EXP_DEVCAP_PAYLOAD 0x00000007 /* Max_Payload_Size */ + #define PCI_EXP_DEVCAP_PHANTOM 0x00000018 /* Phantom functions */ +@@ -795,6 +796,8 @@ + #define PCI_ERR_CAP_ECRC_CHKC 0x00000080 /* ECRC Check Capable */ + #define PCI_ERR_CAP_ECRC_CHKE 0x00000100 /* ECRC Check Enable */ + #define PCI_ERR_CAP_PREFIX_LOG_PRESENT 0x00000800 /* TLP Prefix Log Present */ ++#define PCI_ERR_CAP_TLP_LOG_FLIT 0x00040000 /* TLP was logged in Flit Mode */ ++#define PCI_ERR_CAP_TLP_LOG_SIZE 0x00f80000 /* Logged TLP Size (only in Flit mode) */ + #define PCI_ERR_HEADER_LOG 0x1c /* Header Log Register (16 bytes) */ + #define PCI_ERR_ROOT_COMMAND 0x2c /* Root Error Command */ + #define PCI_ERR_ROOT_CMD_COR_EN 0x00000001 /* Correctable Err Reporting Enable */ +@@ -1013,7 +1016,7 @@ + + /* Resizable BARs */ + #define PCI_REBAR_CAP 4 /* capability register */ +-#define PCI_REBAR_CAP_SIZES 0x00FFFFF0 /* supported BAR sizes */ ++#define PCI_REBAR_CAP_SIZES 0xFFFFFFF0 /* supported BAR sizes */ + #define PCI_REBAR_CTRL 8 /* control register */ + #define PCI_REBAR_CTRL_BAR_IDX 0x00000007 /* BAR index */ + #define PCI_REBAR_CTRL_NBAR_MASK 0x000000E0 /* # of resizable BARs */ +@@ -1061,8 +1064,9 @@ + #define PCI_EXP_DPC_CAP_RP_EXT 0x0020 /* Root Port Extensions */ + #define PCI_EXP_DPC_CAP_POISONED_TLP 0x0040 /* Poisoned TLP Egress Blocking Supported */ + #define PCI_EXP_DPC_CAP_SW_TRIGGER 0x0080 /* Software Triggering Supported */ +-#define PCI_EXP_DPC_RP_PIO_LOG_SIZE 0x0F00 /* RP PIO Log Size */ ++#define PCI_EXP_DPC_RP_PIO_LOG_SIZE 0x0F00 /* RP PIO Log Size [3:0] */ + #define PCI_EXP_DPC_CAP_DL_ACTIVE 0x1000 /* ERR_COR signal on DL_Active supported */ ++#define PCI_EXP_DPC_RP_PIO_LOG_SIZE4 0x2000 /* RP PIO Log Size [4] */ + + #define PCI_EXP_DPC_CTL 0x06 /* DPC control */ + #define PCI_EXP_DPC_CTL_EN_FATAL 0x0001 /* Enable trigger on ERR_FATAL message */ +@@ -1205,9 +1209,12 @@ + #define PCI_DOE_DATA_OBJECT_DISC_REQ_3_INDEX 0x000000ff + #define PCI_DOE_DATA_OBJECT_DISC_REQ_3_VER 0x0000ff00 + #define PCI_DOE_DATA_OBJECT_DISC_RSP_3_VID 0x0000ffff +-#define PCI_DOE_DATA_OBJECT_DISC_RSP_3_PROTOCOL 0x00ff0000 ++#define PCI_DOE_DATA_OBJECT_DISC_RSP_3_TYPE 0x00ff0000 + #define PCI_DOE_DATA_OBJECT_DISC_RSP_3_NEXT_INDEX 0xff000000 + ++/* Deprecated old name, replaced with PCI_DOE_DATA_OBJECT_DISC_RSP_3_TYPE */ ++#define PCI_DOE_DATA_OBJECT_DISC_RSP_3_PROTOCOL PCI_DOE_DATA_OBJECT_DISC_RSP_3_TYPE ++ + /* Compute Express Link (CXL r3.1, sec 8.1.5) */ + #define PCI_DVSEC_CXL_PORT 3 + #define PCI_DVSEC_CXL_PORT_CTL 0x0c +diff --git a/include/standard-headers/linux/virtio_net.h b/include/standard-headers/linux/virtio_net.h +index fc594fe5fc..982e854f14 100644 +--- a/include/standard-headers/linux/virtio_net.h ++++ b/include/standard-headers/linux/virtio_net.h +@@ -327,6 +327,19 @@ struct virtio_net_rss_config { + uint8_t hash_key_data[/* hash_key_length */]; + }; + ++struct virtio_net_rss_config_hdr { ++ uint32_t hash_types; ++ uint16_t indirection_table_mask; ++ uint16_t unclassified_queue; ++ uint16_t indirection_table[/* 1 + indirection_table_mask */]; ++}; ++ ++struct virtio_net_rss_config_trailer { ++ uint16_t max_tx_vq; ++ uint8_t hash_key_length; ++ uint8_t hash_key_data[/* hash_key_length */]; ++}; ++ + #define VIRTIO_NET_CTRL_MQ_RSS_CONFIG 1 + + /* +diff --git a/include/standard-headers/linux/virtio_snd.h b/include/standard-headers/linux/virtio_snd.h +index 860f12e0a4..160d57899f 100644 +--- a/include/standard-headers/linux/virtio_snd.h ++++ b/include/standard-headers/linux/virtio_snd.h +@@ -25,7 +25,7 @@ struct virtio_snd_config { + uint32_t streams; + /* # of available channel maps */ + uint32_t chmaps; +- /* # of available control elements */ ++ /* # of available control elements (if VIRTIO_SND_F_CTLS) */ + uint32_t controls; + }; + +diff --git a/linux-headers/asm-arm64/kvm.h b/linux-headers/asm-arm64/kvm.h +index ec1e82bdc8..4e6aff08df 100644 +--- a/linux-headers/asm-arm64/kvm.h ++++ b/linux-headers/asm-arm64/kvm.h +@@ -105,6 +105,7 @@ struct kvm_regs { + #define KVM_ARM_VCPU_PTRAUTH_ADDRESS 5 /* VCPU uses address authentication */ + #define KVM_ARM_VCPU_PTRAUTH_GENERIC 6 /* VCPU uses generic authentication */ + #define KVM_ARM_VCPU_HAS_EL2 7 /* Support nested virtualization */ ++#define KVM_ARM_VCPU_HAS_EL2_E2H0 8 /* Limit NV support to E2H RES0 */ + + struct kvm_vcpu_init { + __u32 target; +@@ -365,6 +366,7 @@ enum { + KVM_REG_ARM_STD_HYP_BIT_PV_TIME = 0, + }; + ++/* Vendor hyper call function numbers 0-63 */ + #define KVM_REG_ARM_VENDOR_HYP_BMAP KVM_REG_ARM_FW_FEAT_BMAP_REG(2) + + enum { +@@ -372,6 +374,14 @@ enum { + KVM_REG_ARM_VENDOR_HYP_BIT_PTP = 1, + }; + ++/* Vendor hyper call function numbers 64-127 */ ++#define KVM_REG_ARM_VENDOR_HYP_BMAP_2 KVM_REG_ARM_FW_FEAT_BMAP_REG(3) ++ ++enum { ++ KVM_REG_ARM_VENDOR_HYP_BIT_DISCOVER_IMPL_VER = 0, ++ KVM_REG_ARM_VENDOR_HYP_BIT_DISCOVER_IMPL_CPUS = 1, ++}; ++ + /* Device Control API on vm fd */ + #define KVM_ARM_VM_SMCCC_CTRL 0 + #define KVM_ARM_VM_SMCCC_FILTER 0 +@@ -394,6 +404,7 @@ enum { + #define KVM_DEV_ARM_VGIC_GRP_CPU_SYSREGS 6 + #define KVM_DEV_ARM_VGIC_GRP_LEVEL_INFO 7 + #define KVM_DEV_ARM_VGIC_GRP_ITS_REGS 8 ++#define KVM_DEV_ARM_VGIC_GRP_MAINT_IRQ 9 + #define KVM_DEV_ARM_VGIC_LINE_LEVEL_INFO_SHIFT 10 + #define KVM_DEV_ARM_VGIC_LINE_LEVEL_INFO_MASK \ + (0x3fffffULL << KVM_DEV_ARM_VGIC_LINE_LEVEL_INFO_SHIFT) +diff --git a/linux-headers/asm-arm64/unistd_64.h b/linux-headers/asm-arm64/unistd_64.h +index d4e90fff76..ee9aaebdf3 100644 +--- a/linux-headers/asm-arm64/unistd_64.h ++++ b/linux-headers/asm-arm64/unistd_64.h +@@ -323,6 +323,7 @@ + #define __NR_getxattrat 464 + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 ++#define __NR_open_tree_attr 467 + + + #endif /* _ASM_UNISTD_64_H */ +diff --git a/linux-headers/asm-generic/mman-common.h b/linux-headers/asm-generic/mman-common.h +index 1ea2c4c33b..ef1c27fa3c 100644 +--- a/linux-headers/asm-generic/mman-common.h ++++ b/linux-headers/asm-generic/mman-common.h +@@ -85,6 +85,7 @@ + /* compatibility flags */ + #define MAP_FILE 0 + ++#define PKEY_UNRESTRICTED 0x0 + #define PKEY_DISABLE_ACCESS 0x1 + #define PKEY_DISABLE_WRITE 0x2 + #define PKEY_ACCESS_MASK (PKEY_DISABLE_ACCESS |\ +diff --git a/linux-headers/asm-generic/unistd.h b/linux-headers/asm-generic/unistd.h +index 88dc393c2b..2892a45023 100644 +--- a/linux-headers/asm-generic/unistd.h ++++ b/linux-headers/asm-generic/unistd.h +@@ -849,9 +849,11 @@ __SYSCALL(__NR_getxattrat, sys_getxattrat) + __SYSCALL(__NR_listxattrat, sys_listxattrat) + #define __NR_removexattrat 466 + __SYSCALL(__NR_removexattrat, sys_removexattrat) ++#define __NR_open_tree_attr 467 ++__SYSCALL(__NR_open_tree_attr, sys_open_tree_attr) + + #undef __NR_syscalls +-#define __NR_syscalls 467 ++#define __NR_syscalls 468 + + /* + * 32 bit systems traditionally used different +diff --git a/linux-headers/asm-loongarch/unistd_64.h b/linux-headers/asm-loongarch/unistd_64.h +index 23fb96a8a7..50d22df8f7 100644 +--- a/linux-headers/asm-loongarch/unistd_64.h ++++ b/linux-headers/asm-loongarch/unistd_64.h +@@ -319,6 +319,7 @@ + #define __NR_getxattrat 464 + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 ++#define __NR_open_tree_attr 467 + + + #endif /* _ASM_UNISTD_64_H */ +diff --git a/linux-headers/asm-mips/unistd_n32.h b/linux-headers/asm-mips/unistd_n32.h +index 9a75719644..bdcc2f460b 100644 +--- a/linux-headers/asm-mips/unistd_n32.h ++++ b/linux-headers/asm-mips/unistd_n32.h +@@ -395,5 +395,6 @@ + #define __NR_getxattrat (__NR_Linux + 464) + #define __NR_listxattrat (__NR_Linux + 465) + #define __NR_removexattrat (__NR_Linux + 466) ++#define __NR_open_tree_attr (__NR_Linux + 467) + + #endif /* _ASM_UNISTD_N32_H */ +diff --git a/linux-headers/asm-mips/unistd_n64.h b/linux-headers/asm-mips/unistd_n64.h +index 7086783b0c..3b6b0193b6 100644 +--- a/linux-headers/asm-mips/unistd_n64.h ++++ b/linux-headers/asm-mips/unistd_n64.h +@@ -371,5 +371,6 @@ + #define __NR_getxattrat (__NR_Linux + 464) + #define __NR_listxattrat (__NR_Linux + 465) + #define __NR_removexattrat (__NR_Linux + 466) ++#define __NR_open_tree_attr (__NR_Linux + 467) + + #endif /* _ASM_UNISTD_N64_H */ +diff --git a/linux-headers/asm-mips/unistd_o32.h b/linux-headers/asm-mips/unistd_o32.h +index b3825823e4..4609a4b4d3 100644 +--- a/linux-headers/asm-mips/unistd_o32.h ++++ b/linux-headers/asm-mips/unistd_o32.h +@@ -441,5 +441,6 @@ + #define __NR_getxattrat (__NR_Linux + 464) + #define __NR_listxattrat (__NR_Linux + 465) + #define __NR_removexattrat (__NR_Linux + 466) ++#define __NR_open_tree_attr (__NR_Linux + 467) + + #endif /* _ASM_UNISTD_O32_H */ +diff --git a/linux-headers/asm-powerpc/unistd_32.h b/linux-headers/asm-powerpc/unistd_32.h +index 38ee4dc35d..5d38a427e0 100644 +--- a/linux-headers/asm-powerpc/unistd_32.h ++++ b/linux-headers/asm-powerpc/unistd_32.h +@@ -448,6 +448,7 @@ + #define __NR_getxattrat 464 + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 ++#define __NR_open_tree_attr 467 + + + #endif /* _ASM_UNISTD_32_H */ +diff --git a/linux-headers/asm-powerpc/unistd_64.h b/linux-headers/asm-powerpc/unistd_64.h +index 5e5f156834..860a488e4d 100644 +--- a/linux-headers/asm-powerpc/unistd_64.h ++++ b/linux-headers/asm-powerpc/unistd_64.h +@@ -420,6 +420,7 @@ + #define __NR_getxattrat 464 + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 ++#define __NR_open_tree_attr 467 + + + #endif /* _ASM_UNISTD_64_H */ +diff --git a/linux-headers/asm-riscv/kvm.h b/linux-headers/asm-riscv/kvm.h +index f06bc5efcd..5f59fd226c 100644 +--- a/linux-headers/asm-riscv/kvm.h ++++ b/linux-headers/asm-riscv/kvm.h +@@ -182,6 +182,8 @@ enum KVM_RISCV_ISA_EXT_ID { + KVM_RISCV_ISA_EXT_SVVPTC, + KVM_RISCV_ISA_EXT_ZABHA, + KVM_RISCV_ISA_EXT_ZICCRSE, ++ KVM_RISCV_ISA_EXT_ZAAMO, ++ KVM_RISCV_ISA_EXT_ZALRSC, + KVM_RISCV_ISA_EXT_MAX, + }; + +diff --git a/linux-headers/asm-riscv/unistd_32.h b/linux-headers/asm-riscv/unistd_32.h +index 74f6127aed..a5e769f1d9 100644 +--- a/linux-headers/asm-riscv/unistd_32.h ++++ b/linux-headers/asm-riscv/unistd_32.h +@@ -314,6 +314,7 @@ + #define __NR_getxattrat 464 + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 ++#define __NR_open_tree_attr 467 + + + #endif /* _ASM_UNISTD_32_H */ +diff --git a/linux-headers/asm-riscv/unistd_64.h b/linux-headers/asm-riscv/unistd_64.h +index bb6a15a2ec..8df4d64841 100644 +--- a/linux-headers/asm-riscv/unistd_64.h ++++ b/linux-headers/asm-riscv/unistd_64.h +@@ -324,6 +324,7 @@ + #define __NR_getxattrat 464 + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 ++#define __NR_open_tree_attr 467 + + + #endif /* _ASM_UNISTD_64_H */ +diff --git a/linux-headers/asm-s390/unistd_32.h b/linux-headers/asm-s390/unistd_32.h +index 620201cb36..85eedbd18e 100644 +--- a/linux-headers/asm-s390/unistd_32.h ++++ b/linux-headers/asm-s390/unistd_32.h +@@ -439,5 +439,6 @@ + #define __NR_getxattrat 464 + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 ++#define __NR_open_tree_attr 467 + + #endif /* _ASM_S390_UNISTD_32_H */ +diff --git a/linux-headers/asm-s390/unistd_64.h b/linux-headers/asm-s390/unistd_64.h +index e7e4a10aaf..c03b1b9701 100644 +--- a/linux-headers/asm-s390/unistd_64.h ++++ b/linux-headers/asm-s390/unistd_64.h +@@ -387,5 +387,6 @@ + #define __NR_getxattrat 464 + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 ++#define __NR_open_tree_attr 467 + + #endif /* _ASM_S390_UNISTD_64_H */ +diff --git a/linux-headers/asm-x86/kvm.h b/linux-headers/asm-x86/kvm.h +index 86f2c34e7a..dc591fb17e 100644 +--- a/linux-headers/asm-x86/kvm.h ++++ b/linux-headers/asm-x86/kvm.h +@@ -557,6 +557,9 @@ struct kvm_x86_mce { + #define KVM_XEN_HVM_CONFIG_PVCLOCK_TSC_UNSTABLE (1 << 7) + #define KVM_XEN_HVM_CONFIG_SHARED_INFO_HVA (1 << 8) + ++#define KVM_XEN_MSR_MIN_INDEX 0x40000000u ++#define KVM_XEN_MSR_MAX_INDEX 0x4fffffffu ++ + struct kvm_xen_hvm_config { + __u32 flags; + __u32 msr; +diff --git a/linux-headers/asm-x86/unistd_32.h b/linux-headers/asm-x86/unistd_32.h +index a2eb492a75..491d6b4eb6 100644 +--- a/linux-headers/asm-x86/unistd_32.h ++++ b/linux-headers/asm-x86/unistd_32.h +@@ -457,6 +457,7 @@ + #define __NR_getxattrat 464 + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 ++#define __NR_open_tree_attr 467 + + + #endif /* _ASM_UNISTD_32_H */ +diff --git a/linux-headers/asm-x86/unistd_64.h b/linux-headers/asm-x86/unistd_64.h +index 2f5fc400f5..7cf88bf9bd 100644 +--- a/linux-headers/asm-x86/unistd_64.h ++++ b/linux-headers/asm-x86/unistd_64.h +@@ -380,6 +380,7 @@ + #define __NR_getxattrat 464 + #define __NR_listxattrat 465 + #define __NR_removexattrat 466 ++#define __NR_open_tree_attr 467 + + + #endif /* _ASM_UNISTD_64_H */ +diff --git a/linux-headers/asm-x86/unistd_x32.h b/linux-headers/asm-x86/unistd_x32.h +index fecd832e7f..82959111e6 100644 +--- a/linux-headers/asm-x86/unistd_x32.h ++++ b/linux-headers/asm-x86/unistd_x32.h +@@ -333,6 +333,7 @@ + #define __NR_getxattrat (__X32_SYSCALL_BIT + 464) + #define __NR_listxattrat (__X32_SYSCALL_BIT + 465) + #define __NR_removexattrat (__X32_SYSCALL_BIT + 466) ++#define __NR_open_tree_attr (__X32_SYSCALL_BIT + 467) + #define __NR_rt_sigaction (__X32_SYSCALL_BIT + 512) + #define __NR_rt_sigreturn (__X32_SYSCALL_BIT + 513) + #define __NR_ioctl (__X32_SYSCALL_BIT + 514) +diff --git a/linux-headers/linux/bits.h b/linux-headers/linux/bits.h +index c0d00c0a98..58596d18f4 100644 +--- a/linux-headers/linux/bits.h ++++ b/linux-headers/linux/bits.h +@@ -4,13 +4,9 @@ + #ifndef _LINUX_BITS_H + #define _LINUX_BITS_H + +-#define __GENMASK(h, l) \ +- (((~_UL(0)) - (_UL(1) << (l)) + 1) & \ +- (~_UL(0) >> (__BITS_PER_LONG - 1 - (h)))) ++#define __GENMASK(h, l) (((~_UL(0)) << (l)) & (~_UL(0) >> (BITS_PER_LONG - 1 - (h)))) + +-#define __GENMASK_ULL(h, l) \ +- (((~_ULL(0)) - (_ULL(1) << (l)) + 1) & \ +- (~_ULL(0) >> (__BITS_PER_LONG_LONG - 1 - (h)))) ++#define __GENMASK_ULL(h, l) (((~_ULL(0)) << (l)) & (~_ULL(0) >> (BITS_PER_LONG_LONG - 1 - (h)))) + + #define __GENMASK_U128(h, l) \ + ((_BIT128((h)) << 1) - (_BIT128(l))) +diff --git a/linux-headers/linux/const.h b/linux-headers/linux/const.h +index 2122610de7..95ede23342 100644 +--- a/linux-headers/linux/const.h ++++ b/linux-headers/linux/const.h +@@ -33,7 +33,7 @@ + * Missing __asm__ support + * + * __BIT128() would not work in the __asm__ code, as it shifts an +- * 'unsigned __init128' data type as direct representation of ++ * 'unsigned __int128' data type as direct representation of + * 128 bit constants is not supported in the gcc compiler, as + * they get silently truncated. + * +diff --git a/linux-headers/linux/iommufd.h b/linux-headers/linux/iommufd.h +index ccbdca5e11..cb0f7d6b4d 100644 +--- a/linux-headers/linux/iommufd.h ++++ b/linux-headers/linux/iommufd.h +@@ -55,6 +55,7 @@ enum { + IOMMUFD_CMD_VIOMMU_ALLOC = 0x90, + IOMMUFD_CMD_VDEVICE_ALLOC = 0x91, + IOMMUFD_CMD_IOAS_CHANGE_PROCESS = 0x92, ++ IOMMUFD_CMD_VEVENTQ_ALLOC = 0x93, + }; + + /** +@@ -392,6 +393,9 @@ struct iommu_vfio_ioas { + * Any domain attached to the non-PASID part of the + * device must also be flagged, otherwise attaching a + * PASID will blocked. ++ * For the user that wants to attach PASID, ioas is ++ * not recommended for both the non-PASID part ++ * and PASID part of the device. + * If IOMMU does not support PASID it will return + * error (-EOPNOTSUPP). + */ +@@ -608,9 +612,17 @@ enum iommu_hw_info_type { + * IOMMU_HWPT_GET_DIRTY_BITMAP + * IOMMU_HWPT_SET_DIRTY_TRACKING + * ++ * @IOMMU_HW_CAP_PCI_PASID_EXEC: Execute Permission Supported, user ignores it ++ * when the struct ++ * iommu_hw_info::out_max_pasid_log2 is zero. ++ * @IOMMU_HW_CAP_PCI_PASID_PRIV: Privileged Mode Supported, user ignores it ++ * when the struct ++ * iommu_hw_info::out_max_pasid_log2 is zero. + */ + enum iommufd_hw_capabilities { + IOMMU_HW_CAP_DIRTY_TRACKING = 1 << 0, ++ IOMMU_HW_CAP_PCI_PASID_EXEC = 1 << 1, ++ IOMMU_HW_CAP_PCI_PASID_PRIV = 1 << 2, + }; + + /** +@@ -626,6 +638,9 @@ enum iommufd_hw_capabilities { + * iommu_hw_info_type. + * @out_capabilities: Output the generic iommu capability info type as defined + * in the enum iommu_hw_capabilities. ++ * @out_max_pasid_log2: Output the width of PASIDs. 0 means no PASID support. ++ * PCI devices turn to out_capabilities to check if the ++ * specific capabilities is supported or not. + * @__reserved: Must be 0 + * + * Query an iommu type specific hardware information data from an iommu behind +@@ -649,7 +664,8 @@ struct iommu_hw_info { + __u32 data_len; + __aligned_u64 data_uptr; + __u32 out_data_type; +- __u32 __reserved; ++ __u8 out_max_pasid_log2; ++ __u8 __reserved[3]; + __aligned_u64 out_capabilities; + }; + #define IOMMU_GET_HW_INFO _IO(IOMMUFD_TYPE, IOMMUFD_CMD_GET_HW_INFO) +@@ -1014,4 +1030,115 @@ struct iommu_ioas_change_process { + #define IOMMU_IOAS_CHANGE_PROCESS \ + _IO(IOMMUFD_TYPE, IOMMUFD_CMD_IOAS_CHANGE_PROCESS) + ++/** ++ * enum iommu_veventq_flag - flag for struct iommufd_vevent_header ++ * @IOMMU_VEVENTQ_FLAG_LOST_EVENTS: vEVENTQ has lost vEVENTs ++ */ ++enum iommu_veventq_flag { ++ IOMMU_VEVENTQ_FLAG_LOST_EVENTS = (1U << 0), ++}; ++ ++/** ++ * struct iommufd_vevent_header - Virtual Event Header for a vEVENTQ Status ++ * @flags: Combination of enum iommu_veventq_flag ++ * @sequence: The sequence index of a vEVENT in the vEVENTQ, with a range of ++ * [0, INT_MAX] where the following index of INT_MAX is 0 ++ * ++ * Each iommufd_vevent_header reports a sequence index of the following vEVENT: ++ * ++ * +----------------------+-------+----------------------+-------+---+-------+ ++ * | header0 {sequence=0} | data0 | header1 {sequence=1} | data1 |...| dataN | ++ * +----------------------+-------+----------------------+-------+---+-------+ ++ * ++ * And this sequence index is expected to be monotonic to the sequence index of ++ * the previous vEVENT. If two adjacent sequence indexes has a delta larger than ++ * 1, it means that delta - 1 number of vEVENTs has lost, e.g. two lost vEVENTs: ++ * ++ * +-----+----------------------+-------+----------------------+-------+-----+ ++ * | ... | header3 {sequence=3} | data3 | header6 {sequence=6} | data6 | ... | ++ * +-----+----------------------+-------+----------------------+-------+-----+ ++ * ++ * If a vEVENT lost at the tail of the vEVENTQ and there is no following vEVENT ++ * providing the next sequence index, an IOMMU_VEVENTQ_FLAG_LOST_EVENTS header ++ * would be added to the tail, and no data would follow this header: ++ * ++ * +--+----------------------+-------+-----------------------------------------+ ++ * |..| header3 {sequence=3} | data3 | header4 {flags=LOST_EVENTS, sequence=4} | ++ * +--+----------------------+-------+-----------------------------------------+ ++ */ ++struct iommufd_vevent_header { ++ __u32 flags; ++ __u32 sequence; ++}; ++ ++/** ++ * enum iommu_veventq_type - Virtual Event Queue Type ++ * @IOMMU_VEVENTQ_TYPE_DEFAULT: Reserved for future use ++ * @IOMMU_VEVENTQ_TYPE_ARM_SMMUV3: ARM SMMUv3 Virtual Event Queue ++ */ ++enum iommu_veventq_type { ++ IOMMU_VEVENTQ_TYPE_DEFAULT = 0, ++ IOMMU_VEVENTQ_TYPE_ARM_SMMUV3 = 1, ++}; ++ ++/** ++ * struct iommu_vevent_arm_smmuv3 - ARM SMMUv3 Virtual Event ++ * (IOMMU_VEVENTQ_TYPE_ARM_SMMUV3) ++ * @evt: 256-bit ARM SMMUv3 Event record, little-endian. ++ * Reported event records: (Refer to "7.3 Event records" in SMMUv3 HW Spec) ++ * - 0x04 C_BAD_STE ++ * - 0x06 F_STREAM_DISABLED ++ * - 0x08 C_BAD_SUBSTREAMID ++ * - 0x0a C_BAD_CD ++ * - 0x10 F_TRANSLATION ++ * - 0x11 F_ADDR_SIZE ++ * - 0x12 F_ACCESS ++ * - 0x13 F_PERMISSION ++ * ++ * StreamID field reports a virtual device ID. To receive a virtual event for a ++ * device, a vDEVICE must be allocated via IOMMU_VDEVICE_ALLOC. ++ */ ++struct iommu_vevent_arm_smmuv3 { ++ __aligned_le64 evt[4]; ++}; ++ ++/** ++ * struct iommu_veventq_alloc - ioctl(IOMMU_VEVENTQ_ALLOC) ++ * @size: sizeof(struct iommu_veventq_alloc) ++ * @flags: Must be 0 ++ * @viommu_id: virtual IOMMU ID to associate the vEVENTQ with ++ * @type: Type of the vEVENTQ. Must be defined in enum iommu_veventq_type ++ * @veventq_depth: Maximum number of events in the vEVENTQ ++ * @out_veventq_id: The ID of the new vEVENTQ ++ * @out_veventq_fd: The fd of the new vEVENTQ. User space must close the ++ * successfully returned fd after using it ++ * @__reserved: Must be 0 ++ * ++ * Explicitly allocate a virtual event queue interface for a vIOMMU. A vIOMMU ++ * can have multiple FDs for different types, but is confined to one per @type. ++ * User space should open the @out_veventq_fd to read vEVENTs out of a vEVENTQ, ++ * if there are vEVENTs available. A vEVENTQ will lose events due to overflow, ++ * if the number of the vEVENTs hits @veventq_depth. ++ * ++ * Each vEVENT in a vEVENTQ encloses a struct iommufd_vevent_header followed by ++ * a type-specific data structure, in a normal case: ++ * ++ * +-+---------+-------+---------+-------+-----+---------+-------+-+ ++ * | | header0 | data0 | header1 | data1 | ... | headerN | dataN | | ++ * +-+---------+-------+---------+-------+-----+---------+-------+-+ ++ * ++ * unless a tailing IOMMU_VEVENTQ_FLAG_LOST_EVENTS header is logged (refer to ++ * struct iommufd_vevent_header). ++ */ ++struct iommu_veventq_alloc { ++ __u32 size; ++ __u32 flags; ++ __u32 viommu_id; ++ __u32 type; ++ __u32 veventq_depth; ++ __u32 out_veventq_id; ++ __u32 out_veventq_fd; ++ __u32 __reserved; ++}; ++#define IOMMU_VEVENTQ_ALLOC _IO(IOMMUFD_TYPE, IOMMUFD_CMD_VEVENTQ_ALLOC) + #endif +diff --git a/linux-headers/linux/kvm.h b/linux-headers/linux/kvm.h +index 27181b3dd8..e5f3e8b5a0 100644 +--- a/linux-headers/linux/kvm.h ++++ b/linux-headers/linux/kvm.h +@@ -921,6 +921,7 @@ struct kvm_enable_cap { + #define KVM_CAP_PRE_FAULT_MEMORY 236 + #define KVM_CAP_X86_APIC_BUS_CYCLES_NS 237 + #define KVM_CAP_X86_GUEST_MODE 238 ++#define KVM_CAP_ARM_WRITABLE_IMP_ID_REGS 239 + + struct kvm_irq_routing_irqchip { + __u32 irqchip; +diff --git a/linux-headers/linux/psp-sev.h b/linux-headers/linux/psp-sev.h +index 17bf191573..113c4ceb78 100644 +--- a/linux-headers/linux/psp-sev.h ++++ b/linux-headers/linux/psp-sev.h +@@ -73,13 +73,20 @@ typedef enum { + SEV_RET_INVALID_PARAM, + SEV_RET_RESOURCE_LIMIT, + SEV_RET_SECURE_DATA_INVALID, +- SEV_RET_INVALID_KEY = 0x27, +- SEV_RET_INVALID_PAGE_SIZE, +- SEV_RET_INVALID_PAGE_STATE, +- SEV_RET_INVALID_MDATA_ENTRY, +- SEV_RET_INVALID_PAGE_OWNER, +- SEV_RET_INVALID_PAGE_AEAD_OFLOW, +- SEV_RET_RMP_INIT_REQUIRED, ++ SEV_RET_INVALID_PAGE_SIZE = 0x0019, ++ SEV_RET_INVALID_PAGE_STATE = 0x001A, ++ SEV_RET_INVALID_MDATA_ENTRY = 0x001B, ++ SEV_RET_INVALID_PAGE_OWNER = 0x001C, ++ SEV_RET_AEAD_OFLOW = 0x001D, ++ SEV_RET_EXIT_RING_BUFFER = 0x001F, ++ SEV_RET_RMP_INIT_REQUIRED = 0x0020, ++ SEV_RET_BAD_SVN = 0x0021, ++ SEV_RET_BAD_VERSION = 0x0022, ++ SEV_RET_SHUTDOWN_REQUIRED = 0x0023, ++ SEV_RET_UPDATE_FAILED = 0x0024, ++ SEV_RET_RESTORE_REQUIRED = 0x0025, ++ SEV_RET_RMP_INITIALIZATION_FAILED = 0x0026, ++ SEV_RET_INVALID_KEY = 0x0027, + SEV_RET_MAX, + } sev_ret_code; + +diff --git a/linux-headers/linux/stddef.h b/linux-headers/linux/stddef.h +index e1416f7937..e1fcfcf3b3 100644 +--- a/linux-headers/linux/stddef.h ++++ b/linux-headers/linux/stddef.h +@@ -70,4 +70,6 @@ + #define __counted_by_be(m) + #endif + ++#define __kernel_nonstring ++ + #endif /* _LINUX_STDDEF_H */ +diff --git a/linux-headers/linux/vfio.h b/linux-headers/linux/vfio.h +index 1b5e254d6a..79bf8c0cc5 100644 +--- a/linux-headers/linux/vfio.h ++++ b/linux-headers/linux/vfio.h +@@ -671,6 +671,7 @@ enum { + */ + enum { + VFIO_AP_REQ_IRQ_INDEX, ++ VFIO_AP_CFG_CHG_IRQ_INDEX, + VFIO_AP_NUM_IRQS + }; + +@@ -931,29 +932,34 @@ struct vfio_device_bind_iommufd { + * VFIO_DEVICE_ATTACH_IOMMUFD_PT - _IOW(VFIO_TYPE, VFIO_BASE + 19, + * struct vfio_device_attach_iommufd_pt) + * @argsz: User filled size of this data. +- * @flags: Must be 0. ++ * @flags: Flags for attach. + * @pt_id: Input the target id which can represent an ioas or a hwpt + * allocated via iommufd subsystem. + * Output the input ioas id or the attached hwpt id which could + * be the specified hwpt itself or a hwpt automatically created + * for the specified ioas by kernel during the attachment. ++ * @pasid: The pasid to be attached, only meaningful when ++ * VFIO_DEVICE_ATTACH_PASID is set in @flags + * + * Associate the device with an address space within the bound iommufd. + * Undo by VFIO_DEVICE_DETACH_IOMMUFD_PT or device fd close. This is only + * allowed on cdev fds. + * +- * If a vfio device is currently attached to a valid hw_pagetable, without doing +- * a VFIO_DEVICE_DETACH_IOMMUFD_PT, a second VFIO_DEVICE_ATTACH_IOMMUFD_PT ioctl +- * passing in another hw_pagetable (hwpt) id is allowed. This action, also known +- * as a hw_pagetable replacement, will replace the device's currently attached +- * hw_pagetable with a new hw_pagetable corresponding to the given pt_id. ++ * If a vfio device or a pasid of this device is currently attached to a valid ++ * hw_pagetable (hwpt), without doing a VFIO_DEVICE_DETACH_IOMMUFD_PT, a second ++ * VFIO_DEVICE_ATTACH_IOMMUFD_PT ioctl passing in another hwpt id is allowed. ++ * This action, also known as a hw_pagetable replacement, will replace the ++ * currently attached hwpt of the device or the pasid of this device with a new ++ * hwpt corresponding to the given pt_id. + * + * Return: 0 on success, -errno on failure. + */ + struct vfio_device_attach_iommufd_pt { + __u32 argsz; + __u32 flags; ++#define VFIO_DEVICE_ATTACH_PASID (1 << 0) + __u32 pt_id; ++ __u32 pasid; + }; + + #define VFIO_DEVICE_ATTACH_IOMMUFD_PT _IO(VFIO_TYPE, VFIO_BASE + 19) +@@ -962,17 +968,21 @@ struct vfio_device_attach_iommufd_pt { + * VFIO_DEVICE_DETACH_IOMMUFD_PT - _IOW(VFIO_TYPE, VFIO_BASE + 20, + * struct vfio_device_detach_iommufd_pt) + * @argsz: User filled size of this data. +- * @flags: Must be 0. ++ * @flags: Flags for detach. ++ * @pasid: The pasid to be detached, only meaningful when ++ * VFIO_DEVICE_DETACH_PASID is set in @flags + * +- * Remove the association of the device and its current associated address +- * space. After it, the device should be in a blocking DMA state. This is only +- * allowed on cdev fds. ++ * Remove the association of the device or a pasid of the device and its current ++ * associated address space. After it, the device or the pasid should be in a ++ * blocking DMA state. This is only allowed on cdev fds. + * + * Return: 0 on success, -errno on failure. + */ + struct vfio_device_detach_iommufd_pt { + __u32 argsz; + __u32 flags; ++#define VFIO_DEVICE_DETACH_PASID (1 << 0) ++ __u32 pasid; + }; + + #define VFIO_DEVICE_DETACH_IOMMUFD_PT _IO(VFIO_TYPE, VFIO_BASE + 20) +-- +2.50.1 + diff --git a/SOURCES/kvm-linux-headers-update-from-6.15-kvm-next.patch b/SOURCES/kvm-linux-headers-update-from-6.15-kvm-next.patch new file mode 100644 index 0000000..963a4ad --- /dev/null +++ b/SOURCES/kvm-linux-headers-update-from-6.15-kvm-next.patch @@ -0,0 +1,126 @@ +From f1d5a02a236b16c839f4acdbb493d532c95987e0 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:45 +0200 +Subject: [PATCH 028/115] linux-headers: update from 6.15 + kvm/next + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [28/115] 8eeb6840c789d026eff3112d23332bc3172c51be (bonzini/rhel-qemu-kvm) + +This brings in the userspace TDX API. + +Reviewed-by: Xiaoyao Li +Signed-off-by: Paolo Bonzini +(cherry picked from commit 428c0acd953a626dab55e2c07401ce99c2271119) +Signed-off-by: Paolo Bonzini +--- + linux-headers/asm-x86/kvm.h | 71 +++++++++++++++++++++++++++++++++++++ + linux-headers/linux/kvm.h | 1 + + 2 files changed, 72 insertions(+) + +diff --git a/linux-headers/asm-x86/kvm.h b/linux-headers/asm-x86/kvm.h +index dc591fb17e..7fb57ccb2a 100644 +--- a/linux-headers/asm-x86/kvm.h ++++ b/linux-headers/asm-x86/kvm.h +@@ -439,6 +439,7 @@ struct kvm_sync_regs { + #define KVM_X86_QUIRK_MWAIT_NEVER_UD_FAULTS (1 << 6) + #define KVM_X86_QUIRK_SLOT_ZAP_ALL (1 << 7) + #define KVM_X86_QUIRK_STUFF_FEATURE_MSRS (1 << 8) ++#define KVM_X86_QUIRK_IGNORE_GUEST_PAT (1 << 9) + + #define KVM_STATE_NESTED_FORMAT_VMX 0 + #define KVM_STATE_NESTED_FORMAT_SVM 1 +@@ -928,4 +929,74 @@ struct kvm_hyperv_eventfd { + #define KVM_X86_SNP_VM 4 + #define KVM_X86_TDX_VM 5 + ++/* Trust Domain eXtension sub-ioctl() commands. */ ++enum kvm_tdx_cmd_id { ++ KVM_TDX_CAPABILITIES = 0, ++ KVM_TDX_INIT_VM, ++ KVM_TDX_INIT_VCPU, ++ KVM_TDX_INIT_MEM_REGION, ++ KVM_TDX_FINALIZE_VM, ++ KVM_TDX_GET_CPUID, ++ ++ KVM_TDX_CMD_NR_MAX, ++}; ++ ++struct kvm_tdx_cmd { ++ /* enum kvm_tdx_cmd_id */ ++ __u32 id; ++ /* flags for sub-commend. If sub-command doesn't use this, set zero. */ ++ __u32 flags; ++ /* ++ * data for each sub-command. An immediate or a pointer to the actual ++ * data in process virtual address. If sub-command doesn't use it, ++ * set zero. ++ */ ++ __u64 data; ++ /* ++ * Auxiliary error code. The sub-command may return TDX SEAMCALL ++ * status code in addition to -Exxx. ++ */ ++ __u64 hw_error; ++}; ++ ++struct kvm_tdx_capabilities { ++ __u64 supported_attrs; ++ __u64 supported_xfam; ++ __u64 reserved[254]; ++ ++ /* Configurable CPUID bits for userspace */ ++ struct kvm_cpuid2 cpuid; ++}; ++ ++struct kvm_tdx_init_vm { ++ __u64 attributes; ++ __u64 xfam; ++ __u64 mrconfigid[6]; /* sha384 digest */ ++ __u64 mrowner[6]; /* sha384 digest */ ++ __u64 mrownerconfig[6]; /* sha384 digest */ ++ ++ /* The total space for TD_PARAMS before the CPUIDs is 256 bytes */ ++ __u64 reserved[12]; ++ ++ /* ++ * Call KVM_TDX_INIT_VM before vcpu creation, thus before ++ * KVM_SET_CPUID2. ++ * This configuration supersedes KVM_SET_CPUID2s for VCPUs because the ++ * TDX module directly virtualizes those CPUIDs without VMM. The user ++ * space VMM, e.g. qemu, should make KVM_SET_CPUID2 consistent with ++ * those values. If it doesn't, KVM may have wrong idea of vCPUIDs of ++ * the guest, and KVM may wrongly emulate CPUIDs or MSRs that the TDX ++ * module doesn't virtualize. ++ */ ++ struct kvm_cpuid2 cpuid; ++}; ++ ++#define KVM_TDX_MEASURE_MEMORY_REGION _BITULL(0) ++ ++struct kvm_tdx_init_mem_region { ++ __u64 source_addr; ++ __u64 gpa; ++ __u64 nr_pages; ++}; ++ + #endif /* _ASM_X86_KVM_H */ +diff --git a/linux-headers/linux/kvm.h b/linux-headers/linux/kvm.h +index e5f3e8b5a0..99cc82a275 100644 +--- a/linux-headers/linux/kvm.h ++++ b/linux-headers/linux/kvm.h +@@ -369,6 +369,7 @@ struct kvm_run { + #define KVM_SYSTEM_EVENT_WAKEUP 4 + #define KVM_SYSTEM_EVENT_SUSPEND 5 + #define KVM_SYSTEM_EVENT_SEV_TERM 6 ++#define KVM_SYSTEM_EVENT_TDX_FATAL 7 + __u32 type; + __u32 ndata; + union { +-- +2.50.1 + diff --git a/SOURCES/kvm-memory-Change-memory_region_set_ram_discard_manager-.patch b/SOURCES/kvm-memory-Change-memory_region_set_ram_discard_manager-.patch new file mode 100644 index 0000000..facc8c4 --- /dev/null +++ b/SOURCES/kvm-memory-Change-memory_region_set_ram_discard_manager-.patch @@ -0,0 +1,160 @@ +From 9e0c4adad755988ea9f94f57a87e000dd7045962 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:11:27 +0200 +Subject: [PATCH 112/115] memory: Change + memory_region_set_ram_discard_manager() to return the result + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [112/115] 43deb5763dfabf964b55c7a6bb4363b96a5020cc (bonzini/rhel-qemu-kvm) + +Modify memory_region_set_ram_discard_manager() to return -EBUSY if a +RamDiscardManager is already set in the MemoryRegion. The caller must +handle this failure, such as having virtio-mem undo its actions and fail +the realize() process. Opportunistically move the call earlier to avoid +complex error handling. + +This change is beneficial when introducing a new RamDiscardManager +instance besides virtio-mem. After +ram_block_coordinated_discard_require(true) unlocks all +RamDiscardManager instances, only one instance is allowed to be set for +one MemoryRegion at present. + +Suggested-by: David Hildenbrand +Reviewed-by: David Hildenbrand +Reviewed-by: Pankaj Gupta +Tested-by: Alexey Kardashevskiy +Reviewed-by: Alexey Kardashevskiy +Reviewed-by: Xiaoyao Li +Signed-off-by: Chenyi Qiang +Link: https://lore.kernel.org/r/20250612082747.51539-3-chenyi.qiang@intel.com +Signed-off-by: Peter Xu +(cherry picked from commit ff1211154c45c9f7f82116ae9a8c72a848e4a8b5) +Signed-off-by: Paolo Bonzini +--- + hw/virtio/virtio-mem.c | 30 +++++++++++++++++------------- + include/exec/memory.h | 6 +++--- + system/memory.c | 10 +++++++--- + 3 files changed, 27 insertions(+), 19 deletions(-) + +diff --git a/hw/virtio/virtio-mem.c b/hw/virtio/virtio-mem.c +index ae26133f58..6284539243 100644 +--- a/hw/virtio/virtio-mem.c ++++ b/hw/virtio/virtio-mem.c +@@ -1049,6 +1049,17 @@ static void virtio_mem_device_realize(DeviceState *dev, Error **errp) + return; + } + ++ /* ++ * Set ourselves as RamDiscardManager before the plug handler maps the ++ * memory region and exposes it via an address space. ++ */ ++ if (memory_region_set_ram_discard_manager(&vmem->memdev->mr, ++ RAM_DISCARD_MANAGER(vmem))) { ++ error_setg(errp, "Failed to set RamDiscardManager"); ++ ram_block_coordinated_discard_require(false); ++ return; ++ } ++ + /* + * We don't know at this point whether shared RAM is migrated using + * QEMU or migrated using the file content. "x-ignore-shared" will be +@@ -1063,6 +1074,7 @@ static void virtio_mem_device_realize(DeviceState *dev, Error **errp) + ret = ram_block_discard_range(rb, 0, qemu_ram_get_used_length(rb)); + if (ret) { + error_setg_errno(errp, -ret, "Unexpected error discarding RAM"); ++ memory_region_set_ram_discard_manager(&vmem->memdev->mr, NULL); + ram_block_coordinated_discard_require(false); + return; + } +@@ -1124,13 +1136,6 @@ static void virtio_mem_device_realize(DeviceState *dev, Error **errp) + vmem->system_reset = VIRTIO_MEM_SYSTEM_RESET(obj); + vmem->system_reset->vmem = vmem; + qemu_register_resettable(obj); +- +- /* +- * Set ourselves as RamDiscardManager before the plug handler maps the +- * memory region and exposes it via an address space. +- */ +- memory_region_set_ram_discard_manager(&vmem->memdev->mr, +- RAM_DISCARD_MANAGER(vmem)); + } + + static void virtio_mem_device_unrealize(DeviceState *dev) +@@ -1138,12 +1143,6 @@ static void virtio_mem_device_unrealize(DeviceState *dev) + VirtIODevice *vdev = VIRTIO_DEVICE(dev); + VirtIOMEM *vmem = VIRTIO_MEM(dev); + +- /* +- * The unplug handler unmapped the memory region, it cannot be +- * found via an address space anymore. Unset ourselves. +- */ +- memory_region_set_ram_discard_manager(&vmem->memdev->mr, NULL); +- + qemu_unregister_resettable(OBJECT(vmem->system_reset)); + object_unref(OBJECT(vmem->system_reset)); + +@@ -1156,6 +1155,11 @@ static void virtio_mem_device_unrealize(DeviceState *dev) + virtio_del_queue(vdev, 0); + virtio_cleanup(vdev); + g_free(vmem->bitmap); ++ /* ++ * The unplug handler unmapped the memory region, it cannot be ++ * found via an address space anymore. Unset ourselves. ++ */ ++ memory_region_set_ram_discard_manager(&vmem->memdev->mr, NULL); + ram_block_coordinated_discard_require(false); + } + +diff --git a/include/exec/memory.h b/include/exec/memory.h +index 71b4a411cc..7cfbe203f4 100644 +--- a/include/exec/memory.h ++++ b/include/exec/memory.h +@@ -2488,13 +2488,13 @@ static inline bool memory_region_has_ram_discard_manager(MemoryRegion *mr) + * + * This function must not be called for a mapped #MemoryRegion, a #MemoryRegion + * that does not cover RAM, or a #MemoryRegion that already has a +- * #RamDiscardManager assigned. ++ * #RamDiscardManager assigned. Return 0 if the rdm is set successfully. + * + * @mr: the #MemoryRegion + * @rdm: #RamDiscardManager to set + */ +-void memory_region_set_ram_discard_manager(MemoryRegion *mr, +- RamDiscardManager *rdm); ++int memory_region_set_ram_discard_manager(MemoryRegion *mr, ++ RamDiscardManager *rdm); + + /** + * memory_region_find: translate an address/size relative to a +diff --git a/system/memory.c b/system/memory.c +index 5e6eb459d5..35011f731e 100644 +--- a/system/memory.c ++++ b/system/memory.c +@@ -2083,12 +2083,16 @@ RamDiscardManager *memory_region_get_ram_discard_manager(MemoryRegion *mr) + return mr->rdm; + } + +-void memory_region_set_ram_discard_manager(MemoryRegion *mr, +- RamDiscardManager *rdm) ++int memory_region_set_ram_discard_manager(MemoryRegion *mr, ++ RamDiscardManager *rdm) + { + g_assert(memory_region_is_ram(mr)); +- g_assert(!rdm || !mr->rdm); ++ if (mr->rdm && rdm) { ++ return -EBUSY; ++ } ++ + mr->rdm = rdm; ++ return 0; + } + + uint64_t ram_discard_manager_get_min_granularity(const RamDiscardManager *rdm, +-- +2.50.1 + diff --git a/SOURCES/kvm-memory-Export-a-helper-to-get-intersection-of-a-Memo.patch b/SOURCES/kvm-memory-Export-a-helper-to-get-intersection-of-a-Memo.patch new file mode 100644 index 0000000..b76b842 --- /dev/null +++ b/SOURCES/kvm-memory-Export-a-helper-to-get-intersection-of-a-Memo.patch @@ -0,0 +1,157 @@ +From 3a2ef25070d8135a62dfca5bff35114f520c52ad Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:11:27 +0200 +Subject: [PATCH 111/115] memory: Export a helper to get intersection of a + MemoryRegionSection with a given range + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [111/115] d7812c678358993209bbe280985f58fce9a34e18 (bonzini/rhel-qemu-kvm) + +Rename the helper to memory_region_section_intersect_range() to make it +more generic. Meanwhile, define the @end as Int128 and replace the +related operations with Int128_* format since the helper is exported as +a wider API. + +Suggested-by: Alexey Kardashevskiy +Reviewed-by: Alexey Kardashevskiy +Reviewed-by: Pankaj Gupta +Reviewed-by: David Hildenbrand +Reviewed-by: Zhao Liu +Reviewed-by: Xiaoyao Li +Signed-off-by: Chenyi Qiang +Link: https://lore.kernel.org/r/20250612082747.51539-2-chenyi.qiang@intel.com +Signed-off-by: Peter Xu +(cherry picked from commit f47a672a72acd6e2712031f0bc4d4f3ae4b6302c) +Signed-off-by: Paolo Bonzini +--- + hw/virtio/virtio-mem.c | 32 +++++--------------------------- + include/exec/memory.h | 30 ++++++++++++++++++++++++++++++ + 2 files changed, 35 insertions(+), 27 deletions(-) + +diff --git a/hw/virtio/virtio-mem.c b/hw/virtio/virtio-mem.c +index 4977658312..ae26133f58 100644 +--- a/hw/virtio/virtio-mem.c ++++ b/hw/virtio/virtio-mem.c +@@ -244,28 +244,6 @@ static int virtio_mem_for_each_plugged_range(VirtIOMEM *vmem, void *arg, + return ret; + } + +-/* +- * Adjust the memory section to cover the intersection with the given range. +- * +- * Returns false if the intersection is empty, otherwise returns true. +- */ +-static bool virtio_mem_intersect_memory_section(MemoryRegionSection *s, +- uint64_t offset, uint64_t size) +-{ +- uint64_t start = MAX(s->offset_within_region, offset); +- uint64_t end = MIN(s->offset_within_region + int128_get64(s->size), +- offset + size); +- +- if (end <= start) { +- return false; +- } +- +- s->offset_within_address_space += start - s->offset_within_region; +- s->offset_within_region = start; +- s->size = int128_make64(end - start); +- return true; +-} +- + typedef int (*virtio_mem_section_cb)(MemoryRegionSection *s, void *arg); + + static int virtio_mem_for_each_plugged_section(const VirtIOMEM *vmem, +@@ -287,7 +265,7 @@ static int virtio_mem_for_each_plugged_section(const VirtIOMEM *vmem, + first_bit + 1) - 1; + size = (last_bit - first_bit + 1) * vmem->block_size; + +- if (!virtio_mem_intersect_memory_section(&tmp, offset, size)) { ++ if (!memory_region_section_intersect_range(&tmp, offset, size)) { + break; + } + ret = cb(&tmp, arg); +@@ -319,7 +297,7 @@ static int virtio_mem_for_each_unplugged_section(const VirtIOMEM *vmem, + first_bit + 1) - 1; + size = (last_bit - first_bit + 1) * vmem->block_size; + +- if (!virtio_mem_intersect_memory_section(&tmp, offset, size)) { ++ if (!memory_region_section_intersect_range(&tmp, offset, size)) { + break; + } + ret = cb(&tmp, arg); +@@ -355,7 +333,7 @@ static void virtio_mem_notify_unplug(VirtIOMEM *vmem, uint64_t offset, + QLIST_FOREACH(rdl, &vmem->rdl_list, next) { + MemoryRegionSection tmp = *rdl->section; + +- if (!virtio_mem_intersect_memory_section(&tmp, offset, size)) { ++ if (!memory_region_section_intersect_range(&tmp, offset, size)) { + continue; + } + rdl->notify_discard(rdl, &tmp); +@@ -371,7 +349,7 @@ static int virtio_mem_notify_plug(VirtIOMEM *vmem, uint64_t offset, + QLIST_FOREACH(rdl, &vmem->rdl_list, next) { + MemoryRegionSection tmp = *rdl->section; + +- if (!virtio_mem_intersect_memory_section(&tmp, offset, size)) { ++ if (!memory_region_section_intersect_range(&tmp, offset, size)) { + continue; + } + ret = rdl->notify_populate(rdl, &tmp); +@@ -388,7 +366,7 @@ static int virtio_mem_notify_plug(VirtIOMEM *vmem, uint64_t offset, + if (rdl2 == rdl) { + break; + } +- if (!virtio_mem_intersect_memory_section(&tmp, offset, size)) { ++ if (!memory_region_section_intersect_range(&tmp, offset, size)) { + continue; + } + rdl2->notify_discard(rdl2, &tmp); +diff --git a/include/exec/memory.h b/include/exec/memory.h +index 296fd068c0..71b4a411cc 100644 +--- a/include/exec/memory.h ++++ b/include/exec/memory.h +@@ -1200,6 +1200,36 @@ MemoryRegionSection *memory_region_section_new_copy(MemoryRegionSection *s); + */ + void memory_region_section_free_copy(MemoryRegionSection *s); + ++/** ++ * memory_region_section_intersect_range: Adjust the memory section to cover ++ * the intersection with the given range. ++ * ++ * @s: the #MemoryRegionSection to be adjusted ++ * @offset: the offset of the given range in the memory region ++ * @size: the size of the given range ++ * ++ * Returns false if the intersection is empty, otherwise returns true. ++ */ ++static inline bool memory_region_section_intersect_range(MemoryRegionSection *s, ++ uint64_t offset, ++ uint64_t size) ++{ ++ uint64_t start = MAX(s->offset_within_region, offset); ++ Int128 end = int128_min(int128_add(int128_make64(s->offset_within_region), ++ s->size), ++ int128_add(int128_make64(offset), ++ int128_make64(size))); ++ ++ if (int128_le(end, int128_make64(start))) { ++ return false; ++ } ++ ++ s->offset_within_address_space += start - s->offset_within_region; ++ s->offset_within_region = start; ++ s->size = int128_sub(end, int128_make64(start)); ++ return true; ++} ++ + /** + * memory_region_init: Initialize a memory region + * +-- +2.50.1 + diff --git a/SOURCES/kvm-memory-Unify-the-definiton-of-ReplayRamPopulate-and-.patch b/SOURCES/kvm-memory-Unify-the-definiton-of-ReplayRamPopulate-and-.patch new file mode 100644 index 0000000..bf89604 --- /dev/null +++ b/SOURCES/kvm-memory-Unify-the-definiton-of-ReplayRamPopulate-and-.patch @@ -0,0 +1,277 @@ +From a0a6a0d1131461f2695f28a673f410bbb33262ef Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:11:28 +0200 +Subject: [PATCH 113/115] memory: Unify the definiton of ReplayRamPopulate() + and ReplayRamDiscard() + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [113/115] 8ad320473d66f2956d701a4625f815d2fcd8ae4b (bonzini/rhel-qemu-kvm) + +Update ReplayRamDiscard() function to return the result and unify the +ReplayRamPopulate() and ReplayRamDiscard() to ReplayRamDiscardState() at +the same time due to their identical definitions. This unification +simplifies related structures, such as VirtIOMEMReplayData, which makes +it cleaner. + +Reviewed-by: David Hildenbrand +Reviewed-by: Pankaj Gupta +Reviewed-by: Xiaoyao Li +Signed-off-by: Chenyi Qiang +Link: https://lore.kernel.org/r/20250612082747.51539-4-chenyi.qiang@intel.com +Signed-off-by: Peter Xu +(cherry picked from commit 2205b8466733f8c6e3306c964f31c5a7cac69dfa) +Signed-off-by: Paolo Bonzini +--- + hw/virtio/virtio-mem.c | 21 ++++++------ + include/exec/memory.h | 74 ++++++++++++++++++++++++++++++++---------- + migration/ram.c | 5 +-- + system/memory.c | 12 +++---- + 4 files changed, 76 insertions(+), 36 deletions(-) + +diff --git a/hw/virtio/virtio-mem.c b/hw/virtio/virtio-mem.c +index 6284539243..32af43852a 100644 +--- a/hw/virtio/virtio-mem.c ++++ b/hw/virtio/virtio-mem.c +@@ -1736,7 +1736,7 @@ static bool virtio_mem_rdm_is_populated(const RamDiscardManager *rdm, + } + + struct VirtIOMEMReplayData { +- void *fn; ++ ReplayRamDiscardState fn; + void *opaque; + }; + +@@ -1744,12 +1744,12 @@ static int virtio_mem_rdm_replay_populated_cb(MemoryRegionSection *s, void *arg) + { + struct VirtIOMEMReplayData *data = arg; + +- return ((ReplayRamPopulate)data->fn)(s, data->opaque); ++ return data->fn(s, data->opaque); + } + + static int virtio_mem_rdm_replay_populated(const RamDiscardManager *rdm, + MemoryRegionSection *s, +- ReplayRamPopulate replay_fn, ++ ReplayRamDiscardState replay_fn, + void *opaque) + { + const VirtIOMEM *vmem = VIRTIO_MEM(rdm); +@@ -1768,14 +1768,13 @@ static int virtio_mem_rdm_replay_discarded_cb(MemoryRegionSection *s, + { + struct VirtIOMEMReplayData *data = arg; + +- ((ReplayRamDiscard)data->fn)(s, data->opaque); +- return 0; ++ return data->fn(s, data->opaque); + } + +-static void virtio_mem_rdm_replay_discarded(const RamDiscardManager *rdm, +- MemoryRegionSection *s, +- ReplayRamDiscard replay_fn, +- void *opaque) ++static int virtio_mem_rdm_replay_discarded(const RamDiscardManager *rdm, ++ MemoryRegionSection *s, ++ ReplayRamDiscardState replay_fn, ++ void *opaque) + { + const VirtIOMEM *vmem = VIRTIO_MEM(rdm); + struct VirtIOMEMReplayData data = { +@@ -1784,8 +1783,8 @@ static void virtio_mem_rdm_replay_discarded(const RamDiscardManager *rdm, + }; + + g_assert(s->mr == &vmem->memdev->mr); +- virtio_mem_for_each_unplugged_section(vmem, s, &data, +- virtio_mem_rdm_replay_discarded_cb); ++ return virtio_mem_for_each_unplugged_section(vmem, s, &data, ++ virtio_mem_rdm_replay_discarded_cb); + } + + static void virtio_mem_rdm_register_listener(RamDiscardManager *rdm, +diff --git a/include/exec/memory.h b/include/exec/memory.h +index 7cfbe203f4..563353eab6 100644 +--- a/include/exec/memory.h ++++ b/include/exec/memory.h +@@ -566,8 +566,20 @@ static inline void ram_discard_listener_init(RamDiscardListener *rdl, + rdl->double_discard_supported = double_discard_supported; + } + +-typedef int (*ReplayRamPopulate)(MemoryRegionSection *section, void *opaque); +-typedef void (*ReplayRamDiscard)(MemoryRegionSection *section, void *opaque); ++/** ++ * typedef ReplayRamDiscardState: ++ * ++ * The callback handler for #RamDiscardManagerClass.replay_populated/ ++ * #RamDiscardManagerClass.replay_discarded to invoke on populated/discarded ++ * parts. ++ * ++ * @section: the #MemoryRegionSection of populated/discarded part ++ * @opaque: pointer to forward to the callback ++ * ++ * Returns 0 on success, or a negative error if failed. ++ */ ++typedef int (*ReplayRamDiscardState)(MemoryRegionSection *section, ++ void *opaque); + + /* + * RamDiscardManagerClass: +@@ -641,36 +653,38 @@ struct RamDiscardManagerClass { + /** + * @replay_populated: + * +- * Call the #ReplayRamPopulate callback for all populated parts within the +- * #MemoryRegionSection via the #RamDiscardManager. ++ * Call the #ReplayRamDiscardState callback for all populated parts within ++ * the #MemoryRegionSection via the #RamDiscardManager. + * + * In case any call fails, no further calls are made. + * + * @rdm: the #RamDiscardManager + * @section: the #MemoryRegionSection +- * @replay_fn: the #ReplayRamPopulate callback ++ * @replay_fn: the #ReplayRamDiscardState callback + * @opaque: pointer to forward to the callback + * + * Returns 0 on success, or a negative error if any notification failed. + */ + int (*replay_populated)(const RamDiscardManager *rdm, + MemoryRegionSection *section, +- ReplayRamPopulate replay_fn, void *opaque); ++ ReplayRamDiscardState replay_fn, void *opaque); + + /** + * @replay_discarded: + * +- * Call the #ReplayRamDiscard callback for all discarded parts within the +- * #MemoryRegionSection via the #RamDiscardManager. ++ * Call the #ReplayRamDiscardState callback for all discarded parts within ++ * the #MemoryRegionSection via the #RamDiscardManager. + * + * @rdm: the #RamDiscardManager + * @section: the #MemoryRegionSection +- * @replay_fn: the #ReplayRamDiscard callback ++ * @replay_fn: the #ReplayRamDiscardState callback + * @opaque: pointer to forward to the callback ++ * ++ * Returns 0 on success, or a negative error if any notification failed. + */ +- void (*replay_discarded)(const RamDiscardManager *rdm, +- MemoryRegionSection *section, +- ReplayRamDiscard replay_fn, void *opaque); ++ int (*replay_discarded)(const RamDiscardManager *rdm, ++ MemoryRegionSection *section, ++ ReplayRamDiscardState replay_fn, void *opaque); + + /** + * @register_listener: +@@ -711,15 +725,41 @@ uint64_t ram_discard_manager_get_min_granularity(const RamDiscardManager *rdm, + bool ram_discard_manager_is_populated(const RamDiscardManager *rdm, + const MemoryRegionSection *section); + ++/** ++ * ram_discard_manager_replay_populated: ++ * ++ * A wrapper to call the #RamDiscardManagerClass.replay_populated callback ++ * of the #RamDiscardManager. ++ * ++ * @rdm: the #RamDiscardManager ++ * @section: the #MemoryRegionSection ++ * @replay_fn: the #ReplayRamDiscardState callback ++ * @opaque: pointer to forward to the callback ++ * ++ * Returns 0 on success, or a negative error if any notification failed. ++ */ + int ram_discard_manager_replay_populated(const RamDiscardManager *rdm, + MemoryRegionSection *section, +- ReplayRamPopulate replay_fn, ++ ReplayRamDiscardState replay_fn, + void *opaque); + +-void ram_discard_manager_replay_discarded(const RamDiscardManager *rdm, +- MemoryRegionSection *section, +- ReplayRamDiscard replay_fn, +- void *opaque); ++/** ++ * ram_discard_manager_replay_discarded: ++ * ++ * A wrapper to call the #RamDiscardManagerClass.replay_discarded callback ++ * of the #RamDiscardManager. ++ * ++ * @rdm: the #RamDiscardManager ++ * @section: the #MemoryRegionSection ++ * @replay_fn: the #ReplayRamDiscardState callback ++ * @opaque: pointer to forward to the callback ++ * ++ * Returns 0 on success, or a negative error if any notification failed. ++ */ ++int ram_discard_manager_replay_discarded(const RamDiscardManager *rdm, ++ MemoryRegionSection *section, ++ ReplayRamDiscardState replay_fn, ++ void *opaque); + + void ram_discard_manager_register_listener(RamDiscardManager *rdm, + RamDiscardListener *rdl, +diff --git a/migration/ram.c b/migration/ram.c +index 0803f85b8a..aaeea04546 100644 +--- a/migration/ram.c ++++ b/migration/ram.c +@@ -874,8 +874,8 @@ static inline bool migration_bitmap_clear_dirty(RAMState *rs, + return ret; + } + +-static void dirty_bitmap_clear_section(MemoryRegionSection *section, +- void *opaque) ++static int dirty_bitmap_clear_section(MemoryRegionSection *section, ++ void *opaque) + { + const hwaddr offset = section->offset_within_region; + const hwaddr size = int128_get64(section->size); +@@ -894,6 +894,7 @@ static void dirty_bitmap_clear_section(MemoryRegionSection *section, + } + *cleared_bits += bitmap_count_one_with_offset(rb->bmap, start, npages); + bitmap_clear(rb->bmap, start, npages); ++ return 0; + } + + /* +diff --git a/system/memory.c b/system/memory.c +index 35011f731e..8c36ab17b6 100644 +--- a/system/memory.c ++++ b/system/memory.c +@@ -2115,7 +2115,7 @@ bool ram_discard_manager_is_populated(const RamDiscardManager *rdm, + + int ram_discard_manager_replay_populated(const RamDiscardManager *rdm, + MemoryRegionSection *section, +- ReplayRamPopulate replay_fn, ++ ReplayRamDiscardState replay_fn, + void *opaque) + { + RamDiscardManagerClass *rdmc = RAM_DISCARD_MANAGER_GET_CLASS(rdm); +@@ -2124,15 +2124,15 @@ int ram_discard_manager_replay_populated(const RamDiscardManager *rdm, + return rdmc->replay_populated(rdm, section, replay_fn, opaque); + } + +-void ram_discard_manager_replay_discarded(const RamDiscardManager *rdm, +- MemoryRegionSection *section, +- ReplayRamDiscard replay_fn, +- void *opaque) ++int ram_discard_manager_replay_discarded(const RamDiscardManager *rdm, ++ MemoryRegionSection *section, ++ ReplayRamDiscardState replay_fn, ++ void *opaque) + { + RamDiscardManagerClass *rdmc = RAM_DISCARD_MANAGER_GET_CLASS(rdm); + + g_assert(rdmc->replay_discarded); +- rdmc->replay_discarded(rdm, section, replay_fn, opaque); ++ return rdmc->replay_discarded(rdm, section, replay_fn, opaque); + } + + void ram_discard_manager_register_listener(RamDiscardManager *rdm, +-- +2.50.1 + diff --git a/SOURCES/kvm-meson-configure-add-valgrind-option-en-dis-able-valg.patch b/SOURCES/kvm-meson-configure-add-valgrind-option-en-dis-able-valg.patch new file mode 100644 index 0000000..93c87f9 --- /dev/null +++ b/SOURCES/kvm-meson-configure-add-valgrind-option-en-dis-able-valg.patch @@ -0,0 +1,110 @@ +From 0277328b5a2d1df5d9843423ab5f5fa9481bad79 Mon Sep 17 00:00:00 2001 +From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= +Date: Fri, 25 Apr 2025 13:17:12 +0100 +Subject: [PATCH 1/5] meson/configure: add 'valgrind' option & --{en, + dis}able-valgrind flag +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Daniel P. Berrangé +RH-MergeRequest: 359: distro: add an explicit valgrind-devel build dep +RH-Jira: RHEL-88153 +RH-Acked-by: Eric Blake +RH-Acked-by: Jon Maloy +RH-Commit: [1/2] ba9bc44ef9cef6fa76e2092500608575f223f1f7 (berrange/centos-src-qemu) + +Currently valgrind debugging support for coroutine stacks is enabled +unconditionally when valgrind/valgrind.h is found. There is no way +to disable valgrind support if valgrind.h is present in the build env. + +This is bad for distros, as an dependency far down the chain may cause +valgrind.h to become installed, inadvertently enabling QEMU's valgrind +debugging support. It also means if a distro wants valgrind support +there is no way to mandate this. + +The solution is to add a 'valgrind' build feature to meson and thus +configure script. + +Signed-off-by: Daniel P. Berrangé +Reviewed-by: Thomas Huth +Message-ID: <20250425121713.1913424-1-berrange@redhat.com> +Signed-off-by: Thomas Huth +(cherry picked from commit 6b1c744ec0d66d6d568f9a156282153fc11a21cf) + +Conflicts: + meson.build - context from upstream is not present in older tree +--- + meson.build | 13 ++++++++++++- + meson_options.txt | 2 ++ + scripts/meson-buildoptions.sh | 3 +++ + 3 files changed, 17 insertions(+), 1 deletion(-) + +diff --git a/meson.build b/meson.build +index 1dd97c6f49..5bb2b757c3 100644 +--- a/meson.build ++++ b/meson.build +@@ -2463,7 +2463,17 @@ config_host_data.set('CONFIG_FSTRIM', qga_fstrim) + # has_header + config_host_data.set('CONFIG_EPOLL', cc.has_header('sys/epoll.h')) + config_host_data.set('CONFIG_LINUX_MAGIC_H', cc.has_header('linux/magic.h')) +-config_host_data.set('CONFIG_VALGRIND_H', cc.has_header('valgrind/valgrind.h')) ++valgrind = false ++if get_option('valgrind').allowed() ++ if cc.has_header('valgrind/valgrind.h') ++ valgrind = true ++ else ++ if get_option('valgrind').enabled() ++ error('valgrind requested but valgrind.h not found') ++ endif ++ endif ++endif ++config_host_data.set('CONFIG_VALGRIND_H', valgrind) + config_host_data.set('HAVE_BTRFS_H', cc.has_header('linux/btrfs.h')) + config_host_data.set('HAVE_DRM_H', cc.has_header('libdrm/drm.h')) + config_host_data.set('HAVE_PTY_H', cc.has_header('pty.h')) +@@ -4549,6 +4559,7 @@ summary_info += {'libdw': libdw} + if host_os == 'freebsd' + summary_info += {'libinotify-kqueue': inotify} + endif ++summary_info += {'valgrind': valgrind} + summary(summary_info, bool_yn: true, section: 'Dependencies') + + if host_arch == 'unknown' +diff --git a/meson_options.txt b/meson_options.txt +index aa2ba0baef..da06441fdf 100644 +--- a/meson_options.txt ++++ b/meson_options.txt +@@ -113,6 +113,8 @@ option('dbus_display', type: 'feature', value: 'auto', + description: '-display dbus support') + option('tpm', type : 'feature', value : 'auto', + description: 'TPM support') ++option('valgrind', type : 'feature', value: 'auto', ++ description: 'valgrind debug support for coroutine stacks') + + # Do not enable it by default even for Mingw32, because it doesn't + # work on Wine. +diff --git a/scripts/meson-buildoptions.sh b/scripts/meson-buildoptions.sh +index 5f0cbfc725..251470ea6d 100644 +--- a/scripts/meson-buildoptions.sh ++++ b/scripts/meson-buildoptions.sh +@@ -191,6 +191,7 @@ meson_options_help() { + printf "%s\n" ' u2f U2F emulation support' + printf "%s\n" ' uadk UADK Library support' + printf "%s\n" ' usb-redir libusbredir support' ++ printf "%s\n" ' valgrind valgrind debug support for coroutine stacks' + printf "%s\n" ' vde vde network backend support' + printf "%s\n" ' vdi vdi image format support' + printf "%s\n" ' vduse-blk-export' +@@ -509,6 +510,8 @@ _meson_option_parse() { + --disable-uadk) printf "%s" -Duadk=disabled ;; + --enable-usb-redir) printf "%s" -Dusb_redir=enabled ;; + --disable-usb-redir) printf "%s" -Dusb_redir=disabled ;; ++ --enable-valgrind) printf "%s" -Dvalgrind=enabled ;; ++ --disable-valgrind) printf "%s" -Dvalgrind=disabled ;; + --enable-vde) printf "%s" -Dvde=enabled ;; + --disable-vde) printf "%s" -Dvde=disabled ;; + --enable-vdi) printf "%s" -Dvdi=enabled ;; +-- +2.48.1 + diff --git a/SOURCES/kvm-migration-Fix-UAF-for-incoming-migration-on-Migratio.patch b/SOURCES/kvm-migration-Fix-UAF-for-incoming-migration-on-Migratio.patch new file mode 100644 index 0000000..d9d12bf --- /dev/null +++ b/SOURCES/kvm-migration-Fix-UAF-for-incoming-migration-on-Migratio.patch @@ -0,0 +1,180 @@ +From 5d7d7a2ec6301f4d0b0dbea4fbdcab4e41a9cf07 Mon Sep 17 00:00:00 2001 +From: Peter Xu +Date: Thu, 20 Feb 2025 08:24:59 -0500 +Subject: [PATCH 7/9] migration: Fix UAF for incoming migration on + MigrationState + +RH-Author: Peter Xu +RH-MergeRequest: 344: migration: Fix UAF for incoming migration on MigrationState +RH-Jira: RHEL-69775 +RH-Acked-by: Juraj Marcin +RH-Acked-by: Jon Maloy +RH-Commit: [1/1] 106e2b4c1c461202c912b5e3ea7e586c4ab05d8c (peterx/qemu-kvm) + +On the incoming migration side, QEMU uses a coroutine to load all the VM +states. Inside, it may reference MigrationState on global states like +migration capabilities, parameters, error state, shared mutexes and more. + +However there's nothing yet to make sure MigrationState won't get +destroyed (e.g. after migration_shutdown()). Meanwhile there's also no API +available to remove the incoming coroutine in migration_shutdown(), +avoiding it to access the freed elements. + +There's a bug report showing this can happen and crash dest QEMU when +migration is cancelled on source. + +When it happens, the dest main thread is trying to cleanup everything: + + #0 qemu_aio_coroutine_enter + #1 aio_dispatch_handler + #2 aio_poll + #3 monitor_cleanup + #4 qemu_cleanup + #5 qemu_default_main + +Then it found the migration incoming coroutine, schedule it (even after +migration_shutdown()), causing crash: + + #0 __pthread_kill_implementation + #1 __pthread_kill_internal + #2 __GI_raise + #3 __GI_abort + #4 __assert_fail_base + #5 __assert_fail + #6 qemu_mutex_lock_impl + #7 qemu_lockable_mutex_lock + #8 qemu_lockable_lock + #9 qemu_lockable_auto_lock + #10 migrate_set_error + #11 process_incoming_migration_co + #12 coroutine_trampoline + +To fix it, take a refcount after an incoming setup is properly done when +qmp_migrate_incoming() succeeded the 1st time. As it's during a QMP +handler which needs BQL, it means the main loop is still alive (without +going into cleanups, which also needs BQL). + +Releasing the refcount now only until the incoming migration coroutine +finished or failed. Hence the refcount is valid for both (1) setup phase +of incoming ports, mostly IO watches (e.g. qio_channel_add_watch_full()), +and (2) the incoming coroutine itself (process_incoming_migration_co()). + +Note that we can't unref in migration_incoming_state_destroy(), because +both qmp_xen_load_devices_state() and load_snapshot() will use it without +an incoming migration. Those hold BQL so they're not prone to this issue. + +PS: I suspect nobody uses Xen's command at all, as it didn't register yank, +hence AFAIU the command should crash on master when trying to unregister +yank in migration_incoming_state_destroy().. but that's another story. + +Also note that in some incoming failure cases we may not always unref the +MigrationState refcount, which is a trade-off to keep things simple. We +could make it accurate, but it can be an overkill. Some examples: + + - Unlike most of the rest protocols, socket_start_incoming_migration() + may create net listener after incoming port setup sucessfully. + It means we can't unref in migration_channel_process_incoming() as a + generic path because socket protocol might keep using MigrationState. + + - For either socket or file, multiple IO watches might be created, it + means logically each IO watch needs to take one refcount for + MigrationState so as to be 100% accurate on ownership of refcount taken. + +In general, we at least need per-protocol handling to make it accurate, +which can be an overkill if we know incoming failed after all. Add a short +comment to explain that when taking the refcount in qmp_migrate_incoming(). + +Bugzilla: https://issues.redhat.com/browse/RHEL-69775 +Tested-by: Yan Fu +Signed-off-by: Peter Xu +Reviewed-by: Fabiano Rosas +Message-ID: <20250220132459.512610-1-peterx@redhat.com> +Signed-off-by: Fabiano Rosas +(cherry picked from commit d657a14de5d597bbfe7b54e4c4f0646f440e98ad) +Signed-off-by: Peter Xu +--- + migration/migration.c | 40 ++++++++++++++++++++++++++++++++++++++-- + 1 file changed, 38 insertions(+), 2 deletions(-) + +diff --git a/migration/migration.c b/migration/migration.c +index 999d4cac54..aabdc45c16 100644 +--- a/migration/migration.c ++++ b/migration/migration.c +@@ -115,6 +115,27 @@ static void migration_downtime_start(MigrationState *s) + s->downtime_start = qemu_clock_get_ms(QEMU_CLOCK_REALTIME); + } + ++/* ++ * This is unfortunate: incoming migration actually needs the outgoing ++ * migration state (MigrationState) to be there too, e.g. to query ++ * capabilities, parameters, using locks, setup errors, etc. ++ * ++ * NOTE: when calling this, making sure current_migration exists and not ++ * been freed yet! Otherwise trying to access the refcount is already ++ * an use-after-free itself.. ++ * ++ * TODO: Move shared part of incoming / outgoing out into separate object. ++ * Then this is not needed. ++ */ ++static void migrate_incoming_ref_outgoing_state(void) ++{ ++ object_ref(migrate_get_current()); ++} ++static void migrate_incoming_unref_outgoing_state(void) ++{ ++ object_unref(migrate_get_current()); ++} ++ + static void migration_downtime_end(MigrationState *s) + { + int64_t now = qemu_clock_get_ms(QEMU_CLOCK_REALTIME); +@@ -821,7 +842,7 @@ process_incoming_migration_co(void *opaque) + * postcopy thread. + */ + trace_process_incoming_migration_co_postcopy_end_main(); +- return; ++ goto out; + } + /* Else if something went wrong then just fall out of the normal exit */ + } +@@ -837,7 +858,8 @@ process_incoming_migration_co(void *opaque) + } + + migration_bh_schedule(process_incoming_migration_bh, mis); +- return; ++ goto out; ++ + fail: + migrate_set_state(&mis->state, MIGRATION_STATUS_ACTIVE, + MIGRATION_STATUS_FAILED); +@@ -854,6 +876,9 @@ fail: + + exit(EXIT_FAILURE); + } ++out: ++ /* Pairs with the refcount taken in qmp_migrate_incoming() */ ++ migrate_incoming_unref_outgoing_state(); + } + + /** +@@ -1875,6 +1900,17 @@ void qmp_migrate_incoming(const char *uri, bool has_channels, + return; + } + ++ /* ++ * Making sure MigrationState is available until incoming migration ++ * completes. ++ * ++ * NOTE: QEMU _might_ leak this refcount in some failure paths, but ++ * that's OK. This is the minimum change we need to at least making ++ * sure success case is clean on the refcount. We can try harder to ++ * make it accurate for any kind of failures, but it might be an ++ * overkill and doesn't bring us much benefit. ++ */ ++ migrate_incoming_ref_outgoing_state(); + once = false; + } + +-- +2.48.1 + diff --git a/SOURCES/kvm-migration-postcopy-Spatial-locality-page-hint-for-pr.patch b/SOURCES/kvm-migration-postcopy-Spatial-locality-page-hint-for-pr.patch new file mode 100644 index 0000000..bacd275 --- /dev/null +++ b/SOURCES/kvm-migration-postcopy-Spatial-locality-page-hint-for-pr.patch @@ -0,0 +1,237 @@ +From 1fa31324da8ebba64a44c1e9b64f7e59c29f3d75 Mon Sep 17 00:00:00 2001 +From: Peter Xu +Date: Thu, 24 Apr 2025 18:07:05 -0400 +Subject: [PATCH 1/2] migration/postcopy: Spatial locality page hint for + preempt mode +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Peter Xu +RH-MergeRequest: 358: migration/postcopy: Spatial locality page hint for preempt mode +RH-Jira: RHEL-85159 +RH-Acked-by: Juraj Marcin +RH-Acked-by: Daniel P. Berrangé +RH-Commit: [1/1] f5bce349c80f98428c73a3898f87d4d10ec2f4bd (peterx/qemu-kvm) + +The preempt mode postcopy has been introduced for a while. From latency +POV, it should always win the vanilla postcopy. + +However there's one thing missing when preempt mode is enabled right now, +which is the spatial locality hint when there're page requests from the +destination side. + +In vanilla postcopy, as long as a page request was unqueued, it will update +the PSS of the precopy background stream, so that after a page request the +background thread will move the pages after whatever was requested. It's +pretty much a natural behavior when there's only one channel anyway, and +one scanner to send the pages. + +Preempt mode didn't follow that, because preempt mode has its own channel +and its own PSS (which doesn't linearly scan the guest memory, but +dedicated to resolve page requested from destination). So the page request +process and the background migration process are completely separate. + +This patch adds the hint explicitly for preempt mode. With that, whenever +the preempt mode receives a page request on the source, it will service the +remote page fault in the return path, then it'll provide a hint to the +background thread so that we'll start sending the pages right after the +requested ones in the background, assuming the follow up pages have a +higher chance to be accessed later. + +NOTE: since the background migration thread and return path thread run +completely concurrently, it doesn't always mean the hint will be applied +every single time. For example, it's possible that the return path thread +receives multiple page requests in a row without the background thread +getting the chance to consume one. In such case, the preempt thread only +provide the hint if the previous hint has been consumed. After all, +there's no point queuing hints when we only have one linear scanner. + +This could measureably improve the simple sequential memory access pattern +during postcopy (when preempt is on). For random accesses, I can measure a +slight increase of remote page fault latency from ~500us -> ~600us, that +could be a trade-off to have such hint mechanism, and after all that's +still greatly improved comparing to vanilla postcopy on random (~10ms). + +The patch is verified by our QE team in a video streaming test case, to +reduce the pause of the video from ~1min to a few seconds when switching +over to postcopy with preempt mode. + +Reported-by: Xiaohui Li +Tested-by: Xiaohui Li +Reviewed-by: Juraj Marcin +Link: https://lore.kernel.org/r/20250424220705.195544-1-peterx@redhat.com +Signed-off-by: Peter Xu +(cherry picked from commit 20d82622812d888478d04a2d0d8575d70eb5d749) +Signed-off-by: Peter Xu +--- + migration/ram.c | 97 ++++++++++++++++++++++++++++++++++++++++++++++++- + 1 file changed, 96 insertions(+), 1 deletion(-) + +diff --git a/migration/ram.c b/migration/ram.c +index edec1a2d07..0803f85b8a 100644 +--- a/migration/ram.c ++++ b/migration/ram.c +@@ -112,6 +112,36 @@ + + XBZRLECacheStats xbzrle_counters; + ++/* ++ * This structure locates a specific location of a guest page. In QEMU, ++ * it's described in a tuple of (ramblock, offset). ++ */ ++struct PageLocation { ++ RAMBlock *block; ++ unsigned long offset; ++}; ++typedef struct PageLocation PageLocation; ++ ++/** ++ * PageLocationHint: describes a hint to a page location ++ * ++ * @valid set if the hint is vaild and to be consumed ++ * @location: the hint content ++ * ++ * In postcopy preempt mode, the urgent channel may provide hints to the ++ * background channel, so that QEMU source can try to migrate whatever is ++ * right after the requested urgent pages. ++ * ++ * This is based on the assumption that the VM (already running on the ++ * destination side) tends to access the memory with spatial locality. ++ * This is also the default behavior of vanilla postcopy (preempt off). ++ */ ++struct PageLocationHint { ++ bool valid; ++ PageLocation location; ++}; ++typedef struct PageLocationHint PageLocationHint; ++ + /* used by the search for pages to send */ + struct PageSearchStatus { + /* The migration channel used for a specific host page */ +@@ -414,6 +444,13 @@ struct RAMState { + * RAM migration. + */ + unsigned int postcopy_bmap_sync_requested; ++ /* ++ * Page hint during postcopy when preempt mode is on. Return path ++ * thread sets it, while background migration thread consumes it. ++ * ++ * Protected by @bitmap_mutex. ++ */ ++ PageLocationHint page_hint; + }; + typedef struct RAMState RAMState; + +@@ -2091,6 +2128,21 @@ static void pss_host_page_finish(PageSearchStatus *pss) + pss->host_page_start = pss->host_page_end = 0; + } + ++static void ram_page_hint_update(RAMState *rs, PageSearchStatus *pss) ++{ ++ PageLocationHint *hint = &rs->page_hint; ++ ++ /* If there's a pending hint not consumed, don't bother */ ++ if (hint->valid) { ++ return; ++ } ++ ++ /* Provide a hint to the background stream otherwise */ ++ hint->location.block = pss->block; ++ hint->location.offset = pss->page; ++ hint->valid = true; ++} ++ + /* + * Send an urgent host page specified by `pss'. Need to be called with + * bitmap_mutex held. +@@ -2136,6 +2188,7 @@ out: + /* For urgent requests, flush immediately if sent */ + if (sent) { + qemu_fflush(pss->pss_channel); ++ ram_page_hint_update(rs, pss); + } + return ret; + } +@@ -2223,6 +2276,30 @@ static int ram_save_host_page(RAMState *rs, PageSearchStatus *pss) + return (res < 0 ? res : pages); + } + ++static bool ram_page_hint_valid(RAMState *rs) ++{ ++ /* There's only page hint during postcopy preempt mode */ ++ if (!postcopy_preempt_active()) { ++ return false; ++ } ++ ++ return rs->page_hint.valid; ++} ++ ++static void ram_page_hint_collect(RAMState *rs, RAMBlock **block, ++ unsigned long *page) ++{ ++ PageLocationHint *hint = &rs->page_hint; ++ ++ assert(hint->valid); ++ ++ *block = hint->location.block; ++ *page = hint->location.offset; ++ ++ /* Mark the hint consumed */ ++ hint->valid = false; ++} ++ + /** + * ram_find_and_save_block: finds a dirty page and sends it to f + * +@@ -2239,6 +2316,8 @@ static int ram_save_host_page(RAMState *rs, PageSearchStatus *pss) + static int ram_find_and_save_block(RAMState *rs) + { + PageSearchStatus *pss = &rs->pss[RAM_CHANNEL_PRECOPY]; ++ unsigned long next_page; ++ RAMBlock *next_block; + int pages = 0; + + /* No dirty page as there is zero RAM */ +@@ -2258,7 +2337,14 @@ static int ram_find_and_save_block(RAMState *rs) + rs->last_page = 0; + } + +- pss_init(pss, rs->last_seen_block, rs->last_page); ++ if (ram_page_hint_valid(rs)) { ++ ram_page_hint_collect(rs, &next_block, &next_page); ++ } else { ++ next_block = rs->last_seen_block; ++ next_page = rs->last_page; ++ } ++ ++ pss_init(pss, next_block, next_page); + + while (true){ + if (!get_queued_page(rs, pss)) { +@@ -2392,6 +2478,13 @@ static void ram_save_cleanup(void *opaque) + migration_ops = NULL; + } + ++static void ram_page_hint_reset(PageLocationHint *hint) ++{ ++ hint->location.block = NULL; ++ hint->location.offset = 0; ++ hint->valid = false; ++} ++ + static void ram_state_reset(RAMState *rs) + { + int i; +@@ -2404,6 +2497,8 @@ static void ram_state_reset(RAMState *rs) + rs->last_page = 0; + rs->last_version = ram_list.version; + rs->xbzrle_started = false; ++ ++ ram_page_hint_reset(&rs->page_hint); + } + + #define MAX_WAIT 50 /* ms, half buffered_file limit */ +-- +2.48.1 + diff --git a/SOURCES/kvm-mirror-Allow-QMP-override-to-declare-target-already-.patch b/SOURCES/kvm-mirror-Allow-QMP-override-to-declare-target-already-.patch new file mode 100644 index 0000000..c00be71 --- /dev/null +++ b/SOURCES/kvm-mirror-Allow-QMP-override-to-declare-target-already-.patch @@ -0,0 +1,295 @@ +From a5f6042a0c80daf3672fa071b724cb05e6f6e928 Mon Sep 17 00:00:00 2001 +From: Eric Blake +Date: Fri, 9 May 2025 15:40:25 -0500 +Subject: [PATCH 10/16] mirror: Allow QMP override to declare target already + zero + +RH-Author: Eric Blake +RH-MergeRequest: 365: blockdev-mirror: More efficient handling of sparse mirrors +RH-Jira: RHEL-82906 RHEL-83015 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Jon Maloy +RH-Commit: [8/14] fb054864175d83e9d232464295b170808bee0e6c (ebblake/centos-qemu-kvm) + +QEMU has an optimization for a just-created drive-mirror destination +that is not possible for blockdev-mirror (which can't create the +destination) - any time we know the destination starts life as all +zeroes, we can skip a pre-zeroing pass on the destination. Recent +patches have added an improved heuristic for detecting if a file +contains all zeroes, and we plan to use that heuristic in upcoming +patches. But since a heuristic cannot quickly detect all scenarios, +and there may be cases where the caller is aware of information that +QEMU cannot learn quickly, it makes sense to have a way to tell QEMU +to assume facts about the destination that can make the mirror +operation faster. Given our existing example of "qemu-img convert +--target-is-zero", it is time to expose this override in QMP for +blockdev-mirror as well. + +This patch results in some slight redundancy between the older +s->zero_target (set any time mode==FULL and the destination image was +not just created - ie. clear if drive-mirror is asking to skip the +pre-zero pass) and the newly-introduced s->target_is_zero (in addition +to the QMP override, it is set when drive-mirror creates the +destination image); this will be cleaned up in the next patch. + +There is also a subtlety that we must consider. When drive-mirror is +passing target_is_zero on behalf of a just-created image, we know the +image is sparse (skipping the pre-zeroing keeps it that way), so it +doesn't matter whether the destination also has "discard":"unmap" and +"detect-zeroes":"unmap". But now that we are letting the user set the +knob for target-is-zero, if the user passes a pre-existing file that +is fully allocated, it is fine to leave the file fully allocated under +"detect-zeroes":"on", but if the file is open with +"detect-zeroes":"unmap", we should really be trying harder to punch +holes in the destination for every region of zeroes copied from the +source. The easiest way to do this is to still run the pre-zeroing +pass (turning the entire destination file sparse before populating +just the allocated portions of the source), even though that currently +results in double I/O to the portions of the file that are allocated. +A later patch will add further optimizations to reduce redundant +zeroing I/O during the mirror operation. + +Since "target-is-zero":true is designed for optimizations, it is okay +to silently ignore the parameter rather than erroring if the user ever +sets the parameter in a scenario where the mirror job can't exploit it +(for example, when doing "sync":"top" instead of "sync":"full", we +can't pre-zero, so setting the parameter won't make a speed +difference). + +Signed-off-by: Eric Blake +Acked-by: Markus Armbruster +Message-ID: <20250509204341.3553601-23-eblake@redhat.com> +Reviewed-by: Sunny Zhu +Reviewed-by: Stefan Hajnoczi +(cherry picked from commit d17a34bfb94bda3a89d7320ae67255ded1d8c939) +Jira: https://issues.redhat.com/browse/RHEL-82906 +Jira: https://issues.redhat.com/browse/RHEL-83015 +Signed-off-by: Eric Blake +--- + block/mirror.c | 27 ++++++++++++++++++++++---- + blockdev.c | 18 ++++++++++------- + include/block/block_int-global-state.h | 3 ++- + qapi/block-core.json | 8 +++++++- + tests/unit/test-block-iothread.c | 2 +- + 5 files changed, 44 insertions(+), 14 deletions(-) + +diff --git a/block/mirror.c b/block/mirror.c +index c8bbaa0b35..bba3e3b05c 100644 +--- a/block/mirror.c ++++ b/block/mirror.c +@@ -55,6 +55,8 @@ typedef struct MirrorBlockJob { + BlockMirrorBackingMode backing_mode; + /* Whether the target image requires explicit zero-initialization */ + bool zero_target; ++ /* Whether the target should be assumed to be already zero initialized */ ++ bool target_is_zero; + /* + * To be accesssed with atomics. Written only under the BQL (required by the + * current implementation of mirror_change()). +@@ -844,12 +846,26 @@ static int coroutine_fn GRAPH_UNLOCKED mirror_dirty_init(MirrorBlockJob *s) + BlockDriverState *target_bs = blk_bs(s->target); + int ret = -EIO; + int64_t count; ++ bool punch_holes = ++ target_bs->detect_zeroes == BLOCKDEV_DETECT_ZEROES_OPTIONS_UNMAP && ++ bdrv_can_write_zeroes_with_unmap(target_bs); + + bdrv_graph_co_rdlock(); + bs = s->mirror_top_bs->backing->bs; + bdrv_graph_co_rdunlock(); + +- if (s->zero_target) { ++ if (s->zero_target && (!s->target_is_zero || punch_holes)) { ++ /* ++ * Here, we are in FULL mode; our goal is to avoid writing ++ * zeroes if the destination already reads as zero, except ++ * when we are trying to punch holes. This is possible if ++ * zeroing happened externally (s->target_is_zero) or if we ++ * have a fast way to pre-zero the image (the dirty bitmap ++ * will be populated later by the non-zero portions, the same ++ * as for TOP mode). If pre-zeroing is not fast, or we need ++ * to punch holes, then our only recourse is to write the ++ * entire image. ++ */ + if (!bdrv_can_write_zeroes_with_unmap(target_bs)) { + bdrv_set_dirty_bitmap(s->dirty_bitmap, 0, s->bdev_length); + return 0; +@@ -1714,7 +1730,7 @@ static BlockJob *mirror_start_job( + uint32_t granularity, int64_t buf_size, + MirrorSyncMode sync_mode, + BlockMirrorBackingMode backing_mode, +- bool zero_target, ++ bool zero_target, bool target_is_zero, + BlockdevOnError on_source_error, + BlockdevOnError on_target_error, + bool unmap, +@@ -1883,6 +1899,7 @@ static BlockJob *mirror_start_job( + s->sync_mode = sync_mode; + s->backing_mode = backing_mode; + s->zero_target = zero_target; ++ s->target_is_zero = target_is_zero; + qatomic_set(&s->copy_mode, copy_mode); + s->base = base; + s->base_overlay = bdrv_find_overlay(bs, base); +@@ -2011,7 +2028,7 @@ void mirror_start(const char *job_id, BlockDriverState *bs, + int creation_flags, int64_t speed, + uint32_t granularity, int64_t buf_size, + MirrorSyncMode mode, BlockMirrorBackingMode backing_mode, +- bool zero_target, ++ bool zero_target, bool target_is_zero, + BlockdevOnError on_source_error, + BlockdevOnError on_target_error, + bool unmap, const char *filter_node_name, +@@ -2034,7 +2051,8 @@ void mirror_start(const char *job_id, BlockDriverState *bs, + + mirror_start_job(job_id, bs, creation_flags, target, replaces, + speed, granularity, buf_size, mode, backing_mode, +- zero_target, on_source_error, on_target_error, unmap, ++ zero_target, ++ target_is_zero, on_source_error, on_target_error, unmap, + NULL, NULL, &mirror_job_driver, base, false, + filter_node_name, true, copy_mode, false, errp); + } +@@ -2062,6 +2080,7 @@ BlockJob *commit_active_start(const char *job_id, BlockDriverState *bs, + job = mirror_start_job( + job_id, bs, creation_flags, base, NULL, speed, 0, 0, + MIRROR_SYNC_MODE_TOP, MIRROR_LEAVE_BACKING_CHAIN, false, ++ false, + on_error, on_error, true, cb, opaque, + &commit_active_job_driver, base, auto_complete, + filter_node_name, false, MIRROR_COPY_MODE_BACKGROUND, +diff --git a/blockdev.c b/blockdev.c +index 70046b6690..db11a99312 100644 +--- a/blockdev.c ++++ b/blockdev.c +@@ -2795,7 +2795,7 @@ static void blockdev_mirror_common(const char *job_id, BlockDriverState *bs, + const char *replaces, + enum MirrorSyncMode sync, + BlockMirrorBackingMode backing_mode, +- bool zero_target, ++ bool zero_target, bool target_is_zero, + bool has_speed, int64_t speed, + bool has_granularity, uint32_t granularity, + bool has_buf_size, int64_t buf_size, +@@ -2906,11 +2906,10 @@ static void blockdev_mirror_common(const char *job_id, BlockDriverState *bs, + /* pass the node name to replace to mirror start since it's loose coupling + * and will allow to check whether the node still exist at mirror completion + */ +- mirror_start(job_id, bs, target, +- replaces, job_flags, ++ mirror_start(job_id, bs, target, replaces, job_flags, + speed, granularity, buf_size, sync, backing_mode, zero_target, +- on_source_error, on_target_error, unmap, filter_node_name, +- copy_mode, errp); ++ target_is_zero, on_source_error, on_target_error, unmap, ++ filter_node_name, copy_mode, errp); + } + + void qmp_drive_mirror(DriveMirror *arg, Error **errp) +@@ -2925,6 +2924,7 @@ void qmp_drive_mirror(DriveMirror *arg, Error **errp) + int64_t size; + const char *format = arg->format; + bool zero_target; ++ bool target_is_zero; + int ret; + + bs = qmp_get_root_bs(arg->device, errp); +@@ -3041,6 +3041,8 @@ void qmp_drive_mirror(DriveMirror *arg, Error **errp) + zero_target = (arg->sync == MIRROR_SYNC_MODE_FULL && + (arg->mode == NEW_IMAGE_MODE_EXISTING || + !bdrv_has_zero_init(target_bs))); ++ target_is_zero = (arg->mode != NEW_IMAGE_MODE_EXISTING && ++ bdrv_has_zero_init(target_bs)); + bdrv_graph_rdunlock_main_loop(); + + +@@ -3052,7 +3054,7 @@ void qmp_drive_mirror(DriveMirror *arg, Error **errp) + + blockdev_mirror_common(arg->job_id, bs, target_bs, + arg->replaces, arg->sync, +- backing_mode, zero_target, ++ backing_mode, zero_target, target_is_zero, + arg->has_speed, arg->speed, + arg->has_granularity, arg->granularity, + arg->has_buf_size, arg->buf_size, +@@ -3082,6 +3084,7 @@ void qmp_blockdev_mirror(const char *job_id, + bool has_copy_mode, MirrorCopyMode copy_mode, + bool has_auto_finalize, bool auto_finalize, + bool has_auto_dismiss, bool auto_dismiss, ++ bool has_target_is_zero, bool target_is_zero, + Error **errp) + { + BlockDriverState *bs; +@@ -3112,7 +3115,8 @@ void qmp_blockdev_mirror(const char *job_id, + + blockdev_mirror_common(job_id, bs, target_bs, + replaces, sync, backing_mode, +- zero_target, has_speed, speed, ++ zero_target, has_target_is_zero && target_is_zero, ++ has_speed, speed, + has_granularity, granularity, + has_buf_size, buf_size, + has_on_source_error, on_source_error, +diff --git a/include/block/block_int-global-state.h b/include/block/block_int-global-state.h +index eb2d92a226..8cf0003ce7 100644 +--- a/include/block/block_int-global-state.h ++++ b/include/block/block_int-global-state.h +@@ -140,6 +140,7 @@ BlockJob *commit_active_start(const char *job_id, BlockDriverState *bs, + * @mode: Whether to collapse all images in the chain to the target. + * @backing_mode: How to establish the target's backing chain after completion. + * @zero_target: Whether the target should be explicitly zero-initialized ++ * @target_is_zero: Whether the target already is zero-initialized. + * @on_source_error: The action to take upon error reading from the source. + * @on_target_error: The action to take upon error writing to the target. + * @unmap: Whether to unmap target where source sectors only contain zeroes. +@@ -159,7 +160,7 @@ void mirror_start(const char *job_id, BlockDriverState *bs, + int creation_flags, int64_t speed, + uint32_t granularity, int64_t buf_size, + MirrorSyncMode mode, BlockMirrorBackingMode backing_mode, +- bool zero_target, ++ bool zero_target, bool target_is_zero, + BlockdevOnError on_source_error, + BlockdevOnError on_target_error, + bool unmap, const char *filter_node_name, +diff --git a/qapi/block-core.json b/qapi/block-core.json +index c1af3d1f7d..3969c60b93 100644 +--- a/qapi/block-core.json ++++ b/qapi/block-core.json +@@ -2535,6 +2535,11 @@ + # disappear from the query list without user intervention. + # Defaults to true. (Since 3.1) + # ++# @target-is-zero: Assume the destination reads as all zeroes before ++# the mirror started. Setting this to true can speed up the ++# mirror. Setting this to true when the destination is not ++# actually all zero can corrupt the destination. (Since 10.1) ++# + # Since: 2.6 + # + # .. qmp-example:: +@@ -2554,7 +2559,8 @@ + '*on-target-error': 'BlockdevOnError', + '*filter-node-name': 'str', + '*copy-mode': 'MirrorCopyMode', +- '*auto-finalize': 'bool', '*auto-dismiss': 'bool' }, ++ '*auto-finalize': 'bool', '*auto-dismiss': 'bool', ++ '*target-is-zero': 'bool'}, + 'allow-preconfig': true } + + ## +diff --git a/tests/unit/test-block-iothread.c b/tests/unit/test-block-iothread.c +index 373b72fdd8..033711d8d7 100644 +--- a/tests/unit/test-block-iothread.c ++++ b/tests/unit/test-block-iothread.c +@@ -755,7 +755,7 @@ static void test_propagate_mirror(void) + + /* Start a mirror job */ + mirror_start("job0", src, target, NULL, JOB_DEFAULT, 0, 0, 0, +- MIRROR_SYNC_MODE_NONE, MIRROR_OPEN_BACKING_CHAIN, false, ++ MIRROR_SYNC_MODE_NONE, MIRROR_OPEN_BACKING_CHAIN, false, false, + BLOCKDEV_ON_ERROR_REPORT, BLOCKDEV_ON_ERROR_REPORT, + false, "filter_node", MIRROR_COPY_MODE_BACKGROUND, + &error_abort); +-- +2.48.1 + diff --git a/SOURCES/kvm-mirror-Drop-redundant-zero_target-parameter.patch b/SOURCES/kvm-mirror-Drop-redundant-zero_target-parameter.patch new file mode 100644 index 0000000..ee01f81 --- /dev/null +++ b/SOURCES/kvm-mirror-Drop-redundant-zero_target-parameter.patch @@ -0,0 +1,241 @@ +From 5040f835f07f3355ae80b3da2ae83ce35de022e0 Mon Sep 17 00:00:00 2001 +From: Eric Blake +Date: Fri, 9 May 2025 15:40:26 -0500 +Subject: [PATCH 11/16] mirror: Drop redundant zero_target parameter + +RH-Author: Eric Blake +RH-MergeRequest: 365: blockdev-mirror: More efficient handling of sparse mirrors +RH-Jira: RHEL-82906 RHEL-83015 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Jon Maloy +RH-Commit: [9/14] b84a938c69e3761211b9fee4c59b465d55f61855 (ebblake/centos-qemu-kvm) + +The two callers to a mirror job (drive-mirror and blockdev-mirror) set +zero_target precisely when sync mode == FULL, with the one exception +that drive-mirror skips zeroing the target if it was newly created and +reads as zero. But given the previous patch, that exception is +equally captured by target_is_zero. + +Meanwhile, there is another slight wrinkle, fortunately caught by +iotest 185: if the caller uses "sync":"top" but the source has no +backing file, the code in blockdev.c was changing sync to be FULL, but +only after it had set zero_target=false. In mirror.c, prior to recent +patches, this didn't matter: the only places that inspected sync were +setting is_none_mode (both TOP and FULL had set that to false), and +mirror_start() setting base = mode == MIRROR_SYNC_MODE_TOP ? +bdrv_backing_chain_next(bs) : NULL. But now that we are passing sync +around, the slammed sync mode would result in a new pre-zeroing pass +even when the user had passed "sync":"top" in an effort to skip +pre-zeroing. Fortunately, the assignment of base when bs has no +backing chain still works out to NULL if we don't slam things. So +with the forced change of sync ripped out of blockdev.c, the sync mode +is passed through the full callstack unmolested, and we can now +reliably reconstruct the same settings as what used to be passed in by +zero_target=false, without the redundant parameter. + +Signed-off-by: Eric Blake +Message-ID: <20250509204341.3553601-24-eblake@redhat.com> +Reviewed-by: Sunny Zhu +Reviewed-by: Stefan Hajnoczi +[eblake: Fix regression in iotest 185] +Signed-off-by: Eric Blake +(cherry picked from commit 253b43a29077de9266351e120c600a73b82e9c49) +Jira: https://issues.redhat.com/browse/RHEL-82906 +Jira: https://issues.redhat.com/browse/RHEL-83015 +Signed-off-by: Eric Blake +--- + block/mirror.c | 13 +++++-------- + blockdev.c | 19 ++++--------------- + include/block/block_int-global-state.h | 3 +-- + tests/unit/test-block-iothread.c | 2 +- + 4 files changed, 11 insertions(+), 26 deletions(-) + +diff --git a/block/mirror.c b/block/mirror.c +index bba3e3b05c..b35d12adaa 100644 +--- a/block/mirror.c ++++ b/block/mirror.c +@@ -53,8 +53,6 @@ typedef struct MirrorBlockJob { + Error *replace_blocker; + MirrorSyncMode sync_mode; + BlockMirrorBackingMode backing_mode; +- /* Whether the target image requires explicit zero-initialization */ +- bool zero_target; + /* Whether the target should be assumed to be already zero initialized */ + bool target_is_zero; + /* +@@ -854,7 +852,9 @@ static int coroutine_fn GRAPH_UNLOCKED mirror_dirty_init(MirrorBlockJob *s) + bs = s->mirror_top_bs->backing->bs; + bdrv_graph_co_rdunlock(); + +- if (s->zero_target && (!s->target_is_zero || punch_holes)) { ++ if (s->sync_mode == MIRROR_SYNC_MODE_TOP) { ++ /* In TOP mode, there is no benefit to a pre-zeroing pass. */ ++ } else if (!s->target_is_zero || punch_holes) { + /* + * Here, we are in FULL mode; our goal is to avoid writing + * zeroes if the destination already reads as zero, except +@@ -1730,7 +1730,7 @@ static BlockJob *mirror_start_job( + uint32_t granularity, int64_t buf_size, + MirrorSyncMode sync_mode, + BlockMirrorBackingMode backing_mode, +- bool zero_target, bool target_is_zero, ++ bool target_is_zero, + BlockdevOnError on_source_error, + BlockdevOnError on_target_error, + bool unmap, +@@ -1898,7 +1898,6 @@ static BlockJob *mirror_start_job( + s->on_target_error = on_target_error; + s->sync_mode = sync_mode; + s->backing_mode = backing_mode; +- s->zero_target = zero_target; + s->target_is_zero = target_is_zero; + qatomic_set(&s->copy_mode, copy_mode); + s->base = base; +@@ -2028,7 +2027,7 @@ void mirror_start(const char *job_id, BlockDriverState *bs, + int creation_flags, int64_t speed, + uint32_t granularity, int64_t buf_size, + MirrorSyncMode mode, BlockMirrorBackingMode backing_mode, +- bool zero_target, bool target_is_zero, ++ bool target_is_zero, + BlockdevOnError on_source_error, + BlockdevOnError on_target_error, + bool unmap, const char *filter_node_name, +@@ -2051,7 +2050,6 @@ void mirror_start(const char *job_id, BlockDriverState *bs, + + mirror_start_job(job_id, bs, creation_flags, target, replaces, + speed, granularity, buf_size, mode, backing_mode, +- zero_target, + target_is_zero, on_source_error, on_target_error, unmap, + NULL, NULL, &mirror_job_driver, base, false, + filter_node_name, true, copy_mode, false, errp); +@@ -2080,7 +2078,6 @@ BlockJob *commit_active_start(const char *job_id, BlockDriverState *bs, + job = mirror_start_job( + job_id, bs, creation_flags, base, NULL, speed, 0, 0, + MIRROR_SYNC_MODE_TOP, MIRROR_LEAVE_BACKING_CHAIN, false, +- false, + on_error, on_error, true, cb, opaque, + &commit_active_job_driver, base, auto_complete, + filter_node_name, false, MIRROR_COPY_MODE_BACKGROUND, +diff --git a/blockdev.c b/blockdev.c +index db11a99312..04fa759e30 100644 +--- a/blockdev.c ++++ b/blockdev.c +@@ -2795,7 +2795,7 @@ static void blockdev_mirror_common(const char *job_id, BlockDriverState *bs, + const char *replaces, + enum MirrorSyncMode sync, + BlockMirrorBackingMode backing_mode, +- bool zero_target, bool target_is_zero, ++ bool target_is_zero, + bool has_speed, int64_t speed, + bool has_granularity, uint32_t granularity, + bool has_buf_size, int64_t buf_size, +@@ -2862,10 +2862,6 @@ static void blockdev_mirror_common(const char *job_id, BlockDriverState *bs, + return; + } + +- if (!bdrv_backing_chain_next(bs) && sync == MIRROR_SYNC_MODE_TOP) { +- sync = MIRROR_SYNC_MODE_FULL; +- } +- + if (!replaces) { + /* We want to mirror from @bs, but keep implicit filters on top */ + unfiltered_bs = bdrv_skip_implicit_filters(bs); +@@ -2907,7 +2903,7 @@ static void blockdev_mirror_common(const char *job_id, BlockDriverState *bs, + * and will allow to check whether the node still exist at mirror completion + */ + mirror_start(job_id, bs, target, replaces, job_flags, +- speed, granularity, buf_size, sync, backing_mode, zero_target, ++ speed, granularity, buf_size, sync, backing_mode, + target_is_zero, on_source_error, on_target_error, unmap, + filter_node_name, copy_mode, errp); + } +@@ -2923,7 +2919,6 @@ void qmp_drive_mirror(DriveMirror *arg, Error **errp) + int flags; + int64_t size; + const char *format = arg->format; +- bool zero_target; + bool target_is_zero; + int ret; + +@@ -3038,9 +3033,6 @@ void qmp_drive_mirror(DriveMirror *arg, Error **errp) + } + + bdrv_graph_rdlock_main_loop(); +- zero_target = (arg->sync == MIRROR_SYNC_MODE_FULL && +- (arg->mode == NEW_IMAGE_MODE_EXISTING || +- !bdrv_has_zero_init(target_bs))); + target_is_zero = (arg->mode != NEW_IMAGE_MODE_EXISTING && + bdrv_has_zero_init(target_bs)); + bdrv_graph_rdunlock_main_loop(); +@@ -3054,7 +3046,7 @@ void qmp_drive_mirror(DriveMirror *arg, Error **errp) + + blockdev_mirror_common(arg->job_id, bs, target_bs, + arg->replaces, arg->sync, +- backing_mode, zero_target, target_is_zero, ++ backing_mode, target_is_zero, + arg->has_speed, arg->speed, + arg->has_granularity, arg->granularity, + arg->has_buf_size, arg->buf_size, +@@ -3091,7 +3083,6 @@ void qmp_blockdev_mirror(const char *job_id, + BlockDriverState *target_bs; + AioContext *aio_context; + BlockMirrorBackingMode backing_mode = MIRROR_LEAVE_BACKING_CHAIN; +- bool zero_target; + int ret; + + bs = qmp_get_root_bs(device, errp); +@@ -3104,8 +3095,6 @@ void qmp_blockdev_mirror(const char *job_id, + return; + } + +- zero_target = (sync == MIRROR_SYNC_MODE_FULL); +- + aio_context = bdrv_get_aio_context(bs); + + ret = bdrv_try_change_aio_context(target_bs, aio_context, NULL, errp); +@@ -3115,7 +3104,7 @@ void qmp_blockdev_mirror(const char *job_id, + + blockdev_mirror_common(job_id, bs, target_bs, + replaces, sync, backing_mode, +- zero_target, has_target_is_zero && target_is_zero, ++ has_target_is_zero && target_is_zero, + has_speed, speed, + has_granularity, granularity, + has_buf_size, buf_size, +diff --git a/include/block/block_int-global-state.h b/include/block/block_int-global-state.h +index 8cf0003ce7..d21bd7fd2f 100644 +--- a/include/block/block_int-global-state.h ++++ b/include/block/block_int-global-state.h +@@ -139,7 +139,6 @@ BlockJob *commit_active_start(const char *job_id, BlockDriverState *bs, + * @buf_size: The amount of data that can be in flight at one time. + * @mode: Whether to collapse all images in the chain to the target. + * @backing_mode: How to establish the target's backing chain after completion. +- * @zero_target: Whether the target should be explicitly zero-initialized + * @target_is_zero: Whether the target already is zero-initialized. + * @on_source_error: The action to take upon error reading from the source. + * @on_target_error: The action to take upon error writing to the target. +@@ -160,7 +159,7 @@ void mirror_start(const char *job_id, BlockDriverState *bs, + int creation_flags, int64_t speed, + uint32_t granularity, int64_t buf_size, + MirrorSyncMode mode, BlockMirrorBackingMode backing_mode, +- bool zero_target, bool target_is_zero, ++ bool target_is_zero, + BlockdevOnError on_source_error, + BlockdevOnError on_target_error, + bool unmap, const char *filter_node_name, +diff --git a/tests/unit/test-block-iothread.c b/tests/unit/test-block-iothread.c +index 033711d8d7..373b72fdd8 100644 +--- a/tests/unit/test-block-iothread.c ++++ b/tests/unit/test-block-iothread.c +@@ -755,7 +755,7 @@ static void test_propagate_mirror(void) + + /* Start a mirror job */ + mirror_start("job0", src, target, NULL, JOB_DEFAULT, 0, 0, 0, +- MIRROR_SYNC_MODE_NONE, MIRROR_OPEN_BACKING_CHAIN, false, false, ++ MIRROR_SYNC_MODE_NONE, MIRROR_OPEN_BACKING_CHAIN, false, + BLOCKDEV_ON_ERROR_REPORT, BLOCKDEV_ON_ERROR_REPORT, + false, "filter_node", MIRROR_COPY_MODE_BACKGROUND, + &error_abort); +-- +2.48.1 + diff --git a/SOURCES/kvm-mirror-Minor-refactoring.patch b/SOURCES/kvm-mirror-Minor-refactoring.patch new file mode 100644 index 0000000..eda26ee --- /dev/null +++ b/SOURCES/kvm-mirror-Minor-refactoring.patch @@ -0,0 +1,92 @@ +From 0102da22fe5aefde9d398d539fc290ab062346f1 Mon Sep 17 00:00:00 2001 +From: Eric Blake +Date: Fri, 9 May 2025 15:40:23 -0500 +Subject: [PATCH 08/16] mirror: Minor refactoring + +RH-Author: Eric Blake +RH-MergeRequest: 365: blockdev-mirror: More efficient handling of sparse mirrors +RH-Jira: RHEL-82906 RHEL-83015 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Jon Maloy +RH-Commit: [6/14] 886fa2e3249f48f89d3e04ba619d370031851d89 (ebblake/centos-qemu-kvm) + +Commit 5791ba52 (v9.2) pre-initialized ret in mirror_dirty_init to +silence a false positive compiler warning, even though in all code +paths where ret is used, it was guaranteed to be reassigned +beforehand. But since the function returns -errno, and -1 is not +always the right errno, it's better to initialize to -EIO. + +An upcoming patch wants to track two bitmaps in +do_sync_target_write(); this will be easier if the current variables +related to the dirty bitmap are renamed. + +Signed-off-by: Eric Blake +Reviewed-by: Stefan Hajnoczi +Message-ID: <20250509204341.3553601-21-eblake@redhat.com> +(cherry picked from commit 870f8963cf1a84f8ec929b05a6d68906974a76c5) +Conflicts: + block/mirror.c - commit 5791ba52 not present +Jira: https://issues.redhat.com/browse/RHEL-82906 +Jira: https://issues.redhat.com/browse/RHEL-83015 +Signed-off-by: Eric Blake +--- + block/mirror.c | 22 +++++++++++----------- + 1 file changed, 11 insertions(+), 11 deletions(-) + +diff --git a/block/mirror.c b/block/mirror.c +index 61f0a717b7..22f8bd98c4 100644 +--- a/block/mirror.c ++++ b/block/mirror.c +@@ -841,7 +841,7 @@ static int coroutine_fn GRAPH_UNLOCKED mirror_dirty_init(MirrorBlockJob *s) + int64_t offset; + BlockDriverState *bs; + BlockDriverState *target_bs = blk_bs(s->target); +- int ret; ++ int ret = -EIO; + int64_t count; + + bdrv_graph_co_rdlock(); +@@ -1341,7 +1341,7 @@ do_sync_target_write(MirrorBlockJob *job, MirrorMethod method, + { + int ret; + size_t qiov_offset = 0; +- int64_t bitmap_offset, bitmap_end; ++ int64_t dirty_bitmap_offset, dirty_bitmap_end; + + if (!QEMU_IS_ALIGNED(offset, job->granularity) && + bdrv_dirty_bitmap_get(job->dirty_bitmap, offset)) +@@ -1388,11 +1388,11 @@ do_sync_target_write(MirrorBlockJob *job, MirrorMethod method, + * Tails are either clean or shrunk, so for bitmap resetting + * we safely align the range down. + */ +- bitmap_offset = QEMU_ALIGN_UP(offset, job->granularity); +- bitmap_end = QEMU_ALIGN_DOWN(offset + bytes, job->granularity); +- if (bitmap_offset < bitmap_end) { +- bdrv_reset_dirty_bitmap(job->dirty_bitmap, bitmap_offset, +- bitmap_end - bitmap_offset); ++ dirty_bitmap_offset = QEMU_ALIGN_UP(offset, job->granularity); ++ dirty_bitmap_end = QEMU_ALIGN_DOWN(offset + bytes, job->granularity); ++ if (dirty_bitmap_offset < dirty_bitmap_end) { ++ bdrv_reset_dirty_bitmap(job->dirty_bitmap, dirty_bitmap_offset, ++ dirty_bitmap_end - dirty_bitmap_offset); + } + + job_progress_increase_remaining(&job->common.job, bytes); +@@ -1430,10 +1430,10 @@ do_sync_target_write(MirrorBlockJob *job, MirrorMethod method, + * at function start, and they must be still dirty, as we've locked + * the region for in-flight op. + */ +- bitmap_offset = QEMU_ALIGN_DOWN(offset, job->granularity); +- bitmap_end = QEMU_ALIGN_UP(offset + bytes, job->granularity); +- bdrv_set_dirty_bitmap(job->dirty_bitmap, bitmap_offset, +- bitmap_end - bitmap_offset); ++ dirty_bitmap_offset = QEMU_ALIGN_DOWN(offset, job->granularity); ++ dirty_bitmap_end = QEMU_ALIGN_UP(offset + bytes, job->granularity); ++ bdrv_set_dirty_bitmap(job->dirty_bitmap, dirty_bitmap_offset, ++ dirty_bitmap_end - dirty_bitmap_offset); + qatomic_set(&job->actively_synced, false); + + action = mirror_error_action(job, false, -ret); +-- +2.48.1 + diff --git a/SOURCES/kvm-mirror-Pass-full-sync-mode-rather-than-bool-to-inter.patch b/SOURCES/kvm-mirror-Pass-full-sync-mode-rather-than-bool-to-inter.patch new file mode 100644 index 0000000..653bb20 --- /dev/null +++ b/SOURCES/kvm-mirror-Pass-full-sync-mode-rather-than-bool-to-inter.patch @@ -0,0 +1,139 @@ +From 482db3e637a16d5877e523e87c53ddb2579b4b66 Mon Sep 17 00:00:00 2001 +From: Eric Blake +Date: Fri, 9 May 2025 15:40:24 -0500 +Subject: [PATCH 09/16] mirror: Pass full sync mode rather than bool to + internals + +RH-Author: Eric Blake +RH-MergeRequest: 365: blockdev-mirror: More efficient handling of sparse mirrors +RH-Jira: RHEL-82906 RHEL-83015 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Jon Maloy +RH-Commit: [7/14] f45a83a14b0eea07517176d44ab0c49db8233ea0 (ebblake/centos-qemu-kvm) + +Out of the five possible values for MirrorSyncMode, INCREMENTAL and +BITMAP are already rejected up front in mirror_start, leaving NONE, +TOP, and FULL as the remaining values that the code was collapsing +into a single bool is_none_mode. Furthermore, mirror_dirty_init() is +only reachable for modes TOP and FULL, as further guided by +s->zero_target. However, upcoming patches want to further optimize +the pre-zeroing pass of a sync=full mirror in mirror_dirty_init(), +while avoiding that pass on a sync=top action. Instead of throwing +away context by collapsing these two values into +s->is_none_mode=false, it is better to pass s->sync_mode throughout +the entire operation. For active commit, the desired semantics match +sync mode TOP. + +Signed-off-by: Eric Blake +Message-ID: <20250509204341.3553601-22-eblake@redhat.com> +Reviewed-by: Sunny Zhu +Reviewed-by: Stefan Hajnoczi +(cherry picked from commit 9474d97bd7421b4fe7c806ab0949697514d11e88) +Jira: https://issues.redhat.com/browse/RHEL-82906 +Jira: https://issues.redhat.com/browse/RHEL-83015 +Signed-off-by: Eric Blake +--- + block/mirror.c | 24 ++++++++++++------------ + 1 file changed, 12 insertions(+), 12 deletions(-) + +diff --git a/block/mirror.c b/block/mirror.c +index 22f8bd98c4..c8bbaa0b35 100644 +--- a/block/mirror.c ++++ b/block/mirror.c +@@ -51,7 +51,7 @@ typedef struct MirrorBlockJob { + BlockDriverState *to_replace; + /* Used to block operations on the drive-mirror-replace target */ + Error *replace_blocker; +- bool is_none_mode; ++ MirrorSyncMode sync_mode; + BlockMirrorBackingMode backing_mode; + /* Whether the target image requires explicit zero-initialization */ + bool zero_target; +@@ -723,9 +723,10 @@ static int mirror_exit_common(Job *job) + &error_abort); + + if (!abort && s->backing_mode == MIRROR_SOURCE_BACKING_CHAIN) { +- BlockDriverState *backing = s->is_none_mode ? src : s->base; ++ BlockDriverState *backing; + BlockDriverState *unfiltered_target = bdrv_skip_filters(target_bs); + ++ backing = s->sync_mode == MIRROR_SYNC_MODE_NONE ? src : s->base; + if (bdrv_cow_bs(unfiltered_target) != backing) { + bdrv_set_backing_hd(unfiltered_target, backing, &local_err); + if (local_err) { +@@ -1020,7 +1021,7 @@ static int coroutine_fn mirror_run(Job *job, Error **errp) + mirror_free_init(s); + + s->last_pause_ns = qemu_clock_get_ns(QEMU_CLOCK_REALTIME); +- if (!s->is_none_mode) { ++ if (s->sync_mode != MIRROR_SYNC_MODE_NONE) { + ret = mirror_dirty_init(s); + if (ret < 0 || job_is_cancelled(&s->common.job)) { + goto immediate_exit; +@@ -1711,6 +1712,7 @@ static BlockJob *mirror_start_job( + int creation_flags, BlockDriverState *target, + const char *replaces, int64_t speed, + uint32_t granularity, int64_t buf_size, ++ MirrorSyncMode sync_mode, + BlockMirrorBackingMode backing_mode, + bool zero_target, + BlockdevOnError on_source_error, +@@ -1719,7 +1721,7 @@ static BlockJob *mirror_start_job( + BlockCompletionFunc *cb, + void *opaque, + const BlockJobDriver *driver, +- bool is_none_mode, BlockDriverState *base, ++ BlockDriverState *base, + bool auto_complete, const char *filter_node_name, + bool is_mirror, MirrorCopyMode copy_mode, + bool base_ro, +@@ -1878,7 +1880,7 @@ static BlockJob *mirror_start_job( + s->replaces = g_strdup(replaces); + s->on_source_error = on_source_error; + s->on_target_error = on_target_error; +- s->is_none_mode = is_none_mode; ++ s->sync_mode = sync_mode; + s->backing_mode = backing_mode; + s->zero_target = zero_target; + qatomic_set(&s->copy_mode, copy_mode); +@@ -2015,7 +2017,6 @@ void mirror_start(const char *job_id, BlockDriverState *bs, + bool unmap, const char *filter_node_name, + MirrorCopyMode copy_mode, Error **errp) + { +- bool is_none_mode; + BlockDriverState *base; + + GLOBAL_STATE_CODE(); +@@ -2028,14 +2029,13 @@ void mirror_start(const char *job_id, BlockDriverState *bs, + } + + bdrv_graph_rdlock_main_loop(); +- is_none_mode = mode == MIRROR_SYNC_MODE_NONE; + base = mode == MIRROR_SYNC_MODE_TOP ? bdrv_backing_chain_next(bs) : NULL; + bdrv_graph_rdunlock_main_loop(); + + mirror_start_job(job_id, bs, creation_flags, target, replaces, +- speed, granularity, buf_size, backing_mode, zero_target, +- on_source_error, on_target_error, unmap, NULL, NULL, +- &mirror_job_driver, is_none_mode, base, false, ++ speed, granularity, buf_size, mode, backing_mode, ++ zero_target, on_source_error, on_target_error, unmap, ++ NULL, NULL, &mirror_job_driver, base, false, + filter_node_name, true, copy_mode, false, errp); + } + +@@ -2061,9 +2061,9 @@ BlockJob *commit_active_start(const char *job_id, BlockDriverState *bs, + + job = mirror_start_job( + job_id, bs, creation_flags, base, NULL, speed, 0, 0, +- MIRROR_LEAVE_BACKING_CHAIN, false, ++ MIRROR_SYNC_MODE_TOP, MIRROR_LEAVE_BACKING_CHAIN, false, + on_error, on_error, true, cb, opaque, +- &commit_active_job_driver, false, base, auto_complete, ++ &commit_active_job_driver, base, auto_complete, + filter_node_name, false, MIRROR_COPY_MODE_BACKGROUND, + base_read_only, errp); + if (!job) { +-- +2.48.1 + diff --git a/SOURCES/kvm-mirror-Reduce-I-O-when-destination-is-detect-zeroes-.patch b/SOURCES/kvm-mirror-Reduce-I-O-when-destination-is-detect-zeroes-.patch new file mode 100644 index 0000000..3864b3d --- /dev/null +++ b/SOURCES/kvm-mirror-Reduce-I-O-when-destination-is-detect-zeroes-.patch @@ -0,0 +1,58 @@ +From be6ce2c91fe949d1c264de974ab4f6c4efc6976e Mon Sep 17 00:00:00 2001 +From: Eric Blake +Date: Tue, 13 May 2025 17:00:45 -0500 +Subject: [PATCH 16/16] mirror: Reduce I/O when destination is + detect-zeroes:unmap + +RH-Author: Eric Blake +RH-MergeRequest: 365: blockdev-mirror: More efficient handling of sparse mirrors +RH-Jira: RHEL-82906 RHEL-83015 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Jon Maloy +RH-Commit: [14/14] 66f3de2ba9f977c9bc1c54f67d76b366df132e62 (ebblake/centos-qemu-kvm) + +If we are going to punch holes in the mirror destination even for the +portions where the source image is unallocated, it is nicer to treat +the entire image as dirty and punch as we go, rather than pre-zeroing +the entire image just to re-do I/O to the allocated portions of the +image. + +Signed-off-by: Eric Blake +Message-ID: <20250513220142.535200-2-eblake@redhat.com> +Reviewed-by: Stefan Hajnoczi +(cherry picked from commit 9abfc81246c9cc1845080eec5920779961187c07) +Jira: https://issues.redhat.com/browse/RHEL-82906 +Jira: https://issues.redhat.com/browse/RHEL-83015 +Signed-off-by: Eric Blake +--- + block/mirror.c | 13 +++++++++---- + 1 file changed, 9 insertions(+), 4 deletions(-) + +diff --git a/block/mirror.c b/block/mirror.c +index 7f3b5477ce..87c19ddf0d 100644 +--- a/block/mirror.c ++++ b/block/mirror.c +@@ -920,11 +920,16 @@ static int coroutine_fn GRAPH_UNLOCKED mirror_dirty_init(MirrorBlockJob *s) + * zeroing happened externally (ret > 0) or if we have a fast + * way to pre-zero the image (the dirty bitmap will be + * populated later by the non-zero portions, the same as for +- * TOP mode). If pre-zeroing is not fast, then our only +- * recourse is to mark the entire image dirty. The act of +- * pre-zeroing will populate the zero bitmap. ++ * TOP mode). If pre-zeroing is not fast, or we need to visit ++ * the entire image in order to punch holes even in the ++ * non-allocated regions of the source, then just mark the ++ * entire image dirty and leave the zero bitmap clear at this ++ * point in time. Otherwise, it can be faster to pre-zero the ++ * image now, even if we re-write the allocated portions of ++ * the disk later, and the pre-zero pass will populate the ++ * zero bitmap. + */ +- if (!bdrv_can_write_zeroes_with_unmap(target_bs)) { ++ if (!bdrv_can_write_zeroes_with_unmap(target_bs) || punch_holes) { + bdrv_set_dirty_bitmap(s->dirty_bitmap, 0, s->bdev_length); + return 0; + } +-- +2.48.1 + diff --git a/SOURCES/kvm-mirror-Skip-pre-zeroing-destination-if-it-is-already.patch b/SOURCES/kvm-mirror-Skip-pre-zeroing-destination-if-it-is-already.patch new file mode 100644 index 0000000..ea9ad0b --- /dev/null +++ b/SOURCES/kvm-mirror-Skip-pre-zeroing-destination-if-it-is-already.patch @@ -0,0 +1,180 @@ +From 423ce7727eecae647330287e1264ac0d938fa7f9 Mon Sep 17 00:00:00 2001 +From: Eric Blake +Date: Fri, 9 May 2025 15:40:27 -0500 +Subject: [PATCH 12/16] mirror: Skip pre-zeroing destination if it is already + zero + +RH-Author: Eric Blake +RH-MergeRequest: 365: blockdev-mirror: More efficient handling of sparse mirrors +RH-Jira: RHEL-82906 RHEL-83015 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Jon Maloy +RH-Commit: [10/14] e754ae559123099f4aed322f6a4287cf3323f54d (ebblake/centos-qemu-kvm) + +When doing a sync=full mirroring, we can skip pre-zeroing the +destination if it already reads as zeroes and we are not also trying +to punch holes due to detect-zeroes. With this patch, there are fewer +scenarios that have to pass in an explicit target-is-zero, while still +resulting in a sparse destination remaining sparse. + +A later patch will then further improve things to skip writing to the +destination for parts of the image where the source is zero; but even +with just this patch, it is possible to see a difference for any +source that does not report itself as fully allocated, coupled with a +destination BDS that can quickly report that it already reads as zero. +(For a source that reports as fully allocated, such as a file, the +rest of mirror_dirty_init() still sets the entire dirty bitmap to +true, so even though we avoided the pre-zeroing, we are not yet +avoiding all redundant I/O). + +Iotest 194 detects the difference made by this patch: for a file +source (where block status reports the entire image as allocated, and +therefore we end up writing zeroes everywhere in the destination +anyways), the job length remains the same. But for a qcow2 source and +a destination that reads as all zeroes, the dirty bitmap changes to +just tracking the allocated portions of the source, which results in +faster completion and smaller job statistics. For the test to pass +with both ./check -file and -qcow2, a new python filter is needed to +mask out the now-varying job amounts (this matches the shell filters +_filter_block_job_{offset,len} in common.filter). A later test will +also be added which further validates expected sparseness, so it does +not matter that 194 is no longer explicitly looking at how many bytes +were copied. + +Signed-off-by: Eric Blake +Message-ID: <20250509204341.3553601-25-eblake@redhat.com> +Reviewed-by: Sunny Zhu +Reviewed-by: Stefan Hajnoczi +(cherry picked from commit 181a63667adf16c35b57e446def3e41c70f1fea6) +Jira: https://issues.redhat.com/browse/RHEL-82906 +Jira: https://issues.redhat.com/browse/RHEL-83015 +Signed-off-by: Eric Blake +--- + block/mirror.c | 24 ++++++++++++++++-------- + tests/qemu-iotests/194 | 6 ++++-- + tests/qemu-iotests/194.out | 4 ++-- + tests/qemu-iotests/iotests.py | 12 +++++++++++- + 4 files changed, 33 insertions(+), 13 deletions(-) + +diff --git a/block/mirror.c b/block/mirror.c +index b35d12adaa..29cac1777c 100644 +--- a/block/mirror.c ++++ b/block/mirror.c +@@ -848,23 +848,31 @@ static int coroutine_fn GRAPH_UNLOCKED mirror_dirty_init(MirrorBlockJob *s) + target_bs->detect_zeroes == BLOCKDEV_DETECT_ZEROES_OPTIONS_UNMAP && + bdrv_can_write_zeroes_with_unmap(target_bs); + ++ /* Determine if the image is already zero, regardless of sync mode. */ + bdrv_graph_co_rdlock(); + bs = s->mirror_top_bs->backing->bs; ++ if (s->target_is_zero) { ++ ret = 1; ++ } else { ++ ret = bdrv_co_is_all_zeroes(target_bs); ++ } + bdrv_graph_co_rdunlock(); + +- if (s->sync_mode == MIRROR_SYNC_MODE_TOP) { ++ /* Determine if a pre-zeroing pass is necessary. */ ++ if (ret < 0) { ++ return ret; ++ } else if (s->sync_mode == MIRROR_SYNC_MODE_TOP) { + /* In TOP mode, there is no benefit to a pre-zeroing pass. */ +- } else if (!s->target_is_zero || punch_holes) { ++ } else if (ret == 0 || punch_holes) { + /* + * Here, we are in FULL mode; our goal is to avoid writing + * zeroes if the destination already reads as zero, except + * when we are trying to punch holes. This is possible if +- * zeroing happened externally (s->target_is_zero) or if we +- * have a fast way to pre-zero the image (the dirty bitmap +- * will be populated later by the non-zero portions, the same +- * as for TOP mode). If pre-zeroing is not fast, or we need +- * to punch holes, then our only recourse is to write the +- * entire image. ++ * zeroing happened externally (ret > 0) or if we have a fast ++ * way to pre-zero the image (the dirty bitmap will be ++ * populated later by the non-zero portions, the same as for ++ * TOP mode). If pre-zeroing is not fast, or we need to punch ++ * holes, then our only recourse is to write the entire image. + */ + if (!bdrv_can_write_zeroes_with_unmap(target_bs)) { + bdrv_set_dirty_bitmap(s->dirty_bitmap, 0, s->bdev_length); +diff --git a/tests/qemu-iotests/194 b/tests/qemu-iotests/194 +index d0b9c084f5..e114c0b269 100755 +--- a/tests/qemu-iotests/194 ++++ b/tests/qemu-iotests/194 +@@ -62,7 +62,8 @@ with iotests.FilePath('source.img') as source_img_path, \ + + iotests.log('Waiting for `drive-mirror` to complete...') + iotests.log(source_vm.event_wait('BLOCK_JOB_READY'), +- filters=[iotests.filter_qmp_event]) ++ filters=[iotests.filter_qmp_event, ++ iotests.filter_block_job]) + + iotests.log('Starting migration...') + capabilities = [{'capability': 'events', 'state': True}, +@@ -88,7 +89,8 @@ with iotests.FilePath('source.img') as source_img_path, \ + + while True: + event2 = source_vm.event_wait('BLOCK_JOB_COMPLETED') +- iotests.log(event2, filters=[iotests.filter_qmp_event]) ++ iotests.log(event2, filters=[iotests.filter_qmp_event, ++ iotests.filter_block_job]) + if event2['event'] == 'BLOCK_JOB_COMPLETED': + iotests.log('Stopping the NBD server on destination...') + iotests.log(dest_vm.qmp('nbd-server-stop')) +diff --git a/tests/qemu-iotests/194.out b/tests/qemu-iotests/194.out +index 376ed1d2e6..84e0fc34be 100644 +--- a/tests/qemu-iotests/194.out ++++ b/tests/qemu-iotests/194.out +@@ -7,7 +7,7 @@ Launching NBD server on destination... + Starting `drive-mirror` on source... + {"return": {}} + Waiting for `drive-mirror` to complete... +-{"data": {"device": "mirror-job0", "len": 1073741824, "offset": 1073741824, "speed": 0, "type": "mirror"}, "event": "BLOCK_JOB_READY", "timestamp": {"microseconds": "USECS", "seconds": "SECS"}} ++{"data": {"device": "mirror-job0", "len": "LEN", "offset": "OFFSET", "speed": 0, "type": "mirror"}, "event": "BLOCK_JOB_READY", "timestamp": {"microseconds": "USECS", "seconds": "SECS"}} + Starting migration... + {"return": {}} + {"execute": "migrate-start-postcopy", "arguments": {}} +@@ -17,7 +17,7 @@ Starting migration... + {"data": {"status": "completed"}, "event": "MIGRATION", "timestamp": {"microseconds": "USECS", "seconds": "SECS"}} + Gracefully ending the `drive-mirror` job on source... + {"return": {}} +-{"data": {"device": "mirror-job0", "len": 1073741824, "offset": 1073741824, "speed": 0, "type": "mirror"}, "event": "BLOCK_JOB_COMPLETED", "timestamp": {"microseconds": "USECS", "seconds": "SECS"}} ++{"data": {"device": "mirror-job0", "len": "LEN", "offset": "OFFSET", "speed": 0, "type": "mirror"}, "event": "BLOCK_JOB_COMPLETED", "timestamp": {"microseconds": "USECS", "seconds": "SECS"}} + Stopping the NBD server on destination... + {"return": {}} + Wait for migration completion on target... +diff --git a/tests/qemu-iotests/iotests.py b/tests/qemu-iotests/iotests.py +index c8cb028c2d..978bef1499 100644 +--- a/tests/qemu-iotests/iotests.py ++++ b/tests/qemu-iotests/iotests.py +@@ -601,13 +601,23 @@ def filter_chown(msg): + return chown_re.sub("chown UID:GID", msg) + + def filter_qmp_event(event): +- '''Filter a QMP event dict''' ++ '''Filter the timestamp of a QMP event dict''' + event = dict(event) + if 'timestamp' in event: + event['timestamp']['seconds'] = 'SECS' + event['timestamp']['microseconds'] = 'USECS' + return event + ++def filter_block_job(event): ++ '''Filter the offset and length of a QMP block job event dict''' ++ event = dict(event) ++ if 'data' in event: ++ if 'offset' in event['data']: ++ event['data']['offset'] = 'OFFSET' ++ if 'len' in event['data']: ++ event['data']['len'] = 'LEN' ++ return event ++ + def filter_qmp(qmsg, filter_fn): + '''Given a string filter, filter a QMP object's values. + filter_fn takes a (key, value) pair.''' +-- +2.48.1 + diff --git a/SOURCES/kvm-mirror-Skip-writing-zeroes-when-target-is-already-ze.patch b/SOURCES/kvm-mirror-Skip-writing-zeroes-when-target-is-already-ze.patch new file mode 100644 index 0000000..4da52ce --- /dev/null +++ b/SOURCES/kvm-mirror-Skip-writing-zeroes-when-target-is-already-ze.patch @@ -0,0 +1,355 @@ +From 8a2e660ff3ec7f7506fbd4197d4dc8f53db7859a Mon Sep 17 00:00:00 2001 +From: Eric Blake +Date: Fri, 9 May 2025 15:40:28 -0500 +Subject: [PATCH 13/16] mirror: Skip writing zeroes when target is already zero + +RH-Author: Eric Blake +RH-MergeRequest: 365: blockdev-mirror: More efficient handling of sparse mirrors +RH-Jira: RHEL-82906 RHEL-83015 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Jon Maloy +RH-Commit: [11/14] f6bb5e0cecee07af0389aa18c3bddb47d6c5cf54 (ebblake/centos-qemu-kvm) + +When mirroring, the goal is to ensure that the destination reads the +same as the source; this goal is met whether the destination is sparse +or fully-allocated (except when explicitly punching holes, then merely +reading zero is not enough to know if it is sparse, so we still want +to punch the hole). Avoiding a redundant write to zero (whether in +the background because the zero cluster was marked in the dirty +bitmap, or in the foreground because the guest is writing zeroes) when +the destination already reads as zero makes mirroring faster, and +avoids allocating the destination merely because the source reports as +allocated. + +The effect is especially pronounced when the source is a raw file. +That's because when the source is a qcow2 file, the dirty bitmap only +visits the portions of the source that are allocated, which tend to be +non-zero. But when the source is a raw file, +bdrv_co_is_allocated_above() reports the entire file as allocated so +mirror_dirty_init sets the entire dirty bitmap, and it is only later +during mirror_iteration that we change to consulting the more precise +bdrv_co_block_status_above() to learn where the source reads as zero. + +Remember that since a mirror operation can write a cluster more than +once (every time the guest changes the source, the destination is also +changed to keep up), and the guest can change whether a given cluster +reads as zero, is discarded, or has non-zero data over the course of +the mirror operation, we can't take the shortcut of relying on +s->target_is_zero (which is static for the life of the job) in +mirror_co_zero() to see if the destination is already zero, because +that information may be stale. Any solution we use must be dynamic in +the face of the guest writing or discarding a cluster while the mirror +has been ongoing. + +We could just teach mirror_co_zero() to do a block_status() probe of +the destination, and skip the zeroes if the destination already reads +as zero, but we know from past experience that extra block_status() +calls are not always cheap (tmpfs, anyone?), especially when they are +random access rather than linear. Use of block_status() of the source +by the background task in a linear fashion is not our bottleneck (it's +a background task, after all); but since mirroring can be done while +the source is actively being changed, we don't want a slow +block_status() of the destination to occur on the hot path of the +guest trying to do random-access writes to the source. + +So this patch takes a slightly different approach: any time we have to +track dirty clusters, we can also track which clusters are known to +read as zero. For sync=TOP or when we are punching holes from +"detect-zeroes":"unmap", the zero bitmap starts out empty, but +prevents a second write zero to a cluster that was already zero by an +earlier pass; for sync=FULL when we are not punching holes, the zero +bitmap starts out full if the destination reads as zero during +initialization. Either way, I/O to the destination can now avoid +redundant write zero to a cluster that already reads as zero, all +without having to do a block_status() per write on the destination. + +With this patch, if I create a raw sparse destination file, connect it +with QMP 'blockdev-add' while leaving it at the default "discard": +"ignore", then run QMP 'blockdev-mirror' with "sync": "full", the +destination remains sparse rather than fully allocated. Meanwhile, a +destination image that is already fully allocated remains so unless it +was opened with "detect-zeroes": "unmap". And any time writing zeroes +is skipped, the job counters are not incremented. + +Signed-off-by: Eric Blake +Message-ID: <20250509204341.3553601-26-eblake@redhat.com> +Reviewed-by: Stefan Hajnoczi +(cherry picked from commit 7e277545b90874171128804e256a538fb0e8dd7e) +Jira: https://issues.redhat.com/browse/RHEL-82906 +Jira: https://issues.redhat.com/browse/RHEL-83015 +Signed-off-by: Eric Blake +--- + block/mirror.c | 107 ++++++++++++++++++++++++++++++++++++++++++------- + 1 file changed, 93 insertions(+), 14 deletions(-) + +diff --git a/block/mirror.c b/block/mirror.c +index 29cac1777c..7f3b5477ce 100644 +--- a/block/mirror.c ++++ b/block/mirror.c +@@ -73,6 +73,7 @@ typedef struct MirrorBlockJob { + size_t buf_size; + int64_t bdev_length; + unsigned long *cow_bitmap; ++ unsigned long *zero_bitmap; + BdrvDirtyBitmap *dirty_bitmap; + BdrvDirtyBitmapIter *dbi; + uint8_t *buf; +@@ -108,9 +109,12 @@ struct MirrorOp { + int64_t offset; + uint64_t bytes; + +- /* The pointee is set by mirror_co_read(), mirror_co_zero(), and +- * mirror_co_discard() before yielding for the first time */ ++ /* ++ * These pointers are set by mirror_co_read(), mirror_co_zero(), and ++ * mirror_co_discard() before yielding for the first time ++ */ + int64_t *bytes_handled; ++ bool *io_skipped; + + bool is_pseudo_op; + bool is_active_write; +@@ -408,15 +412,34 @@ static void coroutine_fn mirror_co_read(void *opaque) + static void coroutine_fn mirror_co_zero(void *opaque) + { + MirrorOp *op = opaque; +- int ret; ++ bool write_needed = true; ++ int ret = 0; + + op->s->in_flight++; + op->s->bytes_in_flight += op->bytes; + *op->bytes_handled = op->bytes; + op->is_in_flight = true; + +- ret = blk_co_pwrite_zeroes(op->s->target, op->offset, op->bytes, +- op->s->unmap ? BDRV_REQ_MAY_UNMAP : 0); ++ if (op->s->zero_bitmap) { ++ unsigned long end = DIV_ROUND_UP(op->offset + op->bytes, ++ op->s->granularity); ++ assert(QEMU_IS_ALIGNED(op->offset, op->s->granularity)); ++ assert(QEMU_IS_ALIGNED(op->bytes, op->s->granularity) || ++ op->offset + op->bytes == op->s->bdev_length); ++ if (find_next_zero_bit(op->s->zero_bitmap, end, ++ op->offset / op->s->granularity) == end) { ++ write_needed = false; ++ *op->io_skipped = true; ++ } ++ } ++ if (write_needed) { ++ ret = blk_co_pwrite_zeroes(op->s->target, op->offset, op->bytes, ++ op->s->unmap ? BDRV_REQ_MAY_UNMAP : 0); ++ } ++ if (ret >= 0 && op->s->zero_bitmap) { ++ bitmap_set(op->s->zero_bitmap, op->offset / op->s->granularity, ++ DIV_ROUND_UP(op->bytes, op->s->granularity)); ++ } + mirror_write_complete(op, ret); + } + +@@ -435,29 +458,43 @@ static void coroutine_fn mirror_co_discard(void *opaque) + } + + static unsigned mirror_perform(MirrorBlockJob *s, int64_t offset, +- unsigned bytes, MirrorMethod mirror_method) ++ unsigned bytes, MirrorMethod mirror_method, ++ bool *io_skipped) + { + MirrorOp *op; + Coroutine *co; + int64_t bytes_handled = -1; + ++ assert(QEMU_IS_ALIGNED(offset, s->granularity)); ++ assert(QEMU_IS_ALIGNED(bytes, s->granularity) || ++ offset + bytes == s->bdev_length); + op = g_new(MirrorOp, 1); + *op = (MirrorOp){ + .s = s, + .offset = offset, + .bytes = bytes, + .bytes_handled = &bytes_handled, ++ .io_skipped = io_skipped, + }; + qemu_co_queue_init(&op->waiting_requests); + + switch (mirror_method) { + case MIRROR_METHOD_COPY: ++ if (s->zero_bitmap) { ++ bitmap_clear(s->zero_bitmap, offset / s->granularity, ++ DIV_ROUND_UP(bytes, s->granularity)); ++ } + co = qemu_coroutine_create(mirror_co_read, op); + break; + case MIRROR_METHOD_ZERO: ++ /* s->zero_bitmap handled in mirror_co_zero */ + co = qemu_coroutine_create(mirror_co_zero, op); + break; + case MIRROR_METHOD_DISCARD: ++ if (s->zero_bitmap) { ++ bitmap_clear(s->zero_bitmap, offset / s->granularity, ++ DIV_ROUND_UP(bytes, s->granularity)); ++ } + co = qemu_coroutine_create(mirror_co_discard, op); + break; + default: +@@ -568,6 +605,7 @@ static void coroutine_fn GRAPH_UNLOCKED mirror_iteration(MirrorBlockJob *s) + int ret; + int64_t io_bytes; + int64_t io_bytes_acct; ++ bool io_skipped = false; + MirrorMethod mirror_method = MIRROR_METHOD_COPY; + + assert(!(offset % s->granularity)); +@@ -611,8 +649,10 @@ static void coroutine_fn GRAPH_UNLOCKED mirror_iteration(MirrorBlockJob *s) + } + + io_bytes = mirror_clip_bytes(s, offset, io_bytes); +- io_bytes = mirror_perform(s, offset, io_bytes, mirror_method); +- if (mirror_method != MIRROR_METHOD_COPY && write_zeroes_ok) { ++ io_bytes = mirror_perform(s, offset, io_bytes, mirror_method, ++ &io_skipped); ++ if (io_skipped || ++ (mirror_method != MIRROR_METHOD_COPY && write_zeroes_ok)) { + io_bytes_acct = 0; + } else { + io_bytes_acct = io_bytes; +@@ -847,8 +887,10 @@ static int coroutine_fn GRAPH_UNLOCKED mirror_dirty_init(MirrorBlockJob *s) + bool punch_holes = + target_bs->detect_zeroes == BLOCKDEV_DETECT_ZEROES_OPTIONS_UNMAP && + bdrv_can_write_zeroes_with_unmap(target_bs); ++ int64_t bitmap_length = DIV_ROUND_UP(s->bdev_length, s->granularity); + + /* Determine if the image is already zero, regardless of sync mode. */ ++ s->zero_bitmap = bitmap_new(bitmap_length); + bdrv_graph_co_rdlock(); + bs = s->mirror_top_bs->backing->bs; + if (s->target_is_zero) { +@@ -862,7 +904,14 @@ static int coroutine_fn GRAPH_UNLOCKED mirror_dirty_init(MirrorBlockJob *s) + if (ret < 0) { + return ret; + } else if (s->sync_mode == MIRROR_SYNC_MODE_TOP) { +- /* In TOP mode, there is no benefit to a pre-zeroing pass. */ ++ /* ++ * In TOP mode, there is no benefit to a pre-zeroing pass, but ++ * the zero bitmap can be set if the destination already reads ++ * as zero and we are not punching holes. ++ */ ++ if (ret > 0 && !punch_holes) { ++ bitmap_set(s->zero_bitmap, 0, bitmap_length); ++ } + } else if (ret == 0 || punch_holes) { + /* + * Here, we are in FULL mode; our goal is to avoid writing +@@ -871,8 +920,9 @@ static int coroutine_fn GRAPH_UNLOCKED mirror_dirty_init(MirrorBlockJob *s) + * zeroing happened externally (ret > 0) or if we have a fast + * way to pre-zero the image (the dirty bitmap will be + * populated later by the non-zero portions, the same as for +- * TOP mode). If pre-zeroing is not fast, or we need to punch +- * holes, then our only recourse is to write the entire image. ++ * TOP mode). If pre-zeroing is not fast, then our only ++ * recourse is to mark the entire image dirty. The act of ++ * pre-zeroing will populate the zero bitmap. + */ + if (!bdrv_can_write_zeroes_with_unmap(target_bs)) { + bdrv_set_dirty_bitmap(s->dirty_bitmap, 0, s->bdev_length); +@@ -883,6 +933,7 @@ static int coroutine_fn GRAPH_UNLOCKED mirror_dirty_init(MirrorBlockJob *s) + for (offset = 0; offset < s->bdev_length; ) { + int bytes = MIN(s->bdev_length - offset, + QEMU_ALIGN_DOWN(INT_MAX, s->granularity)); ++ bool ignored; + + mirror_throttle(s); + +@@ -898,12 +949,15 @@ static int coroutine_fn GRAPH_UNLOCKED mirror_dirty_init(MirrorBlockJob *s) + continue; + } + +- mirror_perform(s, offset, bytes, MIRROR_METHOD_ZERO); ++ mirror_perform(s, offset, bytes, MIRROR_METHOD_ZERO, &ignored); + offset += bytes; + } + + mirror_wait_for_all_io(s); + s->initial_zeroing_ongoing = false; ++ } else { ++ /* In FULL mode, and image already reads as zero. */ ++ bitmap_set(s->zero_bitmap, 0, bitmap_length); + } + + /* First part, loop on the sectors and initialize the dirty bitmap. */ +@@ -1188,6 +1242,7 @@ immediate_exit: + assert(s->in_flight == 0); + qemu_vfree(s->buf); + g_free(s->cow_bitmap); ++ g_free(s->zero_bitmap); + g_free(s->in_flight_bitmap); + bdrv_dirty_iter_free(s->dbi); + +@@ -1367,6 +1422,7 @@ do_sync_target_write(MirrorBlockJob *job, MirrorMethod method, + int ret; + size_t qiov_offset = 0; + int64_t dirty_bitmap_offset, dirty_bitmap_end; ++ int64_t zero_bitmap_offset, zero_bitmap_end; + + if (!QEMU_IS_ALIGNED(offset, job->granularity) && + bdrv_dirty_bitmap_get(job->dirty_bitmap, offset)) +@@ -1410,8 +1466,9 @@ do_sync_target_write(MirrorBlockJob *job, MirrorMethod method, + } + + /* +- * Tails are either clean or shrunk, so for bitmap resetting +- * we safely align the range down. ++ * Tails are either clean or shrunk, so for dirty bitmap resetting ++ * we safely align the range narrower. But for zero bitmap, round ++ * range wider for checking or clearing, and narrower for setting. + */ + dirty_bitmap_offset = QEMU_ALIGN_UP(offset, job->granularity); + dirty_bitmap_end = QEMU_ALIGN_DOWN(offset + bytes, job->granularity); +@@ -1419,22 +1476,44 @@ do_sync_target_write(MirrorBlockJob *job, MirrorMethod method, + bdrv_reset_dirty_bitmap(job->dirty_bitmap, dirty_bitmap_offset, + dirty_bitmap_end - dirty_bitmap_offset); + } ++ zero_bitmap_offset = offset / job->granularity; ++ zero_bitmap_end = DIV_ROUND_UP(offset + bytes, job->granularity); + + job_progress_increase_remaining(&job->common.job, bytes); + job->active_write_bytes_in_flight += bytes; + + switch (method) { + case MIRROR_METHOD_COPY: ++ if (job->zero_bitmap) { ++ bitmap_clear(job->zero_bitmap, zero_bitmap_offset, ++ zero_bitmap_end - zero_bitmap_offset); ++ } + ret = blk_co_pwritev_part(job->target, offset, bytes, + qiov, qiov_offset, flags); + break; + + case MIRROR_METHOD_ZERO: ++ if (job->zero_bitmap) { ++ if (find_next_zero_bit(job->zero_bitmap, zero_bitmap_end, ++ zero_bitmap_offset) == zero_bitmap_end) { ++ ret = 0; ++ break; ++ } ++ } + assert(!qiov); + ret = blk_co_pwrite_zeroes(job->target, offset, bytes, flags); ++ if (job->zero_bitmap && ret >= 0) { ++ bitmap_set(job->zero_bitmap, dirty_bitmap_offset / job->granularity, ++ (dirty_bitmap_end - dirty_bitmap_offset) / ++ job->granularity); ++ } + break; + + case MIRROR_METHOD_DISCARD: ++ if (job->zero_bitmap) { ++ bitmap_clear(job->zero_bitmap, zero_bitmap_offset, ++ zero_bitmap_end - zero_bitmap_offset); ++ } + assert(!qiov); + ret = blk_co_pdiscard(job->target, offset, bytes); + break; +-- +2.48.1 + diff --git a/SOURCES/kvm-net-socket-skip-automatic-zero-init-of-large-array.patch b/SOURCES/kvm-net-socket-skip-automatic-zero-init-of-large-array.patch index 0ecc437..f5361cf 100644 --- a/SOURCES/kvm-net-socket-skip-automatic-zero-init-of-large-array.patch +++ b/SOURCES/kvm-net-socket-skip-automatic-zero-init-of-large-array.patch @@ -1,16 +1,16 @@ -From 9a941183c365f4c6698c46aa13594e9bb360e8e9 Mon Sep 17 00:00:00 2001 +From 4b9a1a9154467fd65ac2a0a26959d3342d8fcd49 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:37:08 +0100 -Subject: [PATCH 30/31] net/socket: skip automatic zero-init of large array +Subject: [PATCH 55/57] net/socket: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [29/30] e7f35ae85a0ca50367487b1e7ed92b395b247cda +RH-Commit: [29/30] 645ad4d138d1222ea9bd1b2ac3b84d9ff83e2fa2 (stefanha/centos-stream-qemu-kvm) The 'net_socket_send' method has a 68k byte array used for copying data between guest and host. Skip the automatic zero-init of this diff --git a/SOURCES/kvm-net-stream-skip-automatic-zero-init-of-large-array.patch b/SOURCES/kvm-net-stream-skip-automatic-zero-init-of-large-array.patch index 75c2aa8..e9abf3f 100644 --- a/SOURCES/kvm-net-stream-skip-automatic-zero-init-of-large-array.patch +++ b/SOURCES/kvm-net-stream-skip-automatic-zero-init-of-large-array.patch @@ -1,16 +1,16 @@ -From 790b862712841e4b363874e00f4da993ae045a53 Mon Sep 17 00:00:00 2001 +From 94310a4168257297e52058d5d6aea4a2d06630c6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Tue, 10 Jun 2025 13:37:09 +0100 -Subject: [PATCH 31/31] net/stream: skip automatic zero-init of large array +Subject: [PATCH 56/57] net/stream: skip automatic zero-init of large array MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Stefan Hajnoczi -RH-MergeRequest: 461: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED -RH-Jira: RHEL-99887 +RH-MergeRequest: 382: Solve -ftrivial-auto-var-init performance regression with QEMU_UNINITIALIZED +RH-Jira: RHEL-99888 RH-Acked-by: Miroslav Rezanina -RH-Commit: [30/30] 005a9dfc74c8fed3d1aa2e0cdc0473f67f888249 +RH-Commit: [30/30] 9dfec5c0e6358e3557bf58d66eee8e4ba6e93621 (stefanha/centos-stream-qemu-kvm) The 'net_stream_send' method has a 68k byte array used for copying data between guest and host. Skip the automatic zero-init of this diff --git a/SOURCES/kvm-net-vhost-user-add-QAPI-events-to-report-connection-.patch b/SOURCES/kvm-net-vhost-user-add-QAPI-events-to-report-connection-.patch index 543f51c..a792e7c 100644 --- a/SOURCES/kvm-net-vhost-user-add-QAPI-events-to-report-connection-.patch +++ b/SOURCES/kvm-net-vhost-user-add-QAPI-events-to-report-connection-.patch @@ -1,21 +1,18 @@ -From 5285f9d3377d8d8c1872c1f75a6583e42c97686e Mon Sep 17 00:00:00 2001 +From b6de1e19ba778547e92997c6cad77d7cf755c78b Mon Sep 17 00:00:00 2001 From: Laurent Vivier Date: Mon, 17 Feb 2025 10:25:50 +0100 -Subject: [PATCH 1/2] net: vhost-user: add QAPI events to report connection +Subject: [PATCH 1/3] net: vhost-user: add QAPI events to report connection state MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Laurent Vivier -RH-MergeRequest: 432: net: vhost-user: add QAPI events to report connection state -RH-Jira: RHEL-80622 -RH-Acked-by: Cindy Lu +RH-MergeRequest: 371: net: vhost-user: add QAPI events to report connection state +RH-Jira: RHEL-95120 RH-Acked-by: Eugenio Pérez -RH-Acked-by: Jason Wang -RH-Commit: [1/1] f8459679e558339d4fa45ed88ed71dc5d5ef459d - -JIRA: https://issues.redhat.com/browse/RHEL-80622 +RH-Acked-by: Cindy Lu +RH-Commit: [1/1] c8f65026e3548891fe713a1622438388e285dbf3 (lvivier/qemu-kvm-centos) The netdev reports NETDEV_VHOST_USER_CONNECTED event when the chardev is connected, and NETDEV_VHOST_USER_DISCONNECTED @@ -50,7 +47,6 @@ Acked-by: Markus Armbruster Reviewed-by: Michael S. Tsirkin Signed-off-by: Michael S. Tsirkin (cherry picked from commit 02fd9f8aeeb184276b283ae2f404bc3acf1e7b7a) -Signed-off-by: Laurent Vivier --- net/vhost-user.c | 3 +++ qapi/net.json | 40 ++++++++++++++++++++++++++++++++++++++++ diff --git a/SOURCES/kvm-pci-Use-PCI-PM-capability-initializer.patch b/SOURCES/kvm-pci-Use-PCI-PM-capability-initializer.patch new file mode 100644 index 0000000..e2470de --- /dev/null +++ b/SOURCES/kvm-pci-Use-PCI-PM-capability-initializer.patch @@ -0,0 +1,153 @@ +From 978951b390bb7073293c792c4714516ad40cba73 Mon Sep 17 00:00:00 2001 +From: Alex Williamson +Date: Tue, 25 Feb 2025 14:52:26 -0700 +Subject: [PATCH 3/7] pci: Use PCI PM capability initializer +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Eric Auger +RH-MergeRequest: 348: PCI: Implement basic PCI PM capability backing +RH-Jira: RHEL-7301 +RH-Acked-by: Cédric Le Goater +RH-Acked-by: Alex Williamson +RH-Acked-by: Jon Maloy +RH-Commit: [3/6] fd862caa094490a9b8a04b00ad39ba58e0b46a7a (eauger1/centos-qemu-kvm) + +Switch callers directly initializing the PCI PM capability with +pci_add_capability() to use pci_pm_init(). + +Cc: Dmitry Fleytman +Cc: Akihiko Odaki +Cc: Jason Wang +Cc: Stefan Weil +Cc: Sriram Yagnaraman +Cc: Keith Busch +Cc: Klaus Jensen +Cc: Jesper Devantier +Cc: Michael S. Tsirkin +Cc: Marcel Apfelbaum +Cc: Cédric Le Goater +Signed-off-by: Alex Williamson +Reviewed-by: Eric Auger +Reviewed-by: Akihiko Odaki +Reviewed-by: Michael S. Tsirkin +Link: https://lore.kernel.org/qemu-devel/20250225215237.3314011-3-alex.williamson@redhat.com +Signed-off-by: Cédric Le Goater +(cherry picked from commit 0681ec253141d838210b3c5e6bc0d2d71f2e111e) +Signed-off-by: Eric Auger +--- + hw/net/e1000e.c | 3 +-- + hw/net/eepro100.c | 4 +--- + hw/net/igb.c | 3 +-- + hw/nvme/ctrl.c | 3 +-- + hw/pci-bridge/pcie_pci_bridge.c | 2 +- + hw/vfio/pci.c | 7 ++++++- + hw/virtio/virtio-pci.c | 3 +-- + 7 files changed, 12 insertions(+), 13 deletions(-) + +diff --git a/hw/net/e1000e.c b/hw/net/e1000e.c +index 843892ce09..9eb93d049d 100644 +--- a/hw/net/e1000e.c ++++ b/hw/net/e1000e.c +@@ -372,8 +372,7 @@ static int + e1000e_add_pm_capability(PCIDevice *pdev, uint8_t offset, uint16_t pmc) + { + Error *local_err = NULL; +- int ret = pci_add_capability(pdev, PCI_CAP_ID_PM, offset, +- PCI_PM_SIZEOF, &local_err); ++ int ret = pci_pm_init(pdev, offset, &local_err); + + if (local_err) { + error_report_err(local_err); +diff --git a/hw/net/eepro100.c b/hw/net/eepro100.c +index d9a70c4544..668a410055 100644 +--- a/hw/net/eepro100.c ++++ b/hw/net/eepro100.c +@@ -549,9 +549,7 @@ static void e100_pci_reset(EEPRO100State *s, Error **errp) + if (info->power_management) { + /* Power Management Capabilities */ + int cfg_offset = 0xdc; +- int r = pci_add_capability(&s->dev, PCI_CAP_ID_PM, +- cfg_offset, PCI_PM_SIZEOF, +- errp); ++ int r = pci_pm_init(&s->dev, cfg_offset, errp); + if (r < 0) { + return; + } +diff --git a/hw/net/igb.c b/hw/net/igb.c +index b92bba402e..a3c22e2391 100644 +--- a/hw/net/igb.c ++++ b/hw/net/igb.c +@@ -356,8 +356,7 @@ static int + igb_add_pm_capability(PCIDevice *pdev, uint8_t offset, uint16_t pmc) + { + Error *local_err = NULL; +- int ret = pci_add_capability(pdev, PCI_CAP_ID_PM, offset, +- PCI_PM_SIZEOF, &local_err); ++ int ret = pci_pm_init(pdev, offset, &local_err); + + if (local_err) { + error_report_err(local_err); +diff --git a/hw/nvme/ctrl.c b/hw/nvme/ctrl.c +index 9f277b81d8..d451ee0d00 100644 +--- a/hw/nvme/ctrl.c ++++ b/hw/nvme/ctrl.c +@@ -8293,8 +8293,7 @@ static int nvme_add_pm_capability(PCIDevice *pci_dev, uint8_t offset) + Error *err = NULL; + int ret; + +- ret = pci_add_capability(pci_dev, PCI_CAP_ID_PM, offset, +- PCI_PM_SIZEOF, &err); ++ ret = pci_pm_init(pci_dev, offset, &err); + if (err) { + error_report_err(err); + return ret; +diff --git a/hw/pci-bridge/pcie_pci_bridge.c b/hw/pci-bridge/pcie_pci_bridge.c +index 7646ac2397..2f098e3a13 100644 +--- a/hw/pci-bridge/pcie_pci_bridge.c ++++ b/hw/pci-bridge/pcie_pci_bridge.c +@@ -52,7 +52,7 @@ static void pcie_pci_bridge_realize(PCIDevice *d, Error **errp) + goto cap_error; + } + +- pos = pci_add_capability(d, PCI_CAP_ID_PM, 0, PCI_PM_SIZEOF, errp); ++ pos = pci_pm_init(d, 0, errp); + if (pos < 0) { + goto pm_error; + } +diff --git a/hw/vfio/pci.c b/hw/vfio/pci.c +index 82a47edc89..e18b57d864 100644 +--- a/hw/vfio/pci.c ++++ b/hw/vfio/pci.c +@@ -2220,7 +2220,12 @@ static bool vfio_add_std_cap(VFIOPCIDevice *vdev, uint8_t pos, Error **errp) + case PCI_CAP_ID_PM: + vfio_check_pm_reset(vdev, pos); + vdev->pm_cap = pos; +- ret = pci_add_capability(pdev, cap_id, pos, size, errp) >= 0; ++ ret = pci_pm_init(pdev, pos, errp) >= 0; ++ /* ++ * PCI-core config space emulation needs write access to the power ++ * state enabled for tracking BAR mapping relative to PM state. ++ */ ++ pci_set_word(pdev->wmask + pos + PCI_PM_CTRL, PCI_PM_CTRL_STATE_MASK); + break; + case PCI_CAP_ID_AF: + vfio_check_af_flr(vdev, pos); +diff --git a/hw/virtio/virtio-pci.c b/hw/virtio/virtio-pci.c +index 524b63e5c7..4b2aeaad8d 100644 +--- a/hw/virtio/virtio-pci.c ++++ b/hw/virtio/virtio-pci.c +@@ -2195,8 +2195,7 @@ static void virtio_pci_realize(PCIDevice *pci_dev, Error **errp) + pos = pcie_endpoint_cap_init(pci_dev, 0); + assert(pos > 0); + +- pos = pci_add_capability(pci_dev, PCI_CAP_ID_PM, 0, +- PCI_PM_SIZEOF, errp); ++ pos = pci_pm_init(pci_dev, 0, errp); + if (pos < 0) { + return; + } +-- +2.48.1 + diff --git a/SOURCES/kvm-pcie-virtio-Remove-redundant-pm_cap.patch b/SOURCES/kvm-pcie-virtio-Remove-redundant-pm_cap.patch new file mode 100644 index 0000000..15d82a2 --- /dev/null +++ b/SOURCES/kvm-pcie-virtio-Remove-redundant-pm_cap.patch @@ -0,0 +1,99 @@ +From 274e81bcf091c981d1e27e49fbe98e63d5308472 Mon Sep 17 00:00:00 2001 +From: Alex Williamson +Date: Tue, 25 Feb 2025 14:52:28 -0700 +Subject: [PATCH 5/7] pcie, virtio: Remove redundant pm_cap +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Eric Auger +RH-MergeRequest: 348: PCI: Implement basic PCI PM capability backing +RH-Jira: RHEL-7301 +RH-Acked-by: Cédric Le Goater +RH-Acked-by: Alex Williamson +RH-Acked-by: Jon Maloy +RH-Commit: [5/6] 81c6e3c9c52a0b3f0b9269b4ac7f56e8e4b5d68b (eauger1/centos-qemu-kvm) + +The pm_cap on the PCIExpressDevice object can be distilled down +to the new instance on the PCIDevice object. + +Cc: Michael S. Tsirkin +Cc: Marcel Apfelbaum +Reviewed-by: Michael S. Tsirkin +Reviewed-by: Zhenzhong Duan +Reviewed-by: Eric Auger +Signed-off-by: Alex Williamson +Link: https://lore.kernel.org/qemu-devel/20250225215237.3314011-5-alex.williamson@redhat.com +Signed-off-by: Cédric Le Goater +(cherry picked from commit 8b8d08cf293b930d0f55b2d5385d8dd27e0c6b41) +Signed-off-by: Eric Auger +--- + hw/pci-bridge/pcie_pci_bridge.c | 1 - + hw/virtio/virtio-pci.c | 8 +++----- + include/hw/pci/pcie.h | 2 -- + 3 files changed, 3 insertions(+), 8 deletions(-) + +diff --git a/hw/pci-bridge/pcie_pci_bridge.c b/hw/pci-bridge/pcie_pci_bridge.c +index 2f098e3a13..c0ba6d7928 100644 +--- a/hw/pci-bridge/pcie_pci_bridge.c ++++ b/hw/pci-bridge/pcie_pci_bridge.c +@@ -56,7 +56,6 @@ static void pcie_pci_bridge_realize(PCIDevice *d, Error **errp) + if (pos < 0) { + goto pm_error; + } +- d->exp.pm_cap = pos; + pci_set_word(d->config + pos + PCI_PM_PMC, 0x3); + + pcie_cap_arifwd_init(d); +diff --git a/hw/virtio/virtio-pci.c b/hw/virtio/virtio-pci.c +index 4b2aeaad8d..a85787b837 100644 +--- a/hw/virtio/virtio-pci.c ++++ b/hw/virtio/virtio-pci.c +@@ -2200,8 +2200,6 @@ static void virtio_pci_realize(PCIDevice *pci_dev, Error **errp) + return; + } + +- pci_dev->exp.pm_cap = pos; +- + /* + * Indicates that this function complies with revision 1.2 of the + * PCI Power Management Interface Specification. +@@ -2295,11 +2293,11 @@ static bool virtio_pci_no_soft_reset(PCIDevice *dev) + { + uint16_t pmcsr; + +- if (!pci_is_express(dev) || !dev->exp.pm_cap) { ++ if (!pci_is_express(dev) || !(dev->cap_present & QEMU_PCI_CAP_PM)) { + return false; + } + +- pmcsr = pci_get_word(dev->config + dev->exp.pm_cap + PCI_PM_CTRL); ++ pmcsr = pci_get_word(dev->config + dev->pm_cap + PCI_PM_CTRL); + + /* + * When No_Soft_Reset bit is set and the device +@@ -2328,7 +2326,7 @@ static void virtio_pci_bus_reset_hold(Object *obj, ResetType type) + + if (proxy->flags & VIRTIO_PCI_FLAG_INIT_PM) { + pci_word_test_and_clear_mask( +- dev->config + dev->exp.pm_cap + PCI_PM_CTRL, ++ dev->config + dev->pm_cap + PCI_PM_CTRL, + PCI_PM_CTRL_STATE_MASK); + } + } +diff --git a/include/hw/pci/pcie.h b/include/hw/pci/pcie.h +index 5eddb90976..8a30d07fd0 100644 +--- a/include/hw/pci/pcie.h ++++ b/include/hw/pci/pcie.h +@@ -58,8 +58,6 @@ typedef enum { + struct PCIExpressDevice { + /* Offset of express capability in config space */ + uint8_t exp_cap; +- /* Offset of Power Management capability in config space */ +- uint8_t pm_cap; + + /* SLOT */ + bool hpev_notified; /* Logical AND of conditions for hot plug event. +-- +2.48.1 + diff --git a/SOURCES/kvm-physmem-Support-coordinated-discarding-of-RAM-with-g.patch b/SOURCES/kvm-physmem-Support-coordinated-discarding-of-RAM-with-g.patch new file mode 100644 index 0000000..0df872d --- /dev/null +++ b/SOURCES/kvm-physmem-Support-coordinated-discarding-of-RAM-with-g.patch @@ -0,0 +1,129 @@ +From 8d8fd49920c7b4c49d2d7ca1666c3cc7a53a3e96 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:11:28 +0200 +Subject: [PATCH 115/115] physmem: Support coordinated discarding of RAM with + guest_memfd + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [115/115] 5dfe5036d7b5524f0d0af4cde23cdd9ca696717b (bonzini/rhel-qemu-kvm) + +A new field, attributes, was introduced in RAMBlock to link to a +RamBlockAttributes object, which centralizes all guest_memfd related +information (such as fd and status bitmap) within a RAMBlock. + +Create and initialize the RamBlockAttributes object upon ram_block_add(). +Meanwhile, register the object in the target RAMBlock's MemoryRegion. +After that, guest_memfd-backed RAMBlock is associated with the +RamDiscardManager interface, and the users can execute RamDiscardManager +specific handling. For example, VFIO will register the +RamDiscardListener and get notifications when the state_change() helper +invokes. + +As coordinate discarding of RAM with guest_memfd is now supported, only +block uncoordinated discard. + +Tested-by: Alexey Kardashevskiy +Reviewed-by: Alexey Kardashevskiy +Acked-by: David Hildenbrand +Signed-off-by: Chenyi Qiang +Link: https://lore.kernel.org/r/20250612082747.51539-6-chenyi.qiang@intel.com +Signed-off-by: Peter Xu +(cherry picked from commit 2fde3fb916079ee0ff0fc26d9446c813b1d5cc28) +Signed-off-by: Paolo Bonzini + +Conflicts: No CPR +--- + accel/kvm/kvm-all.c | 9 +++++++++ + include/exec/ramblock.h | 1 + + system/physmem.c | 23 +++++++++++++++++++++-- + 3 files changed, 31 insertions(+), 2 deletions(-) + +diff --git a/accel/kvm/kvm-all.c b/accel/kvm/kvm-all.c +index 43c10c82f6..1fd7773a28 100644 +--- a/accel/kvm/kvm-all.c ++++ b/accel/kvm/kvm-all.c +@@ -3073,6 +3073,15 @@ int kvm_convert_memory(hwaddr start, hwaddr size, bool to_private) + addr = memory_region_get_ram_ptr(mr) + section.offset_within_region; + rb = qemu_ram_block_from_host(addr, false, &offset); + ++ ret = ram_block_attributes_state_change(RAM_BLOCK_ATTRIBUTES(mr->rdm), ++ offset, size, to_private); ++ if (ret) { ++ error_report("Failed to notify the listener the state change of " ++ "(0x%"HWADDR_PRIx" + 0x%"HWADDR_PRIx") to %s", ++ start, size, to_private ? "private" : "shared"); ++ goto out_unref; ++ } ++ + if (to_private) { + if (rb->page_size != qemu_real_host_page_size()) { + /* +diff --git a/include/exec/ramblock.h b/include/exec/ramblock.h +index 9ae774d268..a504765038 100644 +--- a/include/exec/ramblock.h ++++ b/include/exec/ramblock.h +@@ -46,6 +46,7 @@ struct RAMBlock { + int fd; + uint64_t fd_offset; + int guest_memfd; ++ RamBlockAttributes *attributes; + size_t page_size; + /* dirty bitmap used during migration */ + unsigned long *bmap; +diff --git a/system/physmem.c b/system/physmem.c +index dccc95030b..35f8f25e22 100644 +--- a/system/physmem.c ++++ b/system/physmem.c +@@ -1889,7 +1889,7 @@ static void ram_block_add(RAMBlock *new_block, Error **errp) + } + assert(new_block->guest_memfd < 0); + +- ret = ram_block_discard_require(true); ++ ret = ram_block_coordinated_discard_require(true); + if (ret < 0) { + error_setg_errno(errp, -ret, + "cannot set up private guest memory: discard currently blocked"); +@@ -1903,6 +1903,24 @@ static void ram_block_add(RAMBlock *new_block, Error **errp) + qemu_mutex_unlock_ramlist(); + goto out_free; + } ++ ++ /* ++ * The attribute bitmap of the RamBlockAttributes is default to ++ * discarded, which mimics the behavior of kvm_set_phys_mem() when it ++ * calls kvm_set_memory_attributes_private(). This leads to a brief ++ * period of inconsistency between the creation of the RAMBlock and its ++ * mapping into the physical address space. However, this is not ++ * problematic, as no users rely on the attribute status to perform ++ * any actions during this interval. ++ */ ++ new_block->attributes = ram_block_attributes_create(new_block); ++ if (!new_block->attributes) { ++ error_setg(errp, "Failed to create ram block attribute"); ++ close(new_block->guest_memfd); ++ ram_block_coordinated_discard_require(false); ++ qemu_mutex_unlock_ramlist(); ++ goto out_free; ++ } + } + + new_ram_size = MAX(old_ram_size, +@@ -2159,8 +2177,9 @@ static void reclaim_ramblock(RAMBlock *block) + } + + if (block->guest_memfd >= 0) { ++ ram_block_attributes_destroy(block->attributes); + close(block->guest_memfd); +- ram_block_discard_require(false); ++ ram_block_coordinated_discard_require(false); + } + + g_free(block); +-- +2.50.1 + diff --git a/SOURCES/kvm-physmem-replace-assertion-with-error.patch b/SOURCES/kvm-physmem-replace-assertion-with-error.patch new file mode 100644 index 0000000..98fa4ce --- /dev/null +++ b/SOURCES/kvm-physmem-replace-assertion-with-error.patch @@ -0,0 +1,72 @@ +From 40d97d335471a77b1491c124d2c109db68bf8ca6 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Mon, 17 Feb 2025 13:08:12 +0100 +Subject: [PATCH 020/115] physmem: replace assertion with error +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [20/115] 5c17949c668857760ee80279e5776e8dcf4c7c11 (bonzini/rhel-qemu-kvm) + +It is possible to start QEMU with a confidential-guest-support object +even in TCG mode. While there is already a check in qemu_machine_creation_done: + + if (machine->cgs && !machine->cgs->ready) { + error_setg(errp, "accelerator does not support confidential guest %s", + object_get_typename(OBJECT(machine->cgs))); + exit(1); + } + +the creation of RAMBlocks happens earlier, in qemu_init_board(), if +the command line does not override the default memory backend with +-M memdev. Then the RAMBlock will try to use guest_memfd (because +machine_require_guest_memfd correctly returns true; at least correctly +according to the current implementation) and trigger the assertion +failure for kvm_enabled(). This happend with a command line as +simple as the following: + + qemu-system-x86_64 -m 512 -nographic -object sev-snp-guest,reduced-phys-bits=48,id=sev0 \ + -M q35,kernel-irqchip=split,confidential-guest-support=sev0 + qemu-system-x86_64: ../system/physmem.c:1871: ram_block_add: Assertion `kvm_enabled()' failed. + +Cc: Xiaoyao Li +Cc: qemu-stable@nongnu.org +Signed-off-by: Paolo Bonzini +Reviewed-by: Daniel P. Berrangé +Reviewed-by: David Hildenbrand +Reviewed-by: Pankaj Gupta +Reviewed-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20250217120812.396522-1-pbonzini@redhat.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 6debfb2cb1795427d2dc6a741c7430a233c76695) +Signed-off-by: Paolo Bonzini +--- + system/physmem.c | 6 +++++- + 1 file changed, 5 insertions(+), 1 deletion(-) + +diff --git a/system/physmem.c b/system/physmem.c +index 94600a33ec..dccc95030b 100644 +--- a/system/physmem.c ++++ b/system/physmem.c +@@ -1882,7 +1882,11 @@ static void ram_block_add(RAMBlock *new_block, Error **errp) + if (new_block->flags & RAM_GUEST_MEMFD) { + int ret; + +- assert(kvm_enabled()); ++ if (!kvm_enabled()) { ++ error_setg(errp, "cannot set up private guest memory for %s: KVM required", ++ object_get_typename(OBJECT(current_machine->cgs))); ++ goto out_free; ++ } + assert(new_block->guest_memfd < 0); + + ret = ram_block_discard_require(true); +-- +2.50.1 + diff --git a/SOURCES/kvm-qga-implement-a-guest-get-load-command.patch b/SOURCES/kvm-qga-implement-a-guest-get-load-command.patch index 640dd49..d4622ff 100644 --- a/SOURCES/kvm-qga-implement-a-guest-get-load-command.patch +++ b/SOURCES/kvm-qga-implement-a-guest-get-load-command.patch @@ -1,17 +1,17 @@ -From 0e3c6791fe9788e8f170818787bbdf609297da7d Mon Sep 17 00:00:00 2001 +From 22f26a93ab94bf87c0724891a5886797a38c23b4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= Date: Mon, 2 Dec 2024 12:19:27 +0000 -Subject: [PATCH 2/2] qga: implement a 'guest-get-load' command +Subject: [PATCH 6/9] qga: implement a 'guest-get-load' command MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Konstantin Kostiuk -RH-MergeRequest: 434: RHEL-83000: qga: implement a 'guest-get-load' command -RH-Jira: RHEL-83000 -RH-Acked-by: Jon Maloy +RH-MergeRequest: 343: RHEL-69622: qga: implement a 'guest-get-load' command +RH-Jira: RHEL-69622 RH-Acked-by: Daniel P. Berrangé -RH-Commit: [1/1] ad1d374a9e2200e0e36727e3484b96e9675bd0e4 +RH-Acked-by: Jon Maloy +RH-Commit: [1/1] 9284c70737ad9f700d37f8c3833f855f2354acb7 (kkostiuk/redhat-centos-stream-src-qemu-kvm) Provide a way to report the process load average, via a new 'guest-get-load' command. diff --git a/SOURCES/kvm-qom-reverse-order-of-instance_post_init-calls.patch b/SOURCES/kvm-qom-reverse-order-of-instance_post_init-calls.patch new file mode 100644 index 0000000..816de07 --- /dev/null +++ b/SOURCES/kvm-qom-reverse-order-of-instance_post_init-calls.patch @@ -0,0 +1,79 @@ +From 60ab87c5cad64e3169ee65f396c6d9f1f7eb0daa Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 022/115] qom: reverse order of instance_post_init calls +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [22/115] e4e2393adffd671dc4d4f7147c620884757c7d74 (bonzini/rhel-qemu-kvm) + +Currently, the instance_post_init calls are performed from the leaf +class and all the way up to Object. This is incorrect because the +leaf class cannot observe property values applied by the superclasses; +for example, a compat property will be set on a device *after* +the class's post_init callback has run. + +In particular this makes it impossible for implementations of +accel_cpu_instance_init() to operate based on the actual values of +the properties, though it seems that cxl_dsp_instance_post_init and +rp_instance_post_init might have similar issues. + +Follow instead the same order as instance_init, starting with Object +and running the child class's instance_post_init after the parent. + +Reviewed-by: Philippe Mathieu-Daudé +Reviewed-by: Alistair Francis +Signed-off-by: Paolo Bonzini +(cherry picked from commit 220c739903cec99df032219ac94c45b5269a0ab5) +Signed-off-by: Paolo Bonzini +--- + include/qom/object.h | 3 ++- + qom/object.c | 8 ++++---- + 2 files changed, 6 insertions(+), 5 deletions(-) + +diff --git a/include/qom/object.h b/include/qom/object.h +index 13d3a655dd..668dd1cc08 100644 +--- a/include/qom/object.h ++++ b/include/qom/object.h +@@ -444,7 +444,8 @@ struct Object + * class will have already been initialized so the type is only responsible + * for initializing its own members. + * @instance_post_init: This function is called to finish initialization of +- * an object, after all @instance_init functions were called. ++ * an object, after all @instance_init functions were called, as well as ++ * @instance_post_init functions for the parent classes. + * @instance_finalize: This function is called during object destruction. This + * is called before the parent @instance_finalize function has been called. + * An object should only free the members that are unique to its type in this +diff --git a/qom/object.c b/qom/object.c +index 157a45c5f8..c03cd3c733 100644 +--- a/qom/object.c ++++ b/qom/object.c +@@ -423,13 +423,13 @@ static void object_init_with_type(Object *obj, TypeImpl *ti) + + static void object_post_init_with_type(Object *obj, TypeImpl *ti) + { +- if (ti->instance_post_init) { +- ti->instance_post_init(obj); +- } +- + if (type_has_parent(ti)) { + object_post_init_with_type(obj, type_get_parent(ti)); + } ++ ++ if (ti->instance_post_init) { ++ ti->instance_post_init(obj); ++ } + } + + bool object_apply_global_props(Object *obj, const GPtrArray *props, +-- +2.50.1 + diff --git a/SOURCES/kvm-ram-block-attributes-Introduce-RamBlockAttributes-to.patch b/SOURCES/kvm-ram-block-attributes-Introduce-RamBlockAttributes-to.patch new file mode 100644 index 0000000..22c3baf --- /dev/null +++ b/SOURCES/kvm-ram-block-attributes-Introduce-RamBlockAttributes-to.patch @@ -0,0 +1,613 @@ +From 13a29003f5a5502fc2cd13cb22f3fd6318e80196 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:11:28 +0200 +Subject: [PATCH 114/115] ram-block-attributes: Introduce RamBlockAttributes to + manage RAMBlock with guest_memfd + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [114/115] eca60c04c6204ee91e4aedb84c993155605e6f5a (bonzini/rhel-qemu-kvm) + +Commit 852f0048f3 ("RAMBlock: make guest_memfd require uncoordinated +discard") highlighted that subsystems like VFIO may disable RAM block +discard. However, guest_memfd relies on discard operations for page +conversion between private and shared memory, potentially leading to +the stale IOMMU mapping issue when assigning hardware devices to +confidential VMs via shared memory. To address this and allow shared +device assignement, it is crucial to ensure the VFIO system refreshes +its IOMMU mappings. + +RamDiscardManager is an existing interface (used by virtio-mem) to +adjust VFIO mappings in relation to VM page assignment. Effectively page +conversion is similar to hot-removing a page in one mode and adding it +back in the other. Therefore, similar actions are required for page +conversion events. Introduce the RamDiscardManager to guest_memfd to +facilitate this process. + +Since guest_memfd is not an object, it cannot directly implement the +RamDiscardManager interface. Implementing it in HostMemoryBackend is +not appropriate because guest_memfd is per RAMBlock, and some RAMBlocks +have a memory backend while others do not. Notably, virtual BIOS +RAMBlocks using memory_region_init_ram_guest_memfd() do not have a +backend. + +To manage RAMBlocks with guest_memfd, define a new object named +RamBlockAttributes to implement the RamDiscardManager interface. This +object can store the guest_memfd information such as the bitmap for +shared memory and the registered listeners for event notifications. A +new state_change() helper function is provided to notify listeners, such +as VFIO, allowing VFIO to do dynamically DMA map and unmap for the shared +memory according to conversion events. Note that in the current context +of RamDiscardManager for guest_memfd, the shared state is analogous to +being populated, while the private state can be considered discarded for +simplicity. In the future, it would be more complicated if considering +more states like private/shared/discarded at the same time. + +In current implementation, memory state tracking is performed at the +host page size granularity, as the minimum conversion size can be one +page per request. Additionally, VFIO expected the DMA mapping for a +specific IOVA to be mapped and unmapped with the same granularity. +Confidential VMs may perform partial conversions, such as conversions on +small regions within a larger one. To prevent such invalid cases and +until support for DMA mapping cut operations is available, all +operations are performed with 4K granularity. + +In addition, memory conversion failures cause QEMU to quit rather than +resuming the guest or retrying the operation at present. It would be +future work to add more error handling or rollback mechanisms once +conversion failures are allowed. For example, in-place conversion of +guest_memfd could retry the unmap operation during the conversion from +shared to private. For now, keep the complex error handling out of the +picture as it is not required. + +Tested-by: Alexey Kardashevskiy +Reviewed-by: Alexey Kardashevskiy +Reviewed-by: Pankaj Gupta +Signed-off-by: Chenyi Qiang +Link: https://lore.kernel.org/r/20250612082747.51539-5-chenyi.qiang@intel.com +[peterx: squash fixup from Chenyi to fix builds] +Signed-off-by: Peter Xu +(cherry picked from commit 5d6483edaa9232d8f3709f68c8eab4bc2033fb70) +Signed-off-by: Paolo Bonzini + +Conflicts: system/->sysemu/ or exec/, context, class_init argument is not const +--- + MAINTAINERS | 1 + + include/exec/ramblock.h | 22 ++ + system/meson.build | 1 + + system/ram-block-attributes.c | 444 ++++++++++++++++++++++++++++++++++ + system/trace-events | 3 + + 5 files changed, 471 insertions(+) + create mode 100644 system/ram-block-attributes.c + +diff --git a/MAINTAINERS b/MAINTAINERS +index f7b7ceffc4..87ba88da84 100644 +--- a/MAINTAINERS ++++ b/MAINTAINERS +@@ -3056,6 +3056,7 @@ F: system/memory.c + F: system/memory_mapping.c + F: system/physmem.c + F: include/exec/memory-internal.h ++F: system/ram-block-attributes.c + F: scripts/coccinelle/memory-region-housekeeping.cocci + + Memory devices +diff --git a/include/exec/ramblock.h b/include/exec/ramblock.h +index 0babd105c0..9ae774d268 100644 +--- a/include/exec/ramblock.h ++++ b/include/exec/ramblock.h +@@ -23,6 +23,10 @@ + #include "cpu-common.h" + #include "qemu/rcu.h" + #include "exec/ramlist.h" ++#include "sysemu/hostmem.h" ++ ++#define TYPE_RAM_BLOCK_ATTRIBUTES "ram-block-attributes" ++OBJECT_DECLARE_SIMPLE_TYPE(RamBlockAttributes, RAM_BLOCK_ATTRIBUTES) + + struct RAMBlock { + struct rcu_head rcu; +@@ -90,5 +94,23 @@ struct RAMBlock { + */ + ram_addr_t postcopy_length; + }; ++ ++struct RamBlockAttributes { ++ Object parent; ++ ++ RAMBlock *ram_block; ++ ++ /* 1-setting of the bitmap represents ram is populated (shared) */ ++ unsigned bitmap_size; ++ unsigned long *bitmap; ++ ++ QLIST_HEAD(, RamDiscardListener) rdl_list; ++}; ++ ++RamBlockAttributes *ram_block_attributes_create(RAMBlock *ram_block); ++void ram_block_attributes_destroy(RamBlockAttributes *attr); ++int ram_block_attributes_state_change(RamBlockAttributes *attr, uint64_t offset, ++ uint64_t size, bool to_discard); ++ + #endif + #endif +diff --git a/system/meson.build b/system/meson.build +index a296270cb0..b13d9e71ff 100644 +--- a/system/meson.build ++++ b/system/meson.build +@@ -16,6 +16,7 @@ system_ss.add(files( + 'dirtylimit.c', + 'dma-helpers.c', + 'globals.c', ++ 'ram-block-attributes.c', + 'memory_mapping.c', + 'qdev-monitor.c', + 'qtest.c', +diff --git a/system/ram-block-attributes.c b/system/ram-block-attributes.c +new file mode 100644 +index 0000000000..0bded54e9c +--- /dev/null ++++ b/system/ram-block-attributes.c +@@ -0,0 +1,444 @@ ++/* ++ * QEMU ram block attributes ++ * ++ * Copyright Intel ++ * ++ * Author: ++ * Chenyi Qiang ++ * ++ * SPDX-License-Identifier: GPL-2.0-or-later ++ */ ++ ++#include "qemu/osdep.h" ++#include "qemu/error-report.h" ++#include "exec/ramblock.h" ++#include "trace.h" ++ ++OBJECT_DEFINE_SIMPLE_TYPE_WITH_INTERFACES(RamBlockAttributes, ++ ram_block_attributes, ++ RAM_BLOCK_ATTRIBUTES, ++ OBJECT, ++ { TYPE_RAM_DISCARD_MANAGER }, ++ { }) ++ ++static size_t ++ram_block_attributes_get_block_size(const RamBlockAttributes *attr) ++{ ++ /* ++ * Because page conversion could be manipulated in the size of at least 4K ++ * or 4K aligned, Use the host page size as the granularity to track the ++ * memory attribute. ++ */ ++ g_assert(attr && attr->ram_block); ++ g_assert(attr->ram_block->page_size == qemu_real_host_page_size()); ++ return attr->ram_block->page_size; ++} ++ ++ ++static bool ++ram_block_attributes_rdm_is_populated(const RamDiscardManager *rdm, ++ const MemoryRegionSection *section) ++{ ++ const RamBlockAttributes *attr = RAM_BLOCK_ATTRIBUTES(rdm); ++ const size_t block_size = ram_block_attributes_get_block_size(attr); ++ const uint64_t first_bit = section->offset_within_region / block_size; ++ const uint64_t last_bit = ++ first_bit + int128_get64(section->size) / block_size - 1; ++ unsigned long first_discarded_bit; ++ ++ first_discarded_bit = find_next_zero_bit(attr->bitmap, last_bit + 1, ++ first_bit); ++ return first_discarded_bit > last_bit; ++} ++ ++typedef int (*ram_block_attributes_section_cb)(MemoryRegionSection *s, ++ void *arg); ++ ++static int ++ram_block_attributes_notify_populate_cb(MemoryRegionSection *section, ++ void *arg) ++{ ++ RamDiscardListener *rdl = arg; ++ ++ return rdl->notify_populate(rdl, section); ++} ++ ++static int ++ram_block_attributes_notify_discard_cb(MemoryRegionSection *section, ++ void *arg) ++{ ++ RamDiscardListener *rdl = arg; ++ ++ rdl->notify_discard(rdl, section); ++ return 0; ++} ++ ++static int ++ram_block_attributes_for_each_populated_section(const RamBlockAttributes *attr, ++ MemoryRegionSection *section, ++ void *arg, ++ ram_block_attributes_section_cb cb) ++{ ++ unsigned long first_bit, last_bit; ++ uint64_t offset, size; ++ const size_t block_size = ram_block_attributes_get_block_size(attr); ++ int ret = 0; ++ ++ first_bit = section->offset_within_region / block_size; ++ first_bit = find_next_bit(attr->bitmap, attr->bitmap_size, ++ first_bit); ++ ++ while (first_bit < attr->bitmap_size) { ++ MemoryRegionSection tmp = *section; ++ ++ offset = first_bit * block_size; ++ last_bit = find_next_zero_bit(attr->bitmap, attr->bitmap_size, ++ first_bit + 1) - 1; ++ size = (last_bit - first_bit + 1) * block_size; ++ ++ if (!memory_region_section_intersect_range(&tmp, offset, size)) { ++ break; ++ } ++ ++ ret = cb(&tmp, arg); ++ if (ret) { ++ error_report("%s: Failed to notify RAM discard listener: %s", ++ __func__, strerror(-ret)); ++ break; ++ } ++ ++ first_bit = find_next_bit(attr->bitmap, attr->bitmap_size, ++ last_bit + 2); ++ } ++ ++ return ret; ++} ++ ++static int ++ram_block_attributes_for_each_discarded_section(const RamBlockAttributes *attr, ++ MemoryRegionSection *section, ++ void *arg, ++ ram_block_attributes_section_cb cb) ++{ ++ unsigned long first_bit, last_bit; ++ uint64_t offset, size; ++ const size_t block_size = ram_block_attributes_get_block_size(attr); ++ int ret = 0; ++ ++ first_bit = section->offset_within_region / block_size; ++ first_bit = find_next_zero_bit(attr->bitmap, attr->bitmap_size, ++ first_bit); ++ ++ while (first_bit < attr->bitmap_size) { ++ MemoryRegionSection tmp = *section; ++ ++ offset = first_bit * block_size; ++ last_bit = find_next_bit(attr->bitmap, attr->bitmap_size, ++ first_bit + 1) - 1; ++ size = (last_bit - first_bit + 1) * block_size; ++ ++ if (!memory_region_section_intersect_range(&tmp, offset, size)) { ++ break; ++ } ++ ++ ret = cb(&tmp, arg); ++ if (ret) { ++ error_report("%s: Failed to notify RAM discard listener: %s", ++ __func__, strerror(-ret)); ++ break; ++ } ++ ++ first_bit = find_next_zero_bit(attr->bitmap, ++ attr->bitmap_size, ++ last_bit + 2); ++ } ++ ++ return ret; ++} ++ ++static uint64_t ++ram_block_attributes_rdm_get_min_granularity(const RamDiscardManager *rdm, ++ const MemoryRegion *mr) ++{ ++ const RamBlockAttributes *attr = RAM_BLOCK_ATTRIBUTES(rdm); ++ ++ g_assert(mr == attr->ram_block->mr); ++ return ram_block_attributes_get_block_size(attr); ++} ++ ++static void ++ram_block_attributes_rdm_register_listener(RamDiscardManager *rdm, ++ RamDiscardListener *rdl, ++ MemoryRegionSection *section) ++{ ++ RamBlockAttributes *attr = RAM_BLOCK_ATTRIBUTES(rdm); ++ int ret; ++ ++ g_assert(section->mr == attr->ram_block->mr); ++ rdl->section = memory_region_section_new_copy(section); ++ ++ QLIST_INSERT_HEAD(&attr->rdl_list, rdl, next); ++ ++ ret = ram_block_attributes_for_each_populated_section(attr, section, rdl, ++ ram_block_attributes_notify_populate_cb); ++ if (ret) { ++ error_report("%s: Failed to register RAM discard listener: %s", ++ __func__, strerror(-ret)); ++ exit(1); ++ } ++} ++ ++static void ++ram_block_attributes_rdm_unregister_listener(RamDiscardManager *rdm, ++ RamDiscardListener *rdl) ++{ ++ RamBlockAttributes *attr = RAM_BLOCK_ATTRIBUTES(rdm); ++ int ret; ++ ++ g_assert(rdl->section); ++ g_assert(rdl->section->mr == attr->ram_block->mr); ++ ++ if (rdl->double_discard_supported) { ++ rdl->notify_discard(rdl, rdl->section); ++ } else { ++ ret = ram_block_attributes_for_each_populated_section(attr, ++ rdl->section, rdl, ram_block_attributes_notify_discard_cb); ++ if (ret) { ++ error_report("%s: Failed to unregister RAM discard listener: %s", ++ __func__, strerror(-ret)); ++ exit(1); ++ } ++ } ++ ++ memory_region_section_free_copy(rdl->section); ++ rdl->section = NULL; ++ QLIST_REMOVE(rdl, next); ++} ++ ++typedef struct RamBlockAttributesReplayData { ++ ReplayRamDiscardState fn; ++ void *opaque; ++} RamBlockAttributesReplayData; ++ ++static int ram_block_attributes_rdm_replay_cb(MemoryRegionSection *section, ++ void *arg) ++{ ++ RamBlockAttributesReplayData *data = arg; ++ ++ return data->fn(section, data->opaque); ++} ++ ++static int ++ram_block_attributes_rdm_replay_populated(const RamDiscardManager *rdm, ++ MemoryRegionSection *section, ++ ReplayRamDiscardState replay_fn, ++ void *opaque) ++{ ++ RamBlockAttributes *attr = RAM_BLOCK_ATTRIBUTES(rdm); ++ RamBlockAttributesReplayData data = { .fn = replay_fn, .opaque = opaque }; ++ ++ g_assert(section->mr == attr->ram_block->mr); ++ return ram_block_attributes_for_each_populated_section(attr, section, &data, ++ ram_block_attributes_rdm_replay_cb); ++} ++ ++static int ++ram_block_attributes_rdm_replay_discarded(const RamDiscardManager *rdm, ++ MemoryRegionSection *section, ++ ReplayRamDiscardState replay_fn, ++ void *opaque) ++{ ++ RamBlockAttributes *attr = RAM_BLOCK_ATTRIBUTES(rdm); ++ RamBlockAttributesReplayData data = { .fn = replay_fn, .opaque = opaque }; ++ ++ g_assert(section->mr == attr->ram_block->mr); ++ return ram_block_attributes_for_each_discarded_section(attr, section, &data, ++ ram_block_attributes_rdm_replay_cb); ++} ++ ++static bool ++ram_block_attributes_is_valid_range(RamBlockAttributes *attr, uint64_t offset, ++ uint64_t size) ++{ ++ MemoryRegion *mr = attr->ram_block->mr; ++ ++ g_assert(mr); ++ ++ uint64_t region_size = memory_region_size(mr); ++ const size_t block_size = ram_block_attributes_get_block_size(attr); ++ ++ if (!QEMU_IS_ALIGNED(offset, block_size) || ++ !QEMU_IS_ALIGNED(size, block_size)) { ++ return false; ++ } ++ if (offset + size <= offset) { ++ return false; ++ } ++ if (offset + size > region_size) { ++ return false; ++ } ++ return true; ++} ++ ++static void ram_block_attributes_notify_discard(RamBlockAttributes *attr, ++ uint64_t offset, ++ uint64_t size) ++{ ++ RamDiscardListener *rdl; ++ ++ QLIST_FOREACH(rdl, &attr->rdl_list, next) { ++ MemoryRegionSection tmp = *rdl->section; ++ ++ if (!memory_region_section_intersect_range(&tmp, offset, size)) { ++ continue; ++ } ++ rdl->notify_discard(rdl, &tmp); ++ } ++} ++ ++static int ++ram_block_attributes_notify_populate(RamBlockAttributes *attr, ++ uint64_t offset, uint64_t size) ++{ ++ RamDiscardListener *rdl; ++ int ret = 0; ++ ++ QLIST_FOREACH(rdl, &attr->rdl_list, next) { ++ MemoryRegionSection tmp = *rdl->section; ++ ++ if (!memory_region_section_intersect_range(&tmp, offset, size)) { ++ continue; ++ } ++ ret = rdl->notify_populate(rdl, &tmp); ++ if (ret) { ++ break; ++ } ++ } ++ ++ return ret; ++} ++ ++int ram_block_attributes_state_change(RamBlockAttributes *attr, ++ uint64_t offset, uint64_t size, ++ bool to_discard) ++{ ++ const size_t block_size = ram_block_attributes_get_block_size(attr); ++ const unsigned long first_bit = offset / block_size; ++ const unsigned long nbits = size / block_size; ++ const unsigned long last_bit = first_bit + nbits - 1; ++ const bool is_discarded = find_next_bit(attr->bitmap, attr->bitmap_size, ++ first_bit) > last_bit; ++ const bool is_populated = find_next_zero_bit(attr->bitmap, ++ attr->bitmap_size, first_bit) > last_bit; ++ unsigned long bit; ++ int ret = 0; ++ ++ if (!ram_block_attributes_is_valid_range(attr, offset, size)) { ++ error_report("%s, invalid range: offset 0x%" PRIx64 ", size " ++ "0x%" PRIx64, __func__, offset, size); ++ return -EINVAL; ++ } ++ ++ trace_ram_block_attributes_state_change(offset, size, ++ is_discarded ? "discarded" : ++ is_populated ? "populated" : ++ "mixture", ++ to_discard ? "discarded" : ++ "populated"); ++ if (to_discard) { ++ if (is_discarded) { ++ /* Already private */ ++ } else if (is_populated) { ++ /* Completely shared */ ++ bitmap_clear(attr->bitmap, first_bit, nbits); ++ ram_block_attributes_notify_discard(attr, offset, size); ++ } else { ++ /* Unexpected mixture: process individual blocks */ ++ for (bit = first_bit; bit < first_bit + nbits; bit++) { ++ if (!test_bit(bit, attr->bitmap)) { ++ continue; ++ } ++ clear_bit(bit, attr->bitmap); ++ ram_block_attributes_notify_discard(attr, bit * block_size, ++ block_size); ++ } ++ } ++ } else { ++ if (is_populated) { ++ /* Already shared */ ++ } else if (is_discarded) { ++ /* Completely private */ ++ bitmap_set(attr->bitmap, first_bit, nbits); ++ ret = ram_block_attributes_notify_populate(attr, offset, size); ++ } else { ++ /* Unexpected mixture: process individual blocks */ ++ for (bit = first_bit; bit < first_bit + nbits; bit++) { ++ if (test_bit(bit, attr->bitmap)) { ++ continue; ++ } ++ set_bit(bit, attr->bitmap); ++ ret = ram_block_attributes_notify_populate(attr, ++ bit * block_size, ++ block_size); ++ if (ret) { ++ break; ++ } ++ } ++ } ++ } ++ ++ return ret; ++} ++ ++RamBlockAttributes *ram_block_attributes_create(RAMBlock *ram_block) ++{ ++ const int block_size = qemu_real_host_page_size(); ++ RamBlockAttributes *attr; ++ MemoryRegion *mr = ram_block->mr; ++ ++ attr = RAM_BLOCK_ATTRIBUTES(object_new(TYPE_RAM_BLOCK_ATTRIBUTES)); ++ ++ attr->ram_block = ram_block; ++ if (memory_region_set_ram_discard_manager(mr, RAM_DISCARD_MANAGER(attr))) { ++ object_unref(OBJECT(attr)); ++ return NULL; ++ } ++ attr->bitmap_size = ++ ROUND_UP(int128_get64(mr->size), block_size) / block_size; ++ attr->bitmap = bitmap_new(attr->bitmap_size); ++ ++ return attr; ++} ++ ++void ram_block_attributes_destroy(RamBlockAttributes *attr) ++{ ++ g_assert(attr); ++ ++ g_free(attr->bitmap); ++ memory_region_set_ram_discard_manager(attr->ram_block->mr, NULL); ++ object_unref(OBJECT(attr)); ++} ++ ++static void ram_block_attributes_init(Object *obj) ++{ ++ RamBlockAttributes *attr = RAM_BLOCK_ATTRIBUTES(obj); ++ ++ QLIST_INIT(&attr->rdl_list); ++} ++ ++static void ram_block_attributes_finalize(Object *obj) ++{ ++} ++ ++static void ram_block_attributes_class_init(ObjectClass *klass, ++ void *data) ++{ ++ RamDiscardManagerClass *rdmc = RAM_DISCARD_MANAGER_CLASS(klass); ++ ++ rdmc->get_min_granularity = ram_block_attributes_rdm_get_min_granularity; ++ rdmc->register_listener = ram_block_attributes_rdm_register_listener; ++ rdmc->unregister_listener = ram_block_attributes_rdm_unregister_listener; ++ rdmc->is_populated = ram_block_attributes_rdm_is_populated; ++ rdmc->replay_populated = ram_block_attributes_rdm_replay_populated; ++ rdmc->replay_discarded = ram_block_attributes_rdm_replay_discarded; ++} +diff --git a/system/trace-events b/system/trace-events +index 2ed1d59b1f..9fd7217472 100644 +--- a/system/trace-events ++++ b/system/trace-events +@@ -44,3 +44,6 @@ dirtylimit_state_finalize(void) + dirtylimit_throttle_pct(int cpu_index, uint64_t pct, int64_t time_us) "CPU[%d] throttle percent: %" PRIu64 ", throttle adjust time %"PRIi64 " us" + dirtylimit_set_vcpu(int cpu_index, uint64_t quota) "CPU[%d] set dirty page rate limit %"PRIu64 + dirtylimit_vcpu_execute(int cpu_index, int64_t sleep_time_us) "CPU[%d] sleep %"PRIi64 " us" ++ ++# ram-block-attributes.c ++ram_block_attributes_state_change(uint64_t offset, uint64_t size, const char *from, const char *to) "offset 0x%"PRIx64" size 0x%"PRIx64" from '%s' to '%s'" +-- +2.50.1 + diff --git a/SOURCES/kvm-rbd-Fix-.bdrv_get_specific_info-implementation.patch b/SOURCES/kvm-rbd-Fix-.bdrv_get_specific_info-implementation.patch index ae5e6b0..e071b64 100644 --- a/SOURCES/kvm-rbd-Fix-.bdrv_get_specific_info-implementation.patch +++ b/SOURCES/kvm-rbd-Fix-.bdrv_get_specific_info-implementation.patch @@ -1,14 +1,14 @@ -From 04a3beab85453e901e76c64b8a7164bfb9fbbc4d Mon Sep 17 00:00:00 2001 +From 181b9ca805f3ae09c24a925eea0460525f30c90e Mon Sep 17 00:00:00 2001 From: Kevin Wolf Date: Mon, 11 Aug 2025 15:40:10 +0200 Subject: [PATCH] rbd: Fix .bdrv_get_specific_info implementation RH-Author: Kevin Wolf -RH-MergeRequest: 473: rbd: Fix .bdrv_get_specific_info implementation -RH-Jira: RHEL-108725 -RH-Acked-by: Stefan Hajnoczi +RH-MergeRequest: 400: rbd: Fix .bdrv_get_specific_info implementation +RH-Jira: RHEL-108726 RH-Acked-by: Hanna Czenczek -RH-Commit: [1/1] 183cf2d34cd1aa32e5f234139c00f6cf925edd56 (kmwolf/rhel-qemu-kvm) +RH-Acked-by: Stefan Hajnoczi +RH-Commit: [1/1] 5a488d6e2355adcec7fc4fd686c6be001808a146 (kmwolf/centos-qemu-kvm) qemu_rbd_get_specific_info() has at least two problems: @@ -57,7 +57,7 @@ Signed-off-by: Kevin Wolf 2 files changed, 76 insertions(+), 37 deletions(-) diff --git a/block/rbd.c b/block/rbd.c -index 9c0fd0cb3f..fa9aab12ab 100644 +index 627f8eb05a..d5546da71b 100644 --- a/block/rbd.c +++ b/block/rbd.c @@ -99,6 +99,14 @@ typedef struct BDRVRBDState { @@ -249,7 +249,7 @@ index 9c0fd0cb3f..fa9aab12ab 100644 return spec_info; diff --git a/qapi/block-core.json b/qapi/block-core.json -index c1af3d1f7d..a2fa277245 100644 +index 3969c60b93..15b91e2d4a 100644 --- a/qapi/block-core.json +++ b/qapi/block-core.json @@ -158,7 +158,14 @@ diff --git a/SOURCES/kvm-redhat-Enable-virtio-mem-on-s390x.patch b/SOURCES/kvm-redhat-Enable-virtio-mem-on-s390x.patch new file mode 100644 index 0000000..ca56520 --- /dev/null +++ b/SOURCES/kvm-redhat-Enable-virtio-mem-on-s390x.patch @@ -0,0 +1,36 @@ +From 7300a435547b7e999227648fd1451db00e9c4867 Mon Sep 17 00:00:00 2001 +From: Thomas Huth +Date: Mon, 24 Mar 2025 18:09:26 +0100 +Subject: [PATCH 26/26] redhat: Enable virtio-mem on s390x + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [26/26] 076b44c8f0262e903c5e17eda676614aec6f5c98 (thuth/qemu-kvm-cs) + +JIRA: https://issues.redhat.com/browse/RHEL-72977 + +Enable virtio-mem on s390x now, too. + +Signed-off-by: Thomas Huth +--- + configs/devices/s390x-softmmu/s390x-rh-devices.mak | 1 + + 1 file changed, 1 insertion(+) + +diff --git a/configs/devices/s390x-softmmu/s390x-rh-devices.mak b/configs/devices/s390x-softmmu/s390x-rh-devices.mak +index 24cf6dbd03..834281d872 100644 +--- a/configs/devices/s390x-softmmu/s390x-rh-devices.mak ++++ b/configs/devices/s390x-softmmu/s390x-rh-devices.mak +@@ -12,6 +12,7 @@ CONFIG_VFIO_CCW=y + CONFIG_VFIO_PCI=y + CONFIG_VHOST_USER=y + CONFIG_VIRTIO_CCW=y ++CONFIG_VIRTIO_MEM=y + CONFIG_WDT_DIAG288=y + CONFIG_VHOST_VSOCK=y + CONFIG_VHOST_USER_VSOCK=y +-- +2.48.1 + diff --git a/SOURCES/kvm-redhat-allow-5-level-paging-for-TDX-VMs.patch b/SOURCES/kvm-redhat-allow-5-level-paging-for-TDX-VMs.patch new file mode 100644 index 0000000..18fd9cf --- /dev/null +++ b/SOURCES/kvm-redhat-allow-5-level-paging-for-TDX-VMs.patch @@ -0,0 +1,33 @@ +From 3cb78d36244833d5e11e88de33bbdbe93a641e16 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:50 +0200 +Subject: [PATCH 110/115] redhat: allow 5-level paging for TDX VMs + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [110/115] dd69bc652e2a735234165832ea4bc674753d7fb7 (bonzini/rhel-qemu-kvm) + +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 1 + + 1 file changed, 1 insertion(+) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 2ff5211794..e65b1727cf 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -754,6 +754,7 @@ static void tdx_cpu_instance_init(X86ConfidentialGuest *cg, CPUState *cpu) + } + + object_property_set_bool(OBJECT(cpu), "pmu", false, &error_abort); ++ object_property_set_int(OBJECT(cpu), "host-phys-bits-limit", 0, &error_abort); + + /* invtsc is fixed1 for TD guest */ + object_property_set_bool(OBJECT(cpu), "invtsc", true, &error_abort); +-- +2.50.1 + diff --git a/SOURCES/kvm-redhat-enable-CONFIG_TDX.patch b/SOURCES/kvm-redhat-enable-CONFIG_TDX.patch new file mode 100644 index 0000000..441d61b --- /dev/null +++ b/SOURCES/kvm-redhat-enable-CONFIG_TDX.patch @@ -0,0 +1,33 @@ +From 2e9c06611ccabf1a31dfe666814395ce48adae2e Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:50 +0200 +Subject: [PATCH 109/115] redhat: enable CONFIG_TDX + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [109/115] 66eac186ba4c28ffd4b8612053feab81ad7ac608 (bonzini/rhel-qemu-kvm) + +Signed-off-by: Paolo Bonzini +--- + configs/devices/x86_64-softmmu/x86_64-rh-devices.mak | 1 + + 1 file changed, 1 insertion(+) + +diff --git a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak +index 2b15fdc2db..6379077253 100644 +--- a/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak ++++ b/configs/devices/x86_64-softmmu/x86_64-rh-devices.mak +@@ -73,6 +73,7 @@ CONFIG_SERIAL_PCI=y + CONFIG_SEV=y + CONFIG_SMBIOS=y + CONFIG_SMBUS_EEPROM=y ++CONFIG_TDX=y + CONFIG_TEST_DEVICES=y + CONFIG_USB=y + CONFIG_USB_EHCI=y +-- +2.50.1 + diff --git a/SOURCES/kvm-redhat-target-i386-add-CPUID-and-MSR-bits-from-Clear.patch b/SOURCES/kvm-redhat-target-i386-add-CPUID-and-MSR-bits-from-Clear.patch new file mode 100644 index 0000000..5c057f6 --- /dev/null +++ b/SOURCES/kvm-redhat-target-i386-add-CPUID-and-MSR-bits-from-Clear.patch @@ -0,0 +1,99 @@ +From 9eae8d0c65d59aacbbd65f522c9d829f5f658c59 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:21:11 +0200 +Subject: [PATCH 021/115] redhat: target/i386: add CPUID and MSR bits from + Clearwater Forest + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [21/115] 9f8fe79556c5a52cec955596a3ec8691c05fd65d (bonzini/rhel-qemu-kvm) + +They are used by TDX. But do not add the model yet. + +Signed-off-by: Tao Su +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/20250121020650.1899618-4-tao1.su@linux.intel.com +Signed-off-by: Paolo Bonzini +(extracted from commit 56e84d898f17606b5d88778726466540af96b234) +--- + target/i386/cpu.h | 33 +++++++++++++++++++++++++++------ + 1 file changed, 27 insertions(+), 6 deletions(-) + +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index cc42a2c520..ee1a1b6622 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -951,6 +951,12 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w); + /* Speculative Store Bypass Disable */ + #define CPUID_7_0_EDX_SPEC_CTRL_SSBD (1U << 31) + ++/* SHA512 Instruction */ ++#define CPUID_7_1_EAX_SHA512 (1U << 0) ++/* SM3 Instruction */ ++#define CPUID_7_1_EAX_SM3 (1U << 1) ++/* SM4 Instruction */ ++#define CPUID_7_1_EAX_SM4 (1U << 2) + /* AVX VNNI Instruction */ + #define CPUID_7_1_EAX_AVX_VNNI (1U << 4) + /* AVX512 BFloat16 Instruction */ +@@ -963,6 +969,12 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w); + #define CPUID_7_1_EAX_FSRS (1U << 11) + /* Fast Short REP CMPS/SCAS */ + #define CPUID_7_1_EAX_FSRC (1U << 12) ++/* Flexible return and event delivery (FRED) */ ++#define CPUID_7_1_EAX_FRED (1U << 17) ++/* Load into IA32_KERNEL_GS_BASE (LKGS) */ ++#define CPUID_7_1_EAX_LKGS (1U << 18) ++/* Non-Serializing Write to Model Specific Register (WRMSRNS) */ ++#define CPUID_7_1_EAX_WRMSRNS (1U << 19) + /* Support Tile Computational Operations on FP16 Numbers */ + #define CPUID_7_1_EAX_AMX_FP16 (1U << 21) + /* Support for VPMADD52[H,L]UQ */ +@@ -976,17 +988,23 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w); + #define CPUID_7_1_EDX_AVX_NE_CONVERT (1U << 5) + /* AMX COMPLEX Instructions */ + #define CPUID_7_1_EDX_AMX_COMPLEX (1U << 8) ++/* AVX-VNNI-INT16 Instructions */ ++#define CPUID_7_1_EDX_AVX_VNNI_INT16 (1U << 10) + /* PREFETCHIT0/1 Instructions */ + #define CPUID_7_1_EDX_PREFETCHITI (1U << 14) + /* Support for Advanced Vector Extensions 10 */ + #define CPUID_7_1_EDX_AVX10 (1U << 19) +-/* Flexible return and event delivery (FRED) */ +-#define CPUID_7_1_EAX_FRED (1U << 17) +-/* Load into IA32_KERNEL_GS_BASE (LKGS) */ +-#define CPUID_7_1_EAX_LKGS (1U << 18) +-/* Non-Serializing Write to Model Specific Register (WRMSRNS) */ +-#define CPUID_7_1_EAX_WRMSRNS (1U << 19) + ++/* Indicate bit 7 of the IA32_SPEC_CTRL MSR is supported */ ++#define CPUID_7_2_EDX_PSFD (1U << 0) ++/* Indicate bits 3 and 4 of the IA32_SPEC_CTRL MSR are supported */ ++#define CPUID_7_2_EDX_IPRED_CTRL (1U << 1) ++/* Indicate bits 5 and 6 of the IA32_SPEC_CTRL MSR are supported */ ++#define CPUID_7_2_EDX_RRSBA_CTRL (1U << 2) ++/* Indicate bit 8 of the IA32_SPEC_CTRL MSR is supported */ ++#define CPUID_7_2_EDX_DDPD_U (1U << 3) ++/* Indicate bit 10 of the IA32_SPEC_CTRL MSR is supported */ ++#define CPUID_7_2_EDX_BHI_CTRL (1U << 4) + /* Do not exhibit MXCSR Configuration Dependent Timing (MCDT) behavior */ + #define CPUID_7_2_EDX_MCDT_NO (1U << 5) + +@@ -1118,7 +1136,10 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w); + #define MSR_ARCH_CAP_FBSDP_NO (1U << 14) + #define MSR_ARCH_CAP_PSDP_NO (1U << 15) + #define MSR_ARCH_CAP_FB_CLEAR (1U << 17) ++#define MSR_ARCH_CAP_BHI_NO (1U << 20) + #define MSR_ARCH_CAP_PBRSB_NO (1U << 24) ++#define MSR_ARCH_CAP_GDS_NO (1U << 26) ++#define MSR_ARCH_CAP_RFDS_NO (1U << 27) + + #define MSR_CORE_CAP_SPLIT_LOCK_DETECT (1U << 5) + +-- +2.50.1 + diff --git a/SOURCES/kvm-reset-Add-RESET_TYPE_WAKEUP.patch b/SOURCES/kvm-reset-Add-RESET_TYPE_WAKEUP.patch new file mode 100644 index 0000000..bcdc6ac --- /dev/null +++ b/SOURCES/kvm-reset-Add-RESET_TYPE_WAKEUP.patch @@ -0,0 +1,94 @@ +From 2de79d978c2cd29ad686dd91e74a86dbf2121f1f Mon Sep 17 00:00:00 2001 +From: Juraj Marcin +Date: Wed, 4 Sep 2024 12:37:13 +0200 +Subject: [PATCH 06/26] reset: Add RESET_TYPE_WAKEUP + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [6/26] 6169fe25bfa5715340c180ee8711d0ad61832106 (thuth/qemu-kvm-cs) + +Some devices need to distinguish cold start reset from waking up from a +suspended state. This patch adds new value to the enum, and updates the +i386 wakeup method to use this new reset type. + +Message-ID: <20240904103722.946194-3-jmarcin@redhat.com> +Reviewed-by: David Hildenbrand +Signed-off-by: Juraj Marcin +Signed-off-by: David Hildenbrand +(cherry picked from commit 759cbb4ee971da13ddfa8ad73befc2351d542044) +Signed-off-by: Thomas Huth +--- + docs/devel/reset.rst | 12 +++++++++++- + hw/i386/pc.c | 2 +- + include/hw/resettable.h | 2 ++ + 3 files changed, 14 insertions(+), 2 deletions(-) + +diff --git a/docs/devel/reset.rst b/docs/devel/reset.rst +index d2799eba7a..44bd51b42e 100644 +--- a/docs/devel/reset.rst ++++ b/docs/devel/reset.rst +@@ -44,6 +44,17 @@ The Resettable interface handles reset types with an enum ``ResetType``: + value on each cold reset, such as RNG seed information, and which they + must not reinitialize on a snapshot-load reset. + ++``RESET_TYPE_WAKEUP`` ++ If the machine supports waking up from a suspended state and needs to reset ++ its devices during wake-up (from the ``MachineClass::wakeup()`` method), this ++ reset type should be used for such a request. Devices can utilize this reset ++ type to differentiate the reset requested during machine wake-up from other ++ reset requests. For example, RAM content must not be lost during wake-up, and ++ memory devices like virtio-mem that provide additional RAM must not reset ++ such state during wake-ups, but might do so during cold resets. However, this ++ reset type should not be used for wake-up detection, as not every machine ++ type issues a device reset request during wake-up. ++ + ``RESET_TYPE_S390_CPU_NORMAL`` + This is only used for S390 CPU objects; it clears interrupts, stops + processing, and clears the TLB, but does not touch register contents. +@@ -53,7 +64,6 @@ The Resettable interface handles reset types with an enum ``ResetType``: + ``RESET_TYPE_S390_CPU_NORMAL`` does and also clears the PSW, prefix, + FPC, timer and control registers. It does not touch gprs, fprs or acrs. + +- + Devices which implement reset methods must treat any unknown ``ResetType`` + as equivalent to ``RESET_TYPE_COLD``; this will reduce the amount of + existing code we need to change if we add more types in future. +diff --git a/hw/i386/pc.c b/hw/i386/pc.c +index fedcf2a65f..fa9f16cbaf 100644 +--- a/hw/i386/pc.c ++++ b/hw/i386/pc.c +@@ -1889,7 +1889,7 @@ static void pc_machine_reset(MachineState *machine, ResetType type) + static void pc_machine_wakeup(MachineState *machine) + { + cpu_synchronize_all_states(); +- pc_machine_reset(machine, RESET_TYPE_COLD); ++ pc_machine_reset(machine, RESET_TYPE_WAKEUP); + cpu_synchronize_all_post_reset(); + } + +diff --git a/include/hw/resettable.h b/include/hw/resettable.h +index 83b561fc83..cf37cd5ead 100644 +--- a/include/hw/resettable.h ++++ b/include/hw/resettable.h +@@ -29,6 +29,7 @@ typedef struct ResettableState ResettableState; + * Types of reset. + * + * + Cold: reset resulting from a power cycle of the object. ++ * + Wakeup: reset resulting from a wake-up from a suspended state. + * + * TODO: Support has to be added to handle more types. In particular, + * ResettableState structure needs to be expanded. +@@ -36,6 +37,7 @@ typedef struct ResettableState ResettableState; + typedef enum ResetType { + RESET_TYPE_COLD, + RESET_TYPE_SNAPSHOT_LOAD, ++ RESET_TYPE_WAKEUP, + RESET_TYPE_S390_CPU_INITIAL, + RESET_TYPE_S390_CPU_NORMAL, + } ResetType; +-- +2.48.1 + diff --git a/SOURCES/kvm-reset-Use-ResetType-for-qemu_devices_reset-and-Machi.patch b/SOURCES/kvm-reset-Use-ResetType-for-qemu_devices_reset-and-Machi.patch new file mode 100644 index 0000000..2b0b933 --- /dev/null +++ b/SOURCES/kvm-reset-Use-ResetType-for-qemu_devices_reset-and-Machi.patch @@ -0,0 +1,360 @@ +From 8d48193b5a661f31c1c1db068d241b31ae379339 Mon Sep 17 00:00:00 2001 +From: Juraj Marcin +Date: Wed, 4 Sep 2024 12:37:12 +0200 +Subject: [PATCH 05/26] reset: Use ResetType for qemu_devices_reset() and + MachineClass::reset() + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [5/26] ea1324b27885d979bcc54cc355dbdf940686776c (thuth/qemu-kvm-cs) + +Currently, both qemu_devices_reset() and MachineClass::reset() use +ShutdownCause for the reason of the reset. However, the Resettable +interface uses ResetState, so ShutdownCause needs to be translated to +ResetType somewhere. Translating it qemu_devices_reset() makes adding +new reset types harder, as they cannot always be matched to a single +ShutdownCause here, and devices may need to check the ResetType to +determine what to reset and if to reset at all. + +This patch moves this translation up in the call stack to +qemu_system_reset() and updates all MachineClass children to use the +ResetType instead. + +Message-ID: <20240904103722.946194-2-jmarcin@redhat.com> +Reviewed-by: David Hildenbrand +Reviewed-by: Peter Maydell +Signed-off-by: Juraj Marcin +Signed-off-by: David Hildenbrand +(cherry picked from commit 1b063fe2df002052cc2d10799764979b8c583480) +Signed-off-by: Thomas Huth +--- + hw/arm/aspeed.c | 4 ++-- + hw/arm/mps2-tz.c | 4 ++-- + hw/core/reset.c | 5 +---- + hw/hppa/machine.c | 4 ++-- + hw/i386/microvm.c | 4 ++-- + hw/i386/pc.c | 6 +++--- + hw/ppc/pegasos2.c | 4 ++-- + hw/ppc/pnv.c | 4 ++-- + hw/ppc/spapr.c | 6 +++--- + hw/s390x/s390-virtio-ccw.c | 4 ++-- + include/hw/boards.h | 3 ++- + include/sysemu/reset.h | 5 +++-- + system/runstate.c | 13 +++++++++++-- + 13 files changed, 37 insertions(+), 29 deletions(-) + +diff --git a/hw/arm/aspeed.c b/hw/arm/aspeed.c +index fd5603f7aa..cbca7685da 100644 +--- a/hw/arm/aspeed.c ++++ b/hw/arm/aspeed.c +@@ -1529,12 +1529,12 @@ static void aspeed_machine_bletchley_class_init(ObjectClass *oc, void *data) + aspeed_machine_class_init_cpus_defaults(mc); + } + +-static void fby35_reset(MachineState *state, ShutdownCause reason) ++static void fby35_reset(MachineState *state, ResetType type) + { + AspeedMachineState *bmc = ASPEED_MACHINE(state); + AspeedGPIOState *gpio = &bmc->soc->gpio; + +- qemu_devices_reset(reason); ++ qemu_devices_reset(type); + + /* Board ID: 7 (Class-1, 4 slots) */ + object_property_set_bool(OBJECT(gpio), "gpioV4", true, &error_fatal); +diff --git a/hw/arm/mps2-tz.c b/hw/arm/mps2-tz.c +index aec57c0d68..8edf57a66d 100644 +--- a/hw/arm/mps2-tz.c ++++ b/hw/arm/mps2-tz.c +@@ -1254,7 +1254,7 @@ static void mps2_set_remap(Object *obj, const char *value, Error **errp) + } + } + +-static void mps2_machine_reset(MachineState *machine, ShutdownCause reason) ++static void mps2_machine_reset(MachineState *machine, ResetType type) + { + MPS2TZMachineState *mms = MPS2TZ_MACHINE(machine); + +@@ -1264,7 +1264,7 @@ static void mps2_machine_reset(MachineState *machine, ShutdownCause reason) + * reset see the correct mapping. + */ + remap_memory(mms, mms->remap); +- qemu_devices_reset(reason); ++ qemu_devices_reset(type); + } + + static void mps2tz_class_init(ObjectClass *oc, void *data) +diff --git a/hw/core/reset.c b/hw/core/reset.c +index 58dfc8db3d..14a2639fbf 100644 +--- a/hw/core/reset.c ++++ b/hw/core/reset.c +@@ -170,11 +170,8 @@ void qemu_unregister_resettable(Object *obj) + resettable_container_remove(get_root_reset_container(), obj); + } + +-void qemu_devices_reset(ShutdownCause reason) ++void qemu_devices_reset(ResetType type) + { +- ResetType type = (reason == SHUTDOWN_CAUSE_SNAPSHOT_LOAD) ? +- RESET_TYPE_SNAPSHOT_LOAD : RESET_TYPE_COLD; +- + /* Reset the simulation */ + resettable_reset(OBJECT(get_root_reset_container()), type); + } +diff --git a/hw/hppa/machine.c b/hw/hppa/machine.c +index 5d0a8739de..8259fe2e38 100644 +--- a/hw/hppa/machine.c ++++ b/hw/hppa/machine.c +@@ -642,12 +642,12 @@ static void machine_HP_C3700_init(MachineState *machine) + machine_HP_common_init_tail(machine, pci_bus, translate); + } + +-static void hppa_machine_reset(MachineState *ms, ShutdownCause reason) ++static void hppa_machine_reset(MachineState *ms, ResetType type) + { + unsigned int smp_cpus = ms->smp.cpus; + int i; + +- qemu_devices_reset(reason); ++ qemu_devices_reset(type); + + /* Start all CPUs at the firmware entry point. + * Monarch CPU will initialize firmware, secondary CPUs +diff --git a/hw/i386/microvm.c b/hw/i386/microvm.c +index 40edcee7af..8ae4dff7f2 100644 +--- a/hw/i386/microvm.c ++++ b/hw/i386/microvm.c +@@ -462,7 +462,7 @@ static void microvm_machine_state_init(MachineState *machine) + microvm_devices_init(mms); + } + +-static void microvm_machine_reset(MachineState *machine, ShutdownCause reason) ++static void microvm_machine_reset(MachineState *machine, ResetType type) + { + MicrovmMachineState *mms = MICROVM_MACHINE(machine); + CPUState *cs; +@@ -475,7 +475,7 @@ static void microvm_machine_reset(MachineState *machine, ShutdownCause reason) + mms->kernel_cmdline_fixed = true; + } + +- qemu_devices_reset(reason); ++ qemu_devices_reset(type); + + CPU_FOREACH(cs) { + cpu = X86_CPU(cs); +diff --git a/hw/i386/pc.c b/hw/i386/pc.c +index fa0e42d072..fedcf2a65f 100644 +--- a/hw/i386/pc.c ++++ b/hw/i386/pc.c +@@ -1869,12 +1869,12 @@ static void pc_machine_initfn(Object *obj) + qemu_add_machine_init_done_notifier(&pcms->machine_done); + } + +-static void pc_machine_reset(MachineState *machine, ShutdownCause reason) ++static void pc_machine_reset(MachineState *machine, ResetType type) + { + CPUState *cs; + X86CPU *cpu; + +- qemu_devices_reset(reason); ++ qemu_devices_reset(type); + + /* Reset APIC after devices have been reset to cancel + * any changes that qemu_devices_reset() might have done. +@@ -1889,7 +1889,7 @@ static void pc_machine_reset(MachineState *machine, ShutdownCause reason) + static void pc_machine_wakeup(MachineState *machine) + { + cpu_synchronize_all_states(); +- pc_machine_reset(machine, SHUTDOWN_CAUSE_NONE); ++ pc_machine_reset(machine, RESET_TYPE_COLD); + cpu_synchronize_all_post_reset(); + } + +diff --git a/hw/ppc/pegasos2.c b/hw/ppc/pegasos2.c +index 9b0a6b70ab..8ff4a00c34 100644 +--- a/hw/ppc/pegasos2.c ++++ b/hw/ppc/pegasos2.c +@@ -291,14 +291,14 @@ static void pegasos2_superio_write(uint8_t addr, uint8_t val) + cpu_physical_memory_write(PCI1_IO_BASE + 0x3f1, &val, 1); + } + +-static void pegasos2_machine_reset(MachineState *machine, ShutdownCause reason) ++static void pegasos2_machine_reset(MachineState *machine, ResetType type) + { + Pegasos2MachineState *pm = PEGASOS2_MACHINE(machine); + void *fdt; + uint64_t d[2]; + int sz; + +- qemu_devices_reset(reason); ++ qemu_devices_reset(type); + if (!pm->vof) { + return; /* Firmware should set up machine so nothing to do */ + } +diff --git a/hw/ppc/pnv.c b/hw/ppc/pnv.c +index 3526852685..988fd55d88 100644 +--- a/hw/ppc/pnv.c ++++ b/hw/ppc/pnv.c +@@ -709,13 +709,13 @@ static void pnv_powerdown_notify(Notifier *n, void *opaque) + } + } + +-static void pnv_reset(MachineState *machine, ShutdownCause reason) ++static void pnv_reset(MachineState *machine, ResetType type) + { + PnvMachineState *pnv = PNV_MACHINE(machine); + IPMIBmc *bmc; + void *fdt; + +- qemu_devices_reset(reason); ++ qemu_devices_reset(type); + + /* + * The machine should provide by default an internal BMC simulator. +diff --git a/hw/ppc/spapr.c b/hw/ppc/spapr.c +index 29e66f1b3f..11c953669a 100644 +--- a/hw/ppc/spapr.c ++++ b/hw/ppc/spapr.c +@@ -1725,7 +1725,7 @@ void spapr_check_mmu_mode(bool guest_radix) + } + } + +-static void spapr_machine_reset(MachineState *machine, ShutdownCause reason) ++static void spapr_machine_reset(MachineState *machine, ResetType type) + { + SpaprMachineState *spapr = SPAPR_MACHINE(machine); + PowerPCCPU *first_ppc_cpu; +@@ -1733,7 +1733,7 @@ static void spapr_machine_reset(MachineState *machine, ShutdownCause reason) + void *fdt; + int rc; + +- if (reason != SHUTDOWN_CAUSE_SNAPSHOT_LOAD) { ++ if (type != RESET_TYPE_SNAPSHOT_LOAD) { + /* + * Record-replay snapshot load must not consume random, this was + * already replayed from initial machine reset. +@@ -1769,7 +1769,7 @@ static void spapr_machine_reset(MachineState *machine, ShutdownCause reason) + spapr_setup_hpt(spapr); + } + +- qemu_devices_reset(reason); ++ qemu_devices_reset(type); + + spapr_ovec_cleanup(spapr->ov5_cas); + spapr->ov5_cas = spapr_ovec_new(); +diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c +index ef2a9687c7..94cad1705b 100644 +--- a/hw/s390x/s390-virtio-ccw.c ++++ b/hw/s390x/s390-virtio-ccw.c +@@ -434,7 +434,7 @@ static void s390_pv_prepare_reset(S390CcwMachineState *ms) + s390_pv_prep_reset(); + } + +-static void s390_machine_reset(MachineState *machine, ShutdownCause reason) ++static void s390_machine_reset(MachineState *machine, ResetType type) + { + S390CcwMachineState *ms = S390_CCW_MACHINE(machine); + enum s390_reset reset_type; +@@ -466,7 +466,7 @@ static void s390_machine_reset(MachineState *machine, ShutdownCause reason) + * Device reset includes CPU clear resets so this has to be + * done AFTER the unprotect call above. + */ +- qemu_devices_reset(reason); ++ qemu_devices_reset(type); + s390_crypto_reset(); + + /* configure and start the ipl CPU only */ +diff --git a/include/hw/boards.h b/include/hw/boards.h +index ffefc0a625..fe011b1e86 100644 +--- a/include/hw/boards.h ++++ b/include/hw/boards.h +@@ -10,6 +10,7 @@ + #include "qemu/module.h" + #include "qom/object.h" + #include "hw/core/cpu.h" ++#include "hw/resettable.h" + + #define TYPE_MACHINE_SUFFIX "-machine" + +@@ -253,7 +254,7 @@ struct MachineClass { + const char *deprecation_reason; + + void (*init)(MachineState *state); +- void (*reset)(MachineState *state, ShutdownCause reason); ++ void (*reset)(MachineState *state, ResetType type); + void (*wakeup)(MachineState *state); + int (*kvm_type)(MachineState *machine, const char *arg); + +diff --git a/include/sysemu/reset.h b/include/sysemu/reset.h +index ae436044a9..0e297c0e02 100644 +--- a/include/sysemu/reset.h ++++ b/include/sysemu/reset.h +@@ -27,6 +27,7 @@ + #ifndef QEMU_SYSEMU_RESET_H + #define QEMU_SYSEMU_RESET_H + ++#include "hw/resettable.h" + #include "qapi/qapi-events-run-state.h" + + typedef void QEMUResetHandler(void *opaque); +@@ -110,7 +111,7 @@ void qemu_unregister_reset(QEMUResetHandler *func, void *opaque); + + /** + * qemu_devices_reset: Perform a complete system reset +- * @reason: reason for the reset ++ * @reason: type of the reset + * + * This function performs the low-level work needed to do a complete reset + * of the system (calling all the callbacks registered with +@@ -121,6 +122,6 @@ void qemu_unregister_reset(QEMUResetHandler *func, void *opaque); + * If you want to trigger a system reset from, for instance, a device + * model, don't use this function. Use qemu_system_reset_request(). + */ +-void qemu_devices_reset(ShutdownCause reason); ++void qemu_devices_reset(ResetType type); + + #endif +diff --git a/system/runstate.c b/system/runstate.c +index a0e2a5fd22..c2c9afa905 100644 +--- a/system/runstate.c ++++ b/system/runstate.c +@@ -32,6 +32,7 @@ + #include "exec/cpu-common.h" + #include "gdbstub/syscalls.h" + #include "hw/boards.h" ++#include "hw/resettable.h" + #include "migration/misc.h" + #include "migration/postcopy-ram.h" + #include "monitor/monitor.h" +@@ -507,15 +508,23 @@ static int qemu_debug_requested(void) + void qemu_system_reset(ShutdownCause reason) + { + MachineClass *mc; ++ ResetType type; + + mc = current_machine ? MACHINE_GET_CLASS(current_machine) : NULL; + + cpu_synchronize_all_states(); + ++ switch (reason) { ++ case SHUTDOWN_CAUSE_SNAPSHOT_LOAD: ++ type = RESET_TYPE_SNAPSHOT_LOAD; ++ break; ++ default: ++ type = RESET_TYPE_COLD; ++ } + if (mc && mc->reset) { +- mc->reset(current_machine, reason); ++ mc->reset(current_machine, type); + } else { +- qemu_devices_reset(reason); ++ qemu_devices_reset(type); + } + switch (reason) { + case SHUTDOWN_CAUSE_NONE: +-- +2.48.1 + diff --git a/SOURCES/kvm-rocker-do-not-pollute-the-namespace.patch b/SOURCES/kvm-rocker-do-not-pollute-the-namespace.patch new file mode 100644 index 0000000..cdaa569 --- /dev/null +++ b/SOURCES/kvm-rocker-do-not-pollute-the-namespace.patch @@ -0,0 +1,241 @@ +From b3366713a19f6a661c724ba3e10cf5a4226dc763 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:45 +0200 +Subject: [PATCH 025/115] rocker: do not pollute the namespace + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [25/115] fc62404804eee1d9d745359f98b4d7528d1c281b (bonzini/rhel-qemu-kvm) + +Do not leave the __le* macros defined, in fact do not use them at all. Fixes a +build failure on Alpine with the TDX patches: + +In file included from ../hw/net/rocker/rocker_of_dpa.c:25: +../hw/net/rocker/rocker_hw.h:14:16: error: conflicting types for 'uint64_t'; have '__u64' {aka 'long long unsigned int'} + 14 | #define __le64 uint64_t + | ^~~~~~~~ +In file included from /usr/include/stdint.h:20, + from ../include/qemu/osdep.h:111, + from ../hw/net/rocker/rocker_of_dpa.c:17: +/usr/include/bits/alltypes.h:136:25: note: previous declaration of 'uint64_t' with type 'uint64_t' {aka 'long unsigned int'} + 136 | typedef unsigned _Int64 uint64_t; + | ^~~~~~~~ + +because the Linux headers include a typedef of __leNN. + +Signed-off-by: Paolo Bonzini +(cherry picked from commit 5150004ccf5fe72c35b3263fbed6f4d06ed3cc6a) +Signed-off-by: Paolo Bonzini +--- + hw/net/rocker/rocker.h | 14 +++--------- + hw/net/rocker/rocker_hw.h | 20 +++++++----------- + hw/net/rocker/rocker_of_dpa.c | 40 +++++++++++++++++------------------ + 3 files changed, 31 insertions(+), 43 deletions(-) + +diff --git a/hw/net/rocker/rocker.h b/hw/net/rocker/rocker.h +index f85354d9d1..fa13ae2993 100644 +--- a/hw/net/rocker/rocker.h ++++ b/hw/net/rocker/rocker.h +@@ -36,15 +36,7 @@ static inline G_GNUC_PRINTF(1, 2) int DPRINTF(const char *fmt, ...) + } + #endif + +-#define __le16 uint16_t +-#define __le32 uint32_t +-#define __le64 uint64_t +- +-#define __be16 uint16_t +-#define __be32 uint32_t +-#define __be64 uint64_t +- +-static inline bool ipv4_addr_is_multicast(__be32 addr) ++static inline bool ipv4_addr_is_multicast(uint32_t addr) + { + return (addr & htonl(0xf0000000)) == htonl(0xe0000000); + } +@@ -52,8 +44,8 @@ static inline bool ipv4_addr_is_multicast(__be32 addr) + typedef struct ipv6_addr { + union { + uint8_t addr8[16]; +- __be16 addr16[8]; +- __be32 addr32[4]; ++ uint16_t addr16[8]; ++ uint32_t addr32[4]; + }; + } Ipv6Addr; + +diff --git a/hw/net/rocker/rocker_hw.h b/hw/net/rocker/rocker_hw.h +index 1786323fa4..7ec6bfbcb9 100644 +--- a/hw/net/rocker/rocker_hw.h ++++ b/hw/net/rocker/rocker_hw.h +@@ -9,10 +9,6 @@ + #ifndef ROCKER_HW_H + #define ROCKER_HW_H + +-#define __le16 uint16_t +-#define __le32 uint32_t +-#define __le64 uint64_t +- + /* + * Return codes + */ +@@ -124,12 +120,12 @@ enum { + */ + + typedef struct rocker_desc { +- __le64 buf_addr; ++ uint64_t buf_addr; + uint64_t cookie; +- __le16 buf_size; +- __le16 tlv_size; +- __le16 rsvd[5]; /* pad to 32 bytes */ +- __le16 comp_err; ++ uint16_t buf_size; ++ uint16_t tlv_size; ++ uint16_t rsvd[5]; /* pad to 32 bytes */ ++ uint16_t comp_err; + } __attribute__((packed, aligned(8))) RockerDesc; + + /* +@@ -137,9 +133,9 @@ typedef struct rocker_desc { + */ + + typedef struct rocker_tlv { +- __le32 type; +- __le16 len; +- __le16 rsvd; ++ uint32_t type; ++ uint16_t len; ++ uint16_t rsvd; + } __attribute__((packed, aligned(8))) RockerTlv; + + /* cmd msg */ +diff --git a/hw/net/rocker/rocker_of_dpa.c b/hw/net/rocker/rocker_of_dpa.c +index 5e16056be6..a298805c89 100644 +--- a/hw/net/rocker/rocker_of_dpa.c ++++ b/hw/net/rocker/rocker_of_dpa.c +@@ -52,10 +52,10 @@ typedef struct of_dpa_flow_key { + uint32_t tunnel_id; /* overlay tunnel id */ + uint32_t tbl_id; /* table id */ + struct { +- __be16 vlan_id; /* 0 if no VLAN */ ++ uint16_t vlan_id; /* 0 if no VLAN */ + MACAddr src; /* ethernet source address */ + MACAddr dst; /* ethernet destination address */ +- __be16 type; /* ethernet frame type */ ++ uint16_t type; /* ethernet frame type */ + } eth; + struct { + uint8_t proto; /* IP protocol or ARP opcode */ +@@ -66,14 +66,14 @@ typedef struct of_dpa_flow_key { + union { + struct { + struct { +- __be32 src; /* IP source address */ +- __be32 dst; /* IP destination address */ ++ uint32_t src; /* IP source address */ ++ uint32_t dst; /* IP destination address */ + } addr; + union { + struct { +- __be16 src; /* TCP/UDP/SCTP source port */ +- __be16 dst; /* TCP/UDP/SCTP destination port */ +- __be16 flags; /* TCP flags */ ++ uint16_t src; /* TCP/UDP/SCTP source port */ ++ uint16_t dst; /* TCP/UDP/SCTP destination port */ ++ uint16_t flags; /* TCP flags */ + } tp; + struct { + MACAddr sha; /* ARP source hardware address */ +@@ -86,11 +86,11 @@ typedef struct of_dpa_flow_key { + Ipv6Addr src; /* IPv6 source address */ + Ipv6Addr dst; /* IPv6 destination address */ + } addr; +- __be32 label; /* IPv6 flow label */ ++ uint32_t label; /* IPv6 flow label */ + struct { +- __be16 src; /* TCP/UDP/SCTP source port */ +- __be16 dst; /* TCP/UDP/SCTP destination port */ +- __be16 flags; /* TCP flags */ ++ uint16_t src; /* TCP/UDP/SCTP source port */ ++ uint16_t dst; /* TCP/UDP/SCTP destination port */ ++ uint16_t flags; /* TCP flags */ + } tp; + struct { + Ipv6Addr target; /* ND target address */ +@@ -112,13 +112,13 @@ typedef struct of_dpa_flow_action { + struct { + uint32_t group_id; + uint32_t tun_log_lport; +- __be16 vlan_id; ++ uint16_t vlan_id; + } write; + struct { +- __be16 new_vlan_id; ++ uint16_t new_vlan_id; + uint32_t out_pport; + uint8_t copy_to_cpu; +- __be16 vlan_id; ++ uint16_t vlan_id; + } apply; + } OfDpaFlowAction; + +@@ -143,7 +143,7 @@ typedef struct of_dpa_flow { + typedef struct of_dpa_flow_pkt_fields { + uint32_t tunnel_id; + struct eth_header *ethhdr; +- __be16 *h_proto; ++ uint16_t *h_proto; + struct vlan_header *vlanhdr; + struct ip_header *ipv4hdr; + struct ip6_header *ipv6hdr; +@@ -180,7 +180,7 @@ typedef struct of_dpa_group { + uint32_t group_id; + MACAddr src_mac; + MACAddr dst_mac; +- __be16 vlan_id; ++ uint16_t vlan_id; + } l2_rewrite; + struct { + uint16_t group_count; +@@ -190,13 +190,13 @@ typedef struct of_dpa_group { + uint32_t group_id; + MACAddr src_mac; + MACAddr dst_mac; +- __be16 vlan_id; ++ uint16_t vlan_id; + uint8_t ttl_check; + } l3_unicast; + }; + } OfDpaGroup; + +-static int of_dpa_mask2prefix(__be32 mask) ++static int of_dpa_mask2prefix(uint32_t mask) + { + int i; + int count = 32; +@@ -451,7 +451,7 @@ static void of_dpa_flow_pkt_parse(OfDpaFlowContext *fc, + fc->iovcnt = iovcnt + 2; + } + +-static void of_dpa_flow_pkt_insert_vlan(OfDpaFlowContext *fc, __be16 vlan_id) ++static void of_dpa_flow_pkt_insert_vlan(OfDpaFlowContext *fc, uint16_t vlan_id) + { + OfDpaFlowPktFields *fields = &fc->fields; + uint16_t h_proto = fields->ethhdr->h_proto; +@@ -486,7 +486,7 @@ static void of_dpa_flow_pkt_strip_vlan(OfDpaFlowContext *fc) + + static void of_dpa_flow_pkt_hdr_rewrite(OfDpaFlowContext *fc, + uint8_t *src_mac, uint8_t *dst_mac, +- __be16 vlan_id) ++ uint16_t vlan_id) + { + OfDpaFlowPktFields *fields = &fc->fields; + +-- +2.50.1 + diff --git a/SOURCES/kvm-s390x-Fix-leak-in-machine_set_loadparm.patch b/SOURCES/kvm-s390x-Fix-leak-in-machine_set_loadparm.patch new file mode 100644 index 0000000..8b2660d --- /dev/null +++ b/SOURCES/kvm-s390x-Fix-leak-in-machine_set_loadparm.patch @@ -0,0 +1,60 @@ +From 4f627e0ae8efb96380070b6a8d50e88c71f40477 Mon Sep 17 00:00:00 2001 +From: Fabiano Rosas +Date: Fri, 9 May 2025 14:49:38 -0300 +Subject: [PATCH 01/57] s390x: Fix leak in machine_set_loadparm +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Thomas Huth +RH-MergeRequest: 387: s390x: Fix memory leaks related to loadparm [rhel-9] +RH-Jira: RHEL-98554 +RH-Acked-by: Cédric Le Goater +RH-Acked-by: Kevin Wolf +RH-Commit: [1/2] dadf5b9e187a644e0a8a8c565b1b913ef7f4dcc8 (thuth/qemu-kvm-cs) + +ASAN spotted a leaking string in machine_set_loadparm(): + +Direct leak of 9 byte(s) in 1 object(s) allocated from: + #0 0x560ffb5bb379 in malloc ../projects/compiler-rt/lib/asan/asan_malloc_linux.cpp:69:3 + #1 0x7f1aca926518 in g_malloc ../glib/gmem.c:106 + #2 0x7f1aca94113e in g_strdup ../glib/gstrfuncs.c:364 + #3 0x560ffc8afbf9 in qobject_input_type_str ../qapi/qobject-input-visitor.c:542:12 + #4 0x560ffc8a80ff in visit_type_str ../qapi/qapi-visit-core.c:349:10 + #5 0x560ffbe6053a in machine_set_loadparm ../hw/s390x/s390-virtio-ccw.c:802:10 + #6 0x560ffc0c5e52 in object_property_set ../qom/object.c:1450:5 + #7 0x560ffc0d4175 in object_property_set_qobject ../qom/qom-qobject.c:28:10 + #8 0x560ffc0c6004 in object_property_set_str ../qom/object.c:1458:15 + #9 0x560ffbe2ae60 in update_machine_ipl_properties ../hw/s390x/ipl.c:569:9 + #10 0x560ffbe2aa65 in s390_ipl_update_diag308 ../hw/s390x/ipl.c:594:5 + #11 0x560ffbdee132 in handle_diag_308 ../target/s390x/diag.c:147:9 + #12 0x560ffbebb956 in helper_diag ../target/s390x/tcg/misc_helper.c:137:9 + #13 0x7f1a3c51c730 (/memfd:tcg-jit (deleted)+0x39730) + +Cc: qemu-stable@nongnu.org +Signed-off-by: Fabiano Rosas +Message-ID: <20250509174938.25935-1-farosas@suse.de> +Fixes: 1fd396e3228 ("s390x: Register TYPE_S390_CCW_MACHINE properties as class properties") +Reviewed-by: Thomas Huth +Reviewed-by: Philippe Mathieu-Daudé +Signed-off-by: Thomas Huth +(cherry picked from commit bdf12f2a56bf3f13c52eb51f0a994bbfe40706b2) +--- + hw/s390x/s390-virtio-ccw.c | 1 + + 1 file changed, 1 insertion(+) + +diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c +index 77a1bde71e..fc18ab575f 100644 +--- a/hw/s390x/s390-virtio-ccw.c ++++ b/hw/s390x/s390-virtio-ccw.c +@@ -782,6 +782,7 @@ static void machine_set_loadparm(Object *obj, Visitor *v, + } + + s390_ipl_fmt_loadparm(ms->loadparm, val, errp); ++ g_free(val); + } + + static void ccw_machine_class_init(ObjectClass *oc, void *data) +-- +2.39.3 + diff --git a/SOURCES/kvm-s390x-introduce-s390_get_memory_limit.patch b/SOURCES/kvm-s390x-introduce-s390_get_memory_limit.patch new file mode 100644 index 0000000..8647d8e --- /dev/null +++ b/SOURCES/kvm-s390x-introduce-s390_get_memory_limit.patch @@ -0,0 +1,144 @@ +From 1dd38383832fc27f2980f33bb5e10ec1af7e3fc3 Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Thu, 19 Dec 2024 15:41:07 +0100 +Subject: [PATCH 15/26] s390x: introduce s390_get_memory_limit() + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [15/26] 5ae6a624a6541283cb15e90ebeb8fef3940c823b (thuth/qemu-kvm-cs) + +Let's add s390_get_memory_limit(), to query what has been successfully +set via s390_set_memory_limit(). Allow setting the limit only once. + +We'll remember the limit in the machine state. Move +s390_set_memory_limit() to machine code, merging it into +set_memory_limit(), because this really is a machine property. + +Message-ID: <20241219144115.2820241-7-david@redhat.com> +Acked-by: Michael S. Tsirkin +Reviewed-by: Thomas Huth +Signed-off-by: David Hildenbrand +(cherry picked from commit 27221b69a3ea49339a1f82b9622126f3928e0915) +Signed-off-by: Thomas Huth +--- + hw/s390x/s390-virtio-ccw.c | 17 ++++++++++++----- + include/hw/s390x/s390-virtio-ccw.h | 8 ++++++++ + target/s390x/cpu-sysemu.c | 8 -------- + target/s390x/cpu.h | 1 - + 4 files changed, 20 insertions(+), 14 deletions(-) + +diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c +index 248ac28d20..f5f147eb92 100644 +--- a/hw/s390x/s390-virtio-ccw.c ++++ b/hw/s390x/s390-virtio-ccw.c +@@ -45,6 +45,7 @@ + #include "migration/blocker.h" + #include "qapi/visitor.h" + #include "hw/s390x/cpu-topology.h" ++#include "kvm/kvm_s390x.h" + #include CONFIG_DEVICES + + static Error *pv_mig_blocker; +@@ -121,12 +122,16 @@ static void subsystem_reset(void) + } + } + +-static void set_memory_limit(uint64_t new_limit) ++static void s390_set_memory_limit(S390CcwMachineState *s390ms, ++ uint64_t new_limit) + { +- uint64_t hw_limit; +- int ret; ++ uint64_t hw_limit = 0; ++ int ret = 0; + +- ret = s390_set_memory_limit(new_limit, &hw_limit); ++ assert(!s390ms->memory_limit && new_limit); ++ if (kvm_enabled()) { ++ ret = kvm_s390_set_mem_limit(new_limit, &hw_limit); ++ } + if (ret == -E2BIG) { + error_report("host supports a maximum of %" PRIu64 " GB", + hw_limit / GiB); +@@ -135,10 +140,12 @@ static void set_memory_limit(uint64_t new_limit) + error_report("setting the guest size failed"); + exit(EXIT_FAILURE); + } ++ s390ms->memory_limit = new_limit; + } + + static void s390_memory_init(MachineState *machine) + { ++ S390CcwMachineState *s390ms = S390_CCW_MACHINE(machine); + MemoryRegion *sysmem = get_system_memory(); + MemoryRegion *ram = machine->ram; + uint64_t ram_size = memory_region_size(ram); +@@ -154,7 +161,7 @@ static void s390_memory_init(MachineState *machine) + exit(EXIT_FAILURE); + } + +- set_memory_limit(ram_size); ++ s390_set_memory_limit(s390ms, ram_size); + + /* Map the initial memory. Must happen after setting the memory limit. */ + memory_region_add_subregion(sysmem, 0, ram); +diff --git a/include/hw/s390x/s390-virtio-ccw.h b/include/hw/s390x/s390-virtio-ccw.h +index 996864a34e..de04336c5a 100644 +--- a/include/hw/s390x/s390-virtio-ccw.h ++++ b/include/hw/s390x/s390-virtio-ccw.h +@@ -29,10 +29,18 @@ struct S390CcwMachineState { + bool dea_key_wrap; + bool pv; + uint8_t loadparm[8]; ++ uint64_t memory_limit; + + SCLPDevice *sclp; + }; + ++static inline uint64_t s390_get_memory_limit(S390CcwMachineState *s390ms) ++{ ++ /* We expect to be called only after the limit was set. */ ++ assert(s390ms->memory_limit); ++ return s390ms->memory_limit; ++} ++ + #define S390_PTF_REASON_NONE (0x00 << 8) + #define S390_PTF_REASON_DONE (0x01 << 8) + #define S390_PTF_REASON_BUSY (0x02 << 8) +diff --git a/target/s390x/cpu-sysemu.c b/target/s390x/cpu-sysemu.c +index 1cd30c1d84..3118a25fee 100644 +--- a/target/s390x/cpu-sysemu.c ++++ b/target/s390x/cpu-sysemu.c +@@ -255,14 +255,6 @@ unsigned int s390_cpu_set_state(uint8_t cpu_state, S390CPU *cpu) + return s390_count_running_cpus(); + } + +-int s390_set_memory_limit(uint64_t new_limit, uint64_t *hw_limit) +-{ +- if (kvm_enabled()) { +- return kvm_s390_set_mem_limit(new_limit, hw_limit); +- } +- return 0; +-} +- + void s390_set_max_pagesize(uint64_t pagesize, Error **errp) + { + if (kvm_enabled()) { +diff --git a/target/s390x/cpu.h b/target/s390x/cpu.h +index 6a64472403..ecaf3191d2 100644 +--- a/target/s390x/cpu.h ++++ b/target/s390x/cpu.h +@@ -881,7 +881,6 @@ static inline void s390_do_cpu_load_normal(CPUState *cs, run_on_cpu_data arg) + + /* cpu.c */ + void s390_crypto_reset(void); +-int s390_set_memory_limit(uint64_t new_limit, uint64_t *hw_limit); + void s390_set_max_pagesize(uint64_t pagesize, Error **errp); + void s390_cmma_reset(void); + void s390_enable_css_support(S390CPU *cpu); +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-pci-add-support-for-guests-that-request-direct.patch b/SOURCES/kvm-s390x-pci-add-support-for-guests-that-request-direct.patch new file mode 100644 index 0000000..da8c553 --- /dev/null +++ b/SOURCES/kvm-s390x-pci-add-support-for-guests-that-request-direct.patch @@ -0,0 +1,256 @@ +From c60d0770ff3f9124e6e9d7beb03e1ef8067e8e26 Mon Sep 17 00:00:00 2001 +From: Christoph Schlameuss +Date: Thu, 12 Jun 2025 13:25:32 +0200 +Subject: [PATCH 01/16] s390x/pci: add support for guests that request direct + mapping +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Christoph Schlameuss +RH-MergeRequest: 376: Draft: KVM: Performance Enhanced Refresh PCI Translation +RH-Jira: RHEL-11430 +RH-Acked-by: Thomas Huth +RH-Acked-by: Cédric Le Goater +RH-Commit: [1/2] 11d1dd9a5add55ae43d5d922588a33945ecbfe27 (cschlame/qemu-kvm) + +JIRA: https://issues.redhat.com/browse/RHEL-11430 +Conflicts: hw/s390x/s390-pci-bus.c old s390_pci_device_properties[] still has DEFINE_PROP_END_OF_LIST() + hw/s390x/s390-pci-inst.c hw_accel.h is still in sysemu + hw/s390x/s390-virtio-ccw.c changes from ccw_machine_9_2_class_options() moved to ccw_rhel_machine_9_6_0_class_options() + +commit dfcee1ea4c52ac60e0a06221eafb7b6253eb10c3 +Author: Matthew Rosato +Date: Wed Feb 26 16:00:12 2025 -0500 + + s390x/pci: add support for guests that request direct mapping + + When receiving a guest mpcifc(4) or mpcifc(6) instruction without the T + bit set, treat this as a request to perform direct mapping instead of + address translation. In order to facilitate this, pin the entirety of + guest memory into the host iommu. + + Pinning for the direct mapping case is handled via vfio and its memory + listener. Additionally, ram discard settings are inherited from vfio: + coordinated discards (e.g. virtio-mem) are allowed while uncoordinated + discards (e.g. virtio-balloon) are disabled. + + Subsequent guest DMA operations are all expected to be of the format + guest_phys+sdma, allowing them to be used as lookup into the host + iommu table. + + Signed-off-by: Matthew Rosato + Reviewed-by: David Hildenbrand + Message-ID: <20250226210013.238349-2-mjrosato@linux.ibm.com> + Signed-off-by: Thomas Huth + +Signed-off-by: Christoph Schlameuss +--- + hw/s390x/s390-pci-bus.c | 39 +++++++++++++++++++++++++++++++-- + hw/s390x/s390-pci-inst.c | 13 +++++++++-- + hw/s390x/s390-pci-vfio.c | 23 +++++++++++++++---- + hw/s390x/s390-virtio-ccw.c | 5 +++++ + include/hw/s390x/s390-pci-bus.h | 3 +++ + 5 files changed, 75 insertions(+), 8 deletions(-) + +diff --git a/hw/s390x/s390-pci-bus.c b/hw/s390x/s390-pci-bus.c +index 3e57d5faca..13bc02d837 100644 +--- a/hw/s390x/s390-pci-bus.c ++++ b/hw/s390x/s390-pci-bus.c +@@ -18,6 +18,8 @@ + #include "hw/s390x/s390-pci-inst.h" + #include "hw/s390x/s390-pci-kvm.h" + #include "hw/s390x/s390-pci-vfio.h" ++#include "hw/s390x/s390-virtio-ccw.h" ++#include "hw/boards.h" + #include "hw/pci/pci_bus.h" + #include "hw/qdev-properties.h" + #include "hw/pci/pci_bridge.h" +@@ -724,12 +726,42 @@ void s390_pci_iommu_enable(S390PCIIOMMU *iommu) + g_free(name); + } + ++void s390_pci_iommu_direct_map_enable(S390PCIIOMMU *iommu) ++{ ++ MachineState *ms = MACHINE(qdev_get_machine()); ++ S390CcwMachineState *s390ms = S390_CCW_MACHINE(ms); ++ ++ /* ++ * For direct-mapping we must map the entire guest address space. Rather ++ * than using an iommu, create a memory region alias that maps GPA X to ++ * IOVA X + SDMA. VFIO will handle pinning via its memory listener. ++ */ ++ g_autofree char *name = g_strdup_printf("iommu-dm-s390-%04x", ++ iommu->pbdev->uid); ++ ++ iommu->dm_mr = g_malloc0(sizeof(*iommu->dm_mr)); ++ memory_region_init_alias(iommu->dm_mr, OBJECT(&iommu->mr), name, ++ get_system_memory(), 0, ++ s390_get_memory_limit(s390ms)); ++ iommu->enabled = true; ++ memory_region_add_subregion(&iommu->mr, iommu->pbdev->zpci_fn.sdma, ++ iommu->dm_mr); ++} ++ + void s390_pci_iommu_disable(S390PCIIOMMU *iommu) + { + iommu->enabled = false; + g_hash_table_remove_all(iommu->iotlb); +- memory_region_del_subregion(&iommu->mr, MEMORY_REGION(&iommu->iommu_mr)); +- object_unparent(OBJECT(&iommu->iommu_mr)); ++ if (iommu->dm_mr) { ++ memory_region_del_subregion(&iommu->mr, iommu->dm_mr); ++ object_unparent(OBJECT(iommu->dm_mr)); ++ g_free(iommu->dm_mr); ++ iommu->dm_mr = NULL; ++ } else { ++ memory_region_del_subregion(&iommu->mr, ++ MEMORY_REGION(&iommu->iommu_mr)); ++ object_unparent(OBJECT(&iommu->iommu_mr)); ++ } + } + + static void s390_pci_iommu_free(S390pciState *s, PCIBus *bus, int32_t devfn) +@@ -1130,6 +1162,7 @@ static void s390_pcihost_plug(HotplugHandler *hotplug_dev, DeviceState *dev, + /* Always intercept emulated devices */ + pbdev->interp = false; + pbdev->forwarding_assist = false; ++ pbdev->rtr_avail = false; + } + + if (s390_pci_msix_init(pbdev) && !pbdev->interp) { +@@ -1488,6 +1521,8 @@ static Property s390_pci_device_properties[] = { + DEFINE_PROP_BOOL("interpret", S390PCIBusDevice, interp, true), + DEFINE_PROP_BOOL("forwarding-assist", S390PCIBusDevice, forwarding_assist, + true), ++ DEFINE_PROP_BOOL("relaxed-translation", S390PCIBusDevice, rtr_avail, ++ true), + DEFINE_PROP_END_OF_LIST(), + }; + +diff --git a/hw/s390x/s390-pci-inst.c b/hw/s390x/s390-pci-inst.c +index 30149546c0..803ebcd9b3 100644 +--- a/hw/s390x/s390-pci-inst.c ++++ b/hw/s390x/s390-pci-inst.c +@@ -16,6 +16,7 @@ + #include "exec/memory.h" + #include "qemu/error-report.h" + #include "sysemu/hw_accel.h" ++#include "hw/boards.h" + #include "hw/pci/pci_device.h" + #include "hw/s390x/s390-pci-inst.h" + #include "hw/s390x/s390-pci-bus.h" +@@ -1008,17 +1009,25 @@ static int reg_ioat(CPUS390XState *env, S390PCIBusDevice *pbdev, ZpciFib fib, + } + + /* currently we only support designation type 1 with translation */ +- if (!(dt == ZPCI_IOTA_RTTO && t)) { ++ if (t && dt != ZPCI_IOTA_RTTO) { + error_report("unsupported ioat dt %d t %d", dt, t); + s390_program_interrupt(env, PGM_OPERAND, ra); + return -EINVAL; ++ } else if (!t && !pbdev->rtr_avail) { ++ error_report("relaxed translation not allowed"); ++ s390_program_interrupt(env, PGM_OPERAND, ra); ++ return -EINVAL; + } + + iommu->pba = pba; + iommu->pal = pal; + iommu->g_iota = g_iota; + +- s390_pci_iommu_enable(iommu); ++ if (t) { ++ s390_pci_iommu_enable(iommu); ++ } else { ++ s390_pci_iommu_direct_map_enable(iommu); ++ } + + return 0; + } +diff --git a/hw/s390x/s390-pci-vfio.c b/hw/s390x/s390-pci-vfio.c +index 7dbbc76823..443e222912 100644 +--- a/hw/s390x/s390-pci-vfio.c ++++ b/hw/s390x/s390-pci-vfio.c +@@ -131,13 +131,28 @@ static void s390_pci_read_base(S390PCIBusDevice *pbdev, + /* Store function type separately for type-specific behavior */ + pbdev->pft = cap->pft; + ++ /* ++ * If the device is a passthrough ISM device, disallow relaxed ++ * translation. ++ */ ++ if (pbdev->pft == ZPCI_PFT_ISM) { ++ pbdev->rtr_avail = false; ++ } ++ + /* + * If appropriate, reduce the size of the supported DMA aperture reported +- * to the guest based upon the vfio DMA limit. ++ * to the guest based upon the vfio DMA limit. This is applicable for ++ * devices that are guaranteed to not use relaxed translation. If the ++ * device is capable of relaxed translation then we must advertise the ++ * full aperture. In this case, if translation is used then we will ++ * rely on the vfio DMA limit counting and use RPCIT CC1 / status 16 ++ * to request that the guest free DMA mappings as necessary. + */ +- vfio_size = pbdev->iommu->max_dma_limit << TARGET_PAGE_BITS; +- if (vfio_size > 0 && vfio_size < cap->end_dma - cap->start_dma + 1) { +- pbdev->zpci_fn.edma = cap->start_dma + vfio_size - 1; ++ if (!pbdev->rtr_avail) { ++ vfio_size = pbdev->iommu->max_dma_limit << TARGET_PAGE_BITS; ++ if (vfio_size > 0 && vfio_size < cap->end_dma - cap->start_dma + 1) { ++ pbdev->zpci_fn.edma = cap->start_dma + vfio_size - 1; ++ } + } + } + +diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c +index 312e8f18aa..77a1bde71e 100644 +--- a/hw/s390x/s390-virtio-ccw.c ++++ b/hw/s390x/s390-virtio-ccw.c +@@ -1348,8 +1348,13 @@ static void ccw_rhel_machine_9_6_0_instance_options(MachineState *machine) + + static void ccw_rhel_machine_9_6_0_class_options(MachineClass *mc) + { ++ static GlobalProperty compat[] = { ++ { TYPE_S390_PCI_DEVICE, "relaxed-translation", "off", }, ++ }; ++ + /* NB: remember to move this line to the *latest* RHEL 9 machine */ + compat_props_add(mc->compat_props, hw_compat_rhel_9, hw_compat_rhel_9_len); ++ compat_props_add(mc->compat_props, compat, G_N_ELEMENTS(compat)); + } + DEFINE_CCW_MACHINE_AS_LATEST(9, 6, 0); + +diff --git a/include/hw/s390x/s390-pci-bus.h b/include/hw/s390x/s390-pci-bus.h +index 2c43ea123f..04944d4fed 100644 +--- a/include/hw/s390x/s390-pci-bus.h ++++ b/include/hw/s390x/s390-pci-bus.h +@@ -277,6 +277,7 @@ struct S390PCIIOMMU { + AddressSpace as; + MemoryRegion mr; + IOMMUMemoryRegion iommu_mr; ++ MemoryRegion *dm_mr; + bool enabled; + uint64_t g_iota; + uint64_t pba; +@@ -362,6 +363,7 @@ struct S390PCIBusDevice { + bool interp; + bool forwarding_assist; + bool aif; ++ bool rtr_avail; + QTAILQ_ENTRY(S390PCIBusDevice) link; + }; + +@@ -389,6 +391,7 @@ int pci_chsc_sei_nt2_have_event(void); + void s390_pci_sclp_configure(SCCB *sccb); + void s390_pci_sclp_deconfigure(SCCB *sccb); + void s390_pci_iommu_enable(S390PCIIOMMU *iommu); ++void s390_pci_iommu_direct_map_enable(S390PCIIOMMU *iommu); + void s390_pci_iommu_disable(S390PCIIOMMU *iommu); + void s390_pci_generate_error_event(uint16_t pec, uint32_t fh, uint32_t fid, + uint64_t faddr, uint32_t e); +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-pci-indicate-QEMU-supports-relaxed-translation.patch b/SOURCES/kvm-s390x-pci-indicate-QEMU-supports-relaxed-translation.patch new file mode 100644 index 0000000..918fb63 --- /dev/null +++ b/SOURCES/kvm-s390x-pci-indicate-QEMU-supports-relaxed-translation.patch @@ -0,0 +1,72 @@ +From 13e8ddbd282da692c8199a6cb9ca847334089e29 Mon Sep 17 00:00:00 2001 +From: Christoph Schlameuss +Date: Thu, 12 Jun 2025 11:48:41 +0200 +Subject: [PATCH 02/16] s390x/pci: indicate QEMU supports relaxed translation + for passthrough +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Christoph Schlameuss +RH-MergeRequest: 376: Draft: KVM: Performance Enhanced Refresh PCI Translation +RH-Jira: RHEL-11430 +RH-Acked-by: Thomas Huth +RH-Acked-by: Cédric Le Goater +RH-Commit: [2/2] afd514268347d0b434a60d7c6c09d20b84e5d902 (cschlame/qemu-kvm) + +JIRA: https://issues.redhat.com/browse/RHEL-11430 + +commit d9b5dfc7122559e5b5959ecf534788b90c3dd102 +Author: Matthew Rosato +Date: Wed Feb 26 16:00:13 2025 -0500 + + s390x/pci: indicate QEMU supports relaxed translation for passthrough + + Specifying this bit in the guest CLP response indicates that the guest + can optionally choose to skip translation and instead use + identity-mapped operations. + + Tested-by: Niklas Schnelle + Reviewed-by: Niklas Schnelle + Signed-off-by: Matthew Rosato + Message-ID: <20250226210013.238349-3-mjrosato@linux.ibm.com> + Signed-off-by: Thomas Huth + +Signed-off-by: Christoph Schlameuss +--- + hw/s390x/s390-pci-vfio.c | 5 ++++- + include/hw/s390x/s390-pci-clp.h | 1 + + 2 files changed, 5 insertions(+), 1 deletion(-) + +diff --git a/hw/s390x/s390-pci-vfio.c b/hw/s390x/s390-pci-vfio.c +index 443e222912..6236ac7f1e 100644 +--- a/hw/s390x/s390-pci-vfio.c ++++ b/hw/s390x/s390-pci-vfio.c +@@ -238,8 +238,11 @@ static void s390_pci_read_group(S390PCIBusDevice *pbdev, + pbdev->pci_group = s390_group_create(pbdev->zpci_fn.pfgid, start_gid); + + resgrp = &pbdev->pci_group->zpci_group; ++ if (pbdev->rtr_avail) { ++ resgrp->fr |= CLP_RSP_QPCIG_MASK_RTR; ++ } + if (cap->flags & VFIO_DEVICE_INFO_ZPCI_FLAG_REFRESH) { +- resgrp->fr = 1; ++ resgrp->fr |= CLP_RSP_QPCIG_MASK_REFRESH; + } + resgrp->dasm = cap->dasm; + resgrp->msia = cap->msi_addr; +diff --git a/include/hw/s390x/s390-pci-clp.h b/include/hw/s390x/s390-pci-clp.h +index 03b7f9ba5f..6a635d693b 100644 +--- a/include/hw/s390x/s390-pci-clp.h ++++ b/include/hw/s390x/s390-pci-clp.h +@@ -158,6 +158,7 @@ typedef struct ClpRspQueryPciGrp { + #define CLP_RSP_QPCIG_MASK_NOI 0xfff + uint16_t i; + uint8_t version; ++#define CLP_RSP_QPCIG_MASK_RTR 0x20 + #define CLP_RSP_QPCIG_MASK_FRAME 0x2 + #define CLP_RSP_QPCIG_MASK_REFRESH 0x1 + uint8_t fr; +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-pv-prepare-for-memory-devices.patch b/SOURCES/kvm-s390x-pv-prepare-for-memory-devices.patch new file mode 100644 index 0000000..16ff2c8 --- /dev/null +++ b/SOURCES/kvm-s390x-pv-prepare-for-memory-devices.patch @@ -0,0 +1,46 @@ +From 9d5420c4370b74d60f082f2aa1225b19150ee629 Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Thu, 19 Dec 2024 15:41:12 +0100 +Subject: [PATCH 20/26] s390x/pv: prepare for memory devices + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [20/26] cdbe71168b9afa9657b94f1e7500568314c707a8 (thuth/qemu-kvm-cs) + +Let's avoid checking for the maxram_size, and instead rely on the memory +limit determined in s390_memory_init(), that might be larger than +maxram_size, for example due to alignment purposes. + +This check now correctly mimics what the kernel will check in +kvm_s390_pv_set_aside(), whereby a VM <= 2 GiB VM would end up using +a segment type ASCE. + +Message-ID: <20241219144115.2820241-12-david@redhat.com> +Acked-by: Michael S. Tsirkin +Reviewed-by: Nina Schoetterl-Glausch +Signed-off-by: David Hildenbrand +(cherry picked from commit a056332e732110c8ef0d40ffd49bd03afc2f04ca) +Signed-off-by: Thomas Huth +--- + target/s390x/kvm/pv.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/target/s390x/kvm/pv.c b/target/s390x/kvm/pv.c +index 424cce75ca..fa66607e7b 100644 +--- a/target/s390x/kvm/pv.c ++++ b/target/s390x/kvm/pv.c +@@ -133,7 +133,7 @@ bool s390_pv_vm_try_disable_async(S390CcwMachineState *ms) + * If the feature is not present or if the VM is not larger than 2 GiB, + * KVM_PV_ASYNC_CLEANUP_PREPARE fill fail; no point in attempting it. + */ +- if ((MACHINE(ms)->ram_size <= 2 * GiB) || ++ if (s390_get_memory_limit(ms) <= 2 * GiB || + !kvm_check_extension(kvm_state, KVM_CAP_S390_PROTECTED_ASYNC_DISABLE)) { + return false; + } +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-remember-the-maximum-page-size.patch b/SOURCES/kvm-s390x-remember-the-maximum-page-size.patch new file mode 100644 index 0000000..b2fd41e --- /dev/null +++ b/SOURCES/kvm-s390x-remember-the-maximum-page-size.patch @@ -0,0 +1,107 @@ +From 5a311d410bca4a5530a51c0b789ce8525d2d0653 Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Thu, 19 Dec 2024 15:41:13 +0100 +Subject: [PATCH 21/26] s390x: remember the maximum page size + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [21/26] 3b97c555b153d42e4fcb27dbb65fbf3edac622a4 (thuth/qemu-kvm-cs) + +Let's remember the value (successfully) set via s390_set_max_pagesize(). +This will be helpful to reject hotplugged memory devices that would exceed +this initially set page size. + +Handle it just like how we handle s390_get_memory_limit(), storing it in +the machine, and moving the handling to machine code. + +Message-ID: <20241219144115.2820241-13-david@redhat.com> +Acked-by: Michael S. Tsirkin +Reviewed-by: Thomas Huth +Signed-off-by: David Hildenbrand +(cherry picked from commit df2ac211a62e6ced7f1495b634fa6f78962f2321) +Signed-off-by: Thomas Huth +--- + hw/s390x/s390-virtio-ccw.c | 12 +++++++++++- + include/hw/s390x/s390-virtio-ccw.h | 1 + + target/s390x/cpu-sysemu.c | 7 ------- + target/s390x/cpu.h | 1 - + 4 files changed, 12 insertions(+), 9 deletions(-) + +diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c +index 824c73536a..bd05a22b4e 100644 +--- a/hw/s390x/s390-virtio-ccw.c ++++ b/hw/s390x/s390-virtio-ccw.c +@@ -143,6 +143,16 @@ static void s390_set_memory_limit(S390CcwMachineState *s390ms, + s390ms->memory_limit = new_limit; + } + ++static void s390_set_max_pagesize(S390CcwMachineState *s390ms, ++ uint64_t pagesize) ++{ ++ assert(!s390ms->max_pagesize && pagesize); ++ if (kvm_enabled()) { ++ kvm_s390_set_max_pagesize(pagesize, &error_fatal); ++ } ++ s390ms->max_pagesize = pagesize; ++} ++ + static void s390_memory_init(MachineState *machine) + { + S390CcwMachineState *s390ms = S390_CCW_MACHINE(machine); +@@ -191,7 +201,7 @@ static void s390_memory_init(MachineState *machine) + * Configure the maximum page size. As no memory devices were created + * yet, this is the page size of initial memory only. + */ +- s390_set_max_pagesize(qemu_maxrampagesize(), &error_fatal); ++ s390_set_max_pagesize(s390ms, qemu_maxrampagesize()); + /* Initialize storage key device */ + s390_skeys_init(); + /* Initialize storage attributes device */ +diff --git a/include/hw/s390x/s390-virtio-ccw.h b/include/hw/s390x/s390-virtio-ccw.h +index de04336c5a..599740a998 100644 +--- a/include/hw/s390x/s390-virtio-ccw.h ++++ b/include/hw/s390x/s390-virtio-ccw.h +@@ -30,6 +30,7 @@ struct S390CcwMachineState { + bool pv; + uint8_t loadparm[8]; + uint64_t memory_limit; ++ uint64_t max_pagesize; + + SCLPDevice *sclp; + }; +diff --git a/target/s390x/cpu-sysemu.c b/target/s390x/cpu-sysemu.c +index 3118a25fee..706a5c53e2 100644 +--- a/target/s390x/cpu-sysemu.c ++++ b/target/s390x/cpu-sysemu.c +@@ -255,13 +255,6 @@ unsigned int s390_cpu_set_state(uint8_t cpu_state, S390CPU *cpu) + return s390_count_running_cpus(); + } + +-void s390_set_max_pagesize(uint64_t pagesize, Error **errp) +-{ +- if (kvm_enabled()) { +- kvm_s390_set_max_pagesize(pagesize, errp); +- } +-} +- + void s390_cmma_reset(void) + { + if (kvm_enabled()) { +diff --git a/target/s390x/cpu.h b/target/s390x/cpu.h +index ecaf3191d2..9770a62ac9 100644 +--- a/target/s390x/cpu.h ++++ b/target/s390x/cpu.h +@@ -881,7 +881,6 @@ static inline void s390_do_cpu_load_normal(CPUState *cs, run_on_cpu_data arg) + + /* cpu.c */ + void s390_crypto_reset(void); +-void s390_set_max_pagesize(uint64_t pagesize, Error **errp); + void s390_cmma_reset(void); + void s390_enable_css_support(S390CPU *cpu); + void s390_do_cpu_set_diag318(CPUState *cs, run_on_cpu_data arg); +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-rename-s390-virtio-hcall-to-s390-hypercall.patch b/SOURCES/kvm-s390x-rename-s390-virtio-hcall-to-s390-hypercall.patch new file mode 100644 index 0000000..90bb4c3 --- /dev/null +++ b/SOURCES/kvm-s390x-rename-s390-virtio-hcall-to-s390-hypercall.patch @@ -0,0 +1,113 @@ +From 2fbdf7e3cf23daea470aaa4a29e16641feb76f3c Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Thu, 19 Dec 2024 15:41:05 +0100 +Subject: [PATCH 13/26] s390x: rename s390-virtio-hcall* to s390-hypercall* + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [13/26] 3c1ef3cbb137517b306871f0a88a61a59740af5a (thuth/qemu-kvm-cs) + +Let's make it clearer that we are talking about general +QEMU/KVM-specific hypercalls. + +Message-ID: <20241219144115.2820241-5-david@redhat.com> +Acked-by: Michael S. Tsirkin +Reviewed-by: Thomas Huth +Signed-off-by: David Hildenbrand +(cherry picked from commit 85489fc3652d0c4433c940f1a80a952e8cb5d3cb) +Signed-off-by: Thomas Huth +--- + hw/s390x/meson.build | 2 +- + hw/s390x/{s390-virtio-hcall.c => s390-hypercall.c} | 2 +- + hw/s390x/{s390-virtio-hcall.h => s390-hypercall.h} | 6 +++--- + target/s390x/kvm/kvm.c | 2 +- + target/s390x/tcg/misc_helper.c | 2 +- + 5 files changed, 7 insertions(+), 7 deletions(-) + rename hw/s390x/{s390-virtio-hcall.c => s390-hypercall.c} (97%) + rename hw/s390x/{s390-virtio-hcall.h => s390-hypercall.h} (86%) + +diff --git a/hw/s390x/meson.build b/hw/s390x/meson.build +index d6c8c33915..e344a3bd8c 100644 +--- a/hw/s390x/meson.build ++++ b/hw/s390x/meson.build +@@ -29,7 +29,7 @@ s390x_ss.add(when: 'CONFIG_TCG', if_true: files( + )) + s390x_ss.add(when: 'CONFIG_S390_CCW_VIRTIO', if_true: files( + 's390-virtio-ccw.c', +- 's390-virtio-hcall.c', ++ 's390-hypercall.c', + )) + s390x_ss.add(when: 'CONFIG_TERMINAL3270', if_true: files('3270-ccw.c')) + s390x_ss.add(when: 'CONFIG_VFIO', if_true: files('s390-pci-vfio.c')) +diff --git a/hw/s390x/s390-virtio-hcall.c b/hw/s390x/s390-hypercall.c +similarity index 97% +rename from hw/s390x/s390-virtio-hcall.c +rename to hw/s390x/s390-hypercall.c +index 5fb78a719e..f816c2b1ef 100644 +--- a/hw/s390x/s390-virtio-hcall.c ++++ b/hw/s390x/s390-hypercall.c +@@ -12,7 +12,7 @@ + #include "qemu/osdep.h" + #include "cpu.h" + #include "hw/boards.h" +-#include "hw/s390x/s390-virtio-hcall.h" ++#include "hw/s390x/s390-hypercall.h" + #include "hw/s390x/ioinst.h" + #include "hw/s390x/css.h" + #include "virtio-ccw.h" +diff --git a/hw/s390x/s390-virtio-hcall.h b/hw/s390x/s390-hypercall.h +similarity index 86% +rename from hw/s390x/s390-virtio-hcall.h +rename to hw/s390x/s390-hypercall.h +index dca456b926..2fa81dbfdd 100644 +--- a/hw/s390x/s390-virtio-hcall.h ++++ b/hw/s390x/s390-hypercall.h +@@ -9,8 +9,8 @@ + * directory. + */ + +-#ifndef HW_S390_VIRTIO_HCALL_H +-#define HW_S390_VIRTIO_HCALL_H ++#ifndef HW_S390_HYPERCALL_H ++#define HW_S390_HYPERCALL_H + + #include "cpu.h" + +@@ -21,4 +21,4 @@ + + void handle_diag_500(S390CPU *cpu, uintptr_t ra); + +-#endif /* HW_S390_VIRTIO_HCALL_H */ ++#endif /* HW_S390_HYPERCALL_H */ +diff --git a/target/s390x/kvm/kvm.c b/target/s390x/kvm/kvm.c +index 42d6a54126..afc8d570c9 100644 +--- a/target/s390x/kvm/kvm.c ++++ b/target/s390x/kvm/kvm.c +@@ -49,7 +49,7 @@ + #include "hw/s390x/ebcdic.h" + #include "exec/memattrs.h" + #include "hw/s390x/s390-virtio-ccw.h" +-#include "hw/s390x/s390-virtio-hcall.h" ++#include "hw/s390x/s390-hypercall.h" + #include "target/s390x/kvm/pv.h" + #include CONFIG_DEVICES + +diff --git a/target/s390x/tcg/misc_helper.c b/target/s390x/tcg/misc_helper.c +index 2b4310003b..b726a95352 100644 +--- a/target/s390x/tcg/misc_helper.c ++++ b/target/s390x/tcg/misc_helper.c +@@ -36,7 +36,7 @@ + #include "sysemu/cpus.h" + #include "sysemu/sysemu.h" + #include "hw/s390x/ebcdic.h" +-#include "hw/s390x/s390-virtio-hcall.h" ++#include "hw/s390x/s390-hypercall.h" + #include "hw/s390x/sclp.h" + #include "hw/s390x/s390_flic.h" + #include "hw/s390x/ioinst.h" +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-s390-hypercall-introduce-DIAG500-STORAGE_LIMIT.patch b/SOURCES/kvm-s390x-s390-hypercall-introduce-DIAG500-STORAGE_LIMIT.patch new file mode 100644 index 0000000..af613d5 --- /dev/null +++ b/SOURCES/kvm-s390x-s390-hypercall-introduce-DIAG500-STORAGE_LIMIT.patch @@ -0,0 +1,99 @@ +From 86417a068f24964422d4fd5ea301d70a0f8142d2 Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Thu, 19 Dec 2024 15:41:08 +0100 +Subject: [PATCH 16/26] s390x/s390-hypercall: introduce DIAG500 STORAGE_LIMIT + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [16/26] c1c341227388735450ddbba0201e7523e0658c07 (thuth/qemu-kvm-cs) + +A guest OS that supports memory hotplug / memory devices must during +boot be aware of the maximum possible physical memory address that it might +have to handle at a later stage during its runtime. + +For example, the maximum possible memory address might be required to +prepare the kernel virtual address space accordingly (e.g., select page +table hierarchy depth). + +On s390x there is currently no such mechanism that is compatible with +paravirtualized memory devices, because the whole SCLP interface was +designed around the idea of "storage increments" and "standby memory". +Paravirtualized memory devices we want to support, such as virtio-mem, have +no intersection with any of that, but could co-exist with them in the +future if ever needed. + +In particular, a guest OS must never detect and use device memory +without the help of a proper device driver. Device memory must not be +exposed in any firmware-provided memory map (SCLP or diag260 on s390x). +For this reason, these memory devices will be places in memory *above* +the "maximum storage increment" exposed via SCLP. + +Let's provide a new diag500 subcode to query the memory limit determined in +s390_memory_init(). + +Message-ID: <20241219144115.2820241-8-david@redhat.com> +Acked-by: Michael S. Tsirkin +Reviewed-by: Thomas Huth +Signed-off-by: David Hildenbrand +(cherry picked from commit f7c168657816486527727d860b73747d41f0c5f6) +Signed-off-by: Thomas Huth +--- + hw/s390x/s390-hypercall.c | 12 +++++++++++- + hw/s390x/s390-hypercall.h | 1 + + 2 files changed, 12 insertions(+), 1 deletion(-) + +diff --git a/hw/s390x/s390-hypercall.c b/hw/s390x/s390-hypercall.c +index f816c2b1ef..ac1b08b2cd 100644 +--- a/hw/s390x/s390-hypercall.c ++++ b/hw/s390x/s390-hypercall.c +@@ -11,7 +11,7 @@ + + #include "qemu/osdep.h" + #include "cpu.h" +-#include "hw/boards.h" ++#include "hw/s390x/s390-virtio-ccw.h" + #include "hw/s390x/s390-hypercall.h" + #include "hw/s390x/ioinst.h" + #include "hw/s390x/css.h" +@@ -57,6 +57,13 @@ static int handle_virtio_ccw_notify(uint64_t subch_id, uint64_t data) + return 0; + } + ++static uint64_t handle_storage_limit(void) ++{ ++ S390CcwMachineState *s390ms = S390_CCW_MACHINE(qdev_get_machine()); ++ ++ return s390_get_memory_limit(s390ms) - 1; ++} ++ + void handle_diag_500(S390CPU *cpu, uintptr_t ra) + { + CPUS390XState *env = &cpu->env; +@@ -69,6 +76,9 @@ void handle_diag_500(S390CPU *cpu, uintptr_t ra) + case DIAG500_VIRTIO_CCW_NOTIFY: + env->regs[2] = handle_virtio_ccw_notify(env->regs[2], env->regs[3]); + break; ++ case DIAG500_STORAGE_LIMIT: ++ env->regs[2] = handle_storage_limit(); ++ break; + default: + s390_program_interrupt(env, PGM_SPECIFICATION, ra); + } +diff --git a/hw/s390x/s390-hypercall.h b/hw/s390x/s390-hypercall.h +index 2fa81dbfdd..4f07209128 100644 +--- a/hw/s390x/s390-hypercall.h ++++ b/hw/s390x/s390-hypercall.h +@@ -18,6 +18,7 @@ + #define DIAG500_VIRTIO_RESET 1 /* legacy */ + #define DIAG500_VIRTIO_SET_STATUS 2 /* legacy */ + #define DIAG500_VIRTIO_CCW_NOTIFY 3 /* KVM_S390_VIRTIO_CCW_NOTIFY */ ++#define DIAG500_STORAGE_LIMIT 4 + + void handle_diag_500(S390CPU *cpu, uintptr_t ra); + +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-s390-skeys-prepare-for-memory-devices.patch b/SOURCES/kvm-s390x-s390-skeys-prepare-for-memory-devices.patch new file mode 100644 index 0000000..885ebf1 --- /dev/null +++ b/SOURCES/kvm-s390x-s390-skeys-prepare-for-memory-devices.patch @@ -0,0 +1,56 @@ +From 53d1b43699c6b30583f41a18a33c28893718aeac Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Thu, 19 Dec 2024 15:41:10 +0100 +Subject: [PATCH 18/26] s390x/s390-skeys: prepare for memory devices + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [18/26] 47edda0eeb6d5932f81633f2d9d294b1ca5f413c (thuth/qemu-kvm-cs) + +With memory devices, we will have storage keys for memory that +exceeds the initial ram size. + +The TODO already states that current handling is subopimal, +but we won't worry about improving that (TCG-only) thing for now. + +Message-ID: <20241219144115.2820241-10-david@redhat.com> +Acked-by: Michael S. Tsirkin +Reviewed-by: Thomas Huth +Signed-off-by: David Hildenbrand +(cherry picked from commit d1e3c2ac41b3f73708682e4e8212c32ad35013b9) +Signed-off-by: Thomas Huth +--- + hw/s390x/s390-skeys.c | 6 +++--- + 1 file changed, 3 insertions(+), 3 deletions(-) + +diff --git a/hw/s390x/s390-skeys.c b/hw/s390x/s390-skeys.c +index bf22d6863e..e4297b3b8a 100644 +--- a/hw/s390x/s390-skeys.c ++++ b/hw/s390x/s390-skeys.c +@@ -11,7 +11,7 @@ + + #include "qemu/osdep.h" + #include "qemu/units.h" +-#include "hw/boards.h" ++#include "hw/s390x/s390-virtio-ccw.h" + #include "hw/qdev-properties.h" + #include "hw/s390x/storage-keys.h" + #include "qapi/error.h" +@@ -251,9 +251,9 @@ static bool qemu_s390_enable_skeys(S390SKeysState *ss) + * g_once_init_enter() is good enough. + */ + if (g_once_init_enter(&initialized)) { +- MachineState *machine = MACHINE(qdev_get_machine()); ++ S390CcwMachineState *s390ms = S390_CCW_MACHINE(qdev_get_machine()); + +- skeys->key_count = machine->ram_size / TARGET_PAGE_SIZE; ++ skeys->key_count = s390_get_memory_limit(s390ms) / TARGET_PAGE_SIZE; + skeys->keydata = g_malloc0(skeys->key_count); + g_once_init_leave(&initialized, 1); + } +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-s390-stattrib-kvm-prepare-for-memory-devices-a.patch b/SOURCES/kvm-s390x-s390-stattrib-kvm-prepare-for-memory-devices-a.patch new file mode 100644 index 0000000..e8102a4 --- /dev/null +++ b/SOURCES/kvm-s390x-s390-stattrib-kvm-prepare-for-memory-devices-a.patch @@ -0,0 +1,155 @@ +From 1195c91d10892a888870248fd881612955b9e1eb Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Thu, 19 Dec 2024 15:41:09 +0100 +Subject: [PATCH 17/26] s390x/s390-stattrib-kvm: prepare for memory devices and + sparse memory layouts + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [17/26] 799aa7b2b9cc2a948e9f391bc0ecf739254c78b1 (thuth/qemu-kvm-cs) + +With memory devices, we will have storage attributes for memory that +exceeds the initial ram size. Further, we can easily have memory holes, +for which there (currently) are no storage attributes. + +In particular, with memory holes, KVM_S390_SET_CMMA_BITS will fail to set +some storage attributes. + +So let's do it like we handle storage keys migration, relying on +guest_phys_blocks_append(). However, in contrast to storage key +migration, we will handle it on the migration destination. + +This is a preparation for virtio-mem support. Note that ever since the +"early migration" feature was added (x-early-migration), the state +of device blocks (plugged/unplugged) is migrated early such that +guest_phys_blocks_append() will properly consider all currently plugged +memory blocks and skip any unplugged ones. + +In the future, we should try getting rid of the large temporary buffer +and also not send any attributes for any memory holes, just so they +get ignored on the destination. + +Message-ID: <20241219144115.2820241-9-david@redhat.com> +Acked-by: Michael S. Tsirkin +Reviewed-by: Thomas Huth +Signed-off-by: David Hildenbrand +(cherry picked from commit 241e6b2d27b090b17cda5b011b2064544b0c458b) +Signed-off-by: Thomas Huth +--- + hw/s390x/s390-stattrib-kvm.c | 67 +++++++++++++++++++++++------------- + 1 file changed, 43 insertions(+), 24 deletions(-) + +diff --git a/hw/s390x/s390-stattrib-kvm.c b/hw/s390x/s390-stattrib-kvm.c +index eeaa811098..33ec91422a 100644 +--- a/hw/s390x/s390-stattrib-kvm.c ++++ b/hw/s390x/s390-stattrib-kvm.c +@@ -10,11 +10,12 @@ + */ + + #include "qemu/osdep.h" +-#include "hw/boards.h" ++#include "hw/s390x/s390-virtio-ccw.h" + #include "migration/qemu-file.h" + #include "hw/s390x/storage-attributes.h" + #include "qemu/error-report.h" + #include "sysemu/kvm.h" ++#include "sysemu/memory_mapping.h" + #include "exec/ram_addr.h" + #include "kvm/kvm_s390x.h" + #include "qapi/error.h" +@@ -84,8 +85,8 @@ static int kvm_s390_stattrib_set_stattr(S390StAttribState *sa, + uint8_t *values) + { + KVMS390StAttribState *sas = KVM_S390_STATTRIB(sa); +- MachineState *machine = MACHINE(qdev_get_machine()); +- unsigned long max = machine->ram_size / TARGET_PAGE_SIZE; ++ S390CcwMachineState *s390ms = S390_CCW_MACHINE(qdev_get_machine()); ++ unsigned long max = s390_get_memory_limit(s390ms) / TARGET_PAGE_SIZE; + + if (start_gfn + count > max) { + error_report("Out of memory bounds when setting storage attributes"); +@@ -103,39 +104,57 @@ static int kvm_s390_stattrib_set_stattr(S390StAttribState *sa, + static void kvm_s390_stattrib_synchronize(S390StAttribState *sa) + { + KVMS390StAttribState *sas = KVM_S390_STATTRIB(sa); +- MachineState *machine = MACHINE(qdev_get_machine()); +- unsigned long max = machine->ram_size / TARGET_PAGE_SIZE; +- /* We do not need to reach the maximum buffer size allowed */ +- unsigned long cx, len = KVM_S390_SKEYS_MAX / 2; ++ S390CcwMachineState *s390ms = S390_CCW_MACHINE(qdev_get_machine()); ++ unsigned long max = s390_get_memory_limit(s390ms) / TARGET_PAGE_SIZE; ++ unsigned long start_gfn, end_gfn, pages; ++ GuestPhysBlockList guest_phys_blocks; ++ GuestPhysBlock *block; + int r; + struct kvm_s390_cmma_log clog = { + .flags = 0, + .mask = ~0ULL, + }; + +- if (sas->incoming_buffer) { +- for (cx = 0; cx + len <= max; cx += len) { +- clog.start_gfn = cx; +- clog.count = len; +- clog.values = (uint64_t)(sas->incoming_buffer + cx); +- r = kvm_vm_ioctl(kvm_state, KVM_S390_SET_CMMA_BITS, &clog); +- if (r) { +- error_report("KVM_S390_SET_CMMA_BITS failed: %s", strerror(-r)); +- return; +- } +- } +- if (cx < max) { +- clog.start_gfn = cx; +- clog.count = max - cx; +- clog.values = (uint64_t)(sas->incoming_buffer + cx); ++ if (!sas->incoming_buffer) { ++ return; ++ } ++ guest_phys_blocks_init(&guest_phys_blocks); ++ guest_phys_blocks_append(&guest_phys_blocks); ++ ++ QTAILQ_FOREACH(block, &guest_phys_blocks.head, next) { ++ assert(QEMU_IS_ALIGNED(block->target_start, TARGET_PAGE_SIZE)); ++ assert(QEMU_IS_ALIGNED(block->target_end, TARGET_PAGE_SIZE)); ++ ++ start_gfn = block->target_start / TARGET_PAGE_SIZE; ++ end_gfn = block->target_end / TARGET_PAGE_SIZE; ++ ++ while (start_gfn < end_gfn) { ++ /* Don't exceed the maximum buffer size. */ ++ pages = MIN(end_gfn - start_gfn, KVM_S390_SKEYS_MAX / 2); ++ ++ /* ++ * If we ever get guest physical memory beyond the configured ++ * memory limit, something went very wrong. ++ */ ++ assert(start_gfn + pages <= max); ++ ++ clog.start_gfn = start_gfn; ++ clog.count = pages; ++ clog.values = (uint64_t)(sas->incoming_buffer + start_gfn); + r = kvm_vm_ioctl(kvm_state, KVM_S390_SET_CMMA_BITS, &clog); + if (r) { + error_report("KVM_S390_SET_CMMA_BITS failed: %s", strerror(-r)); ++ goto out; + } ++ ++ start_gfn += pages; + } +- g_free(sas->incoming_buffer); +- sas->incoming_buffer = NULL; + } ++ ++out: ++ guest_phys_blocks_free(&guest_phys_blocks); ++ g_free(sas->incoming_buffer); ++ sas->incoming_buffer = NULL; + } + + static int kvm_s390_stattrib_set_migrationmode(S390StAttribState *sa, bool val, +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-s390-virtio-ccw-don-t-crash-on-weird-RAM-sizes.patch b/SOURCES/kvm-s390x-s390-virtio-ccw-don-t-crash-on-weird-RAM-sizes.patch new file mode 100644 index 0000000..fa99566 --- /dev/null +++ b/SOURCES/kvm-s390x-s390-virtio-ccw-don-t-crash-on-weird-RAM-sizes.patch @@ -0,0 +1,63 @@ +From 4ee3076ac566622929f9410636483c4f0b2da967 Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Thu, 19 Dec 2024 15:41:02 +0100 +Subject: [PATCH 10/26] s390x/s390-virtio-ccw: don't crash on weird RAM sizes + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [10/26] 55738da52f3cf4746bee2b17780a10720fa05863 (thuth/qemu-kvm-cs) + +KVM is not happy when starting a VM with weird RAM sizes: + + # qemu-system-s390x --enable-kvm --nographic -m 1234K + qemu-system-s390x: kvm_set_user_memory_region: KVM_SET_USER_MEMORY_REGION + failed, slot=0, start=0x0, size=0x244000: Invalid argument + kvm_set_phys_mem: error registering slot: Invalid argument + Aborted (core dumped) + +Let's handle that in a better way by rejecting such weird RAM sizes +right from the start: + + # qemu-system-s390x --enable-kvm --nographic -m 1234K + qemu-system-s390x: ram size must be multiples of 1 MiB + +Message-ID: <20241219144115.2820241-2-david@redhat.com> +Acked-by: Michael S. Tsirkin +Reviewed-by: Eric Farman +Reviewed-by: Thomas Huth +Acked-by: Janosch Frank +Signed-off-by: David Hildenbrand +(cherry picked from commit 14e568ab4836347481af2e334009c385f456a734) +Signed-off-by: Thomas Huth +--- + hw/s390x/s390-virtio-ccw.c | 11 +++++++++++ + 1 file changed, 11 insertions(+) + +diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c +index 94cad1705b..82ded9666c 100644 +--- a/hw/s390x/s390-virtio-ccw.c ++++ b/hw/s390x/s390-virtio-ccw.c +@@ -180,6 +180,17 @@ static void s390_memory_init(MemoryRegion *ram) + { + MemoryRegion *sysmem = get_system_memory(); + ++ if (!QEMU_IS_ALIGNED(memory_region_size(ram), 1 * MiB)) { ++ /* ++ * SCLP cannot possibly expose smaller granularity right now and KVM ++ * cannot handle smaller granularity. As we don't support NUMA, the ++ * region size directly corresponds to machine->ram_size, and the region ++ * is a single RAM memory region. ++ */ ++ error_report("ram size must be multiples of 1 MiB"); ++ exit(EXIT_FAILURE); ++ } ++ + /* allocate RAM for core */ + memory_region_add_subregion(sysmem, 0, ram); + +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-s390-virtio-ccw-move-setting-the-maximum-guest.patch b/SOURCES/kvm-s390x-s390-virtio-ccw-move-setting-the-maximum-guest.patch new file mode 100644 index 0000000..5a498d6 --- /dev/null +++ b/SOURCES/kvm-s390x-s390-virtio-ccw-move-setting-the-maximum-guest.patch @@ -0,0 +1,140 @@ +From 9ec2d356210f1e66f50519cc4d58633a13db9004 Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Thu, 19 Dec 2024 15:41:06 +0100 +Subject: [PATCH 14/26] s390x/s390-virtio-ccw: move setting the maximum guest + size from sclp to machine code + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [14/26] a5970c1c6d8d09a473a25a7eee533ec3a6711ec8 (thuth/qemu-kvm-cs) + +Nowadays, it feels more natural to have that code located in +s390_memory_init(), where we also have direct access to the machine +object. + +While at it, use the actual RAM size, not the maximum RAM size which +cannot currently be reached without support for any memory devices. +Consequently update s390_pv_vm_try_disable_async() to rely on the RAM size +as well, to avoid temporary issues while we further rework that +handling. + +set_memory_limit() is temporary, we'll merge it with +s390_set_memory_limit() next. + +Message-ID: <20241219144115.2820241-6-david@redhat.com> +Acked-by: Michael S. Tsirkin +Reviewed-by: Thomas Huth +Signed-off-by: David Hildenbrand +(cherry picked from commit 3c6fb557d295949bea291c3bf88ee9c83392e78c) +Signed-off-by: Thomas Huth +--- + hw/s390x/s390-virtio-ccw.c | 28 ++++++++++++++++++++++++---- + hw/s390x/sclp.c | 11 ----------- + target/s390x/kvm/pv.c | 2 +- + 3 files changed, 25 insertions(+), 16 deletions(-) + +diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c +index d47e99028e..248ac28d20 100644 +--- a/hw/s390x/s390-virtio-ccw.c ++++ b/hw/s390x/s390-virtio-ccw.c +@@ -121,11 +121,29 @@ static void subsystem_reset(void) + } + } + +-static void s390_memory_init(MemoryRegion *ram) ++static void set_memory_limit(uint64_t new_limit) ++{ ++ uint64_t hw_limit; ++ int ret; ++ ++ ret = s390_set_memory_limit(new_limit, &hw_limit); ++ if (ret == -E2BIG) { ++ error_report("host supports a maximum of %" PRIu64 " GB", ++ hw_limit / GiB); ++ exit(EXIT_FAILURE); ++ } else if (ret) { ++ error_report("setting the guest size failed"); ++ exit(EXIT_FAILURE); ++ } ++} ++ ++static void s390_memory_init(MachineState *machine) + { + MemoryRegion *sysmem = get_system_memory(); ++ MemoryRegion *ram = machine->ram; ++ uint64_t ram_size = memory_region_size(ram); + +- if (!QEMU_IS_ALIGNED(memory_region_size(ram), 1 * MiB)) { ++ if (!QEMU_IS_ALIGNED(ram_size, 1 * MiB)) { + /* + * SCLP cannot possibly expose smaller granularity right now and KVM + * cannot handle smaller granularity. As we don't support NUMA, the +@@ -136,7 +154,9 @@ static void s390_memory_init(MemoryRegion *ram) + exit(EXIT_FAILURE); + } + +- /* allocate RAM for core */ ++ set_memory_limit(ram_size); ++ ++ /* Map the initial memory. Must happen after setting the memory limit. */ + memory_region_add_subregion(sysmem, 0, ram); + + /* +@@ -211,7 +231,7 @@ static void ccw_init(MachineState *machine) + qdev_realize_and_unref(DEVICE(ms->sclp), NULL, &error_fatal); + + /* init memory + setup max page size. Required for the CPU model */ +- s390_memory_init(machine->ram); ++ s390_memory_init(machine); + + /* init CPUs (incl. CPU model) early so s390_has_feature() works */ + s390_init_cpus(machine); +diff --git a/hw/s390x/sclp.c b/hw/s390x/sclp.c +index 8757626b5c..73e88ab4eb 100644 +--- a/hw/s390x/sclp.c ++++ b/hw/s390x/sclp.c +@@ -376,10 +376,7 @@ void sclp_service_interrupt(uint32_t sccb) + /* qemu object creation and initialization functions */ + static void sclp_realize(DeviceState *dev, Error **errp) + { +- MachineState *machine = MACHINE(qdev_get_machine()); + SCLPDevice *sclp = SCLP(dev); +- uint64_t hw_limit; +- int ret; + + /* + * qdev_device_add searches the sysbus for TYPE_SCLP_EVENTS_BUS. As long +@@ -389,14 +386,6 @@ static void sclp_realize(DeviceState *dev, Error **errp) + if (!sysbus_realize(SYS_BUS_DEVICE(sclp->event_facility), errp)) { + return; + } +- +- ret = s390_set_memory_limit(machine->maxram_size, &hw_limit); +- if (ret == -E2BIG) { +- error_setg(errp, "host supports a maximum of %" PRIu64 " GB", +- hw_limit / GiB); +- } else if (ret) { +- error_setg(errp, "setting the guest size failed"); +- } + } + + static void sclp_memory_init(SCLPDevice *sclp) +diff --git a/target/s390x/kvm/pv.c b/target/s390x/kvm/pv.c +index dde836d21a..424cce75ca 100644 +--- a/target/s390x/kvm/pv.c ++++ b/target/s390x/kvm/pv.c +@@ -133,7 +133,7 @@ bool s390_pv_vm_try_disable_async(S390CcwMachineState *ms) + * If the feature is not present or if the VM is not larger than 2 GiB, + * KVM_PV_ASYNC_CLEANUP_PREPARE fill fail; no point in attempting it. + */ +- if ((MACHINE(ms)->maxram_size <= 2 * GiB) || ++ if ((MACHINE(ms)->ram_size <= 2 * GiB) || + !kvm_check_extension(kvm_state, KVM_CAP_S390_PROTECTED_ASYNC_DISABLE)) { + return false; + } +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-s390-virtio-ccw-prepare-for-memory-devices.patch b/SOURCES/kvm-s390x-s390-virtio-ccw-prepare-for-memory-devices.patch new file mode 100644 index 0000000..17e7e6a --- /dev/null +++ b/SOURCES/kvm-s390x-s390-virtio-ccw-prepare-for-memory-devices.patch @@ -0,0 +1,117 @@ +From 0e7d7bf86fb242c1ea90bf9648fb061626790eda Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Thu, 19 Dec 2024 15:41:11 +0100 +Subject: [PATCH 19/26] s390x/s390-virtio-ccw: prepare for memory devices + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [19/26] 2441c8c5f5a06d5ca93188dd44e8a08f06d1722b (thuth/qemu-kvm-cs) + +Let's prepare our address space for memory devices if enabled via +"maxmem" and if we have CONFIG_MEM_DEVICE enabled at all. Note that +CONFIG_MEM_DEVICE will be selected automatically once we add support +for devices. + +Just like on other architectures, the region container for memory devices +is placed directly above our initial memory. For now, we only align the +start address of the region up to 1 GiB, but we won't add any additional +space to the region for internal alignment purposes; this can be done in +the future if really required. + +The RAM size returned via SCLP is not modified, as this only +covers initial RAM (and standby memory we don't implement) and not memory +devices; clarify that in the docs of read_SCP_info(). Existing OSes without +support for memory devices will keep working as is, even when memory +devices would be attached the VM. + +Guest OSs which support memory devices, such as virtio-mem, will +consult diag500(), to find out the maximum possible pfn. Guest OSes that +don't support memory devices, don't have to be changed and will continue +relying on information provided by SCLP. + +There are no remaining maxram_size users in s390x code, and the remaining +ram_size users only care about initial RAM: +* hw/s390x/ipl.c +* hw/s390x/s390-hypercall.c +* hw/s390x/sclp.c +* target/s390x/kvm/pv.c + +Message-ID: <20241219144115.2820241-11-david@redhat.com> +Acked-by: Michael S. Tsirkin +Reviewed-by: Thomas Huth +Signed-off-by: David Hildenbrand +(cherry picked from commit 1e86400298cf0fed5f7d49427db477775b859093) +Signed-off-by: Thomas Huth +--- + hw/s390x/s390-virtio-ccw.c | 23 ++++++++++++++++++++++- + hw/s390x/sclp.c | 6 +++++- + 2 files changed, 27 insertions(+), 2 deletions(-) + +diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c +index f5f147eb92..824c73536a 100644 +--- a/hw/s390x/s390-virtio-ccw.c ++++ b/hw/s390x/s390-virtio-ccw.c +@@ -149,6 +149,7 @@ static void s390_memory_init(MachineState *machine) + MemoryRegion *sysmem = get_system_memory(); + MemoryRegion *ram = machine->ram; + uint64_t ram_size = memory_region_size(ram); ++ uint64_t devmem_base, devmem_size; + + if (!QEMU_IS_ALIGNED(ram_size, 1 * MiB)) { + /* +@@ -161,11 +162,31 @@ static void s390_memory_init(MachineState *machine) + exit(EXIT_FAILURE); + } + +- s390_set_memory_limit(s390ms, ram_size); ++ devmem_size = 0; ++ devmem_base = ram_size; ++#ifdef CONFIG_MEM_DEVICE ++ if (machine->ram_size < machine->maxram_size) { ++ ++ /* ++ * Make sure memory devices have a sane default alignment, even ++ * when weird initial memory sizes are specified. ++ */ ++ devmem_base = QEMU_ALIGN_UP(devmem_base, 1 * GiB); ++ devmem_size = machine->maxram_size - machine->ram_size; ++ } ++#endif ++ s390_set_memory_limit(s390ms, devmem_base + devmem_size); + + /* Map the initial memory. Must happen after setting the memory limit. */ + memory_region_add_subregion(sysmem, 0, ram); + ++ /* Initialize address space for memory devices. */ ++#ifdef CONFIG_MEM_DEVICE ++ if (devmem_size) { ++ machine_memory_devices_init(machine, devmem_base, devmem_size); ++ } ++#endif /* CONFIG_MEM_DEVICE */ ++ + /* + * Configure the maximum page size. As no memory devices were created + * yet, this is the page size of initial memory only. +diff --git a/hw/s390x/sclp.c b/hw/s390x/sclp.c +index 73e88ab4eb..5945c9b1d8 100644 +--- a/hw/s390x/sclp.c ++++ b/hw/s390x/sclp.c +@@ -161,7 +161,11 @@ static void read_SCP_info(SCLPDevice *sclp, SCCB *sccb) + read_info->rnsize2 = cpu_to_be32(rnsize); + } + +- /* we don't support standby memory, maxram_size is never exposed */ ++ /* ++ * We don't support standby memory. maxram_size is used for sizing the ++ * memory device region, which is not exposed through SCLP but through ++ * diag500. ++ */ + rnmax = machine->ram_size >> sclp->increment_size; + if (rnmax < 0x10000) { + read_info->rnmax = cpu_to_be16(rnmax); +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-s390-virtio-hcall-prepare-for-more-diag500-hyp.patch b/SOURCES/kvm-s390x-s390-virtio-hcall-prepare-for-more-diag500-hyp.patch new file mode 100644 index 0000000..4bc283e --- /dev/null +++ b/SOURCES/kvm-s390x-s390-virtio-hcall-prepare-for-more-diag500-hyp.patch @@ -0,0 +1,163 @@ +From d2764db41fc6edcead9ad27b8d31e7bff524c0c0 Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Thu, 19 Dec 2024 15:41:04 +0100 +Subject: [PATCH 12/26] s390x/s390-virtio-hcall: prepare for more diag500 + hypercalls + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [12/26] 6573602d71b9e70679a48315f913309be29d6239 (thuth/qemu-kvm-cs) + +Let's generalize, abstracting the virtio bits. diag500 is now a generic +hypercall to handle QEMU/KVM specific things. Explicitly specify all +already defined subcodes, including legacy ones (so we know what we can +use for new hypercalls). + +Move the PGM_SPECIFICATION injection into the renamed function +handle_diag_500(), so we can turn it into a void function. + +We'll rename the files separately, so git properly detects the rename. + +Message-ID: <20241219144115.2820241-4-david@redhat.com> +Acked-by: Michael S. Tsirkin +Reviewed-by: Thomas Huth +Signed-off-by: David Hildenbrand +(cherry picked from commit 6e9cc2da4e8b997fd6ff3249034f436b84fc7974) +Signed-off-by: Thomas Huth +--- + hw/s390x/s390-virtio-hcall.c | 15 ++++++++------- + hw/s390x/s390-virtio-hcall.h | 11 ++++++----- + target/s390x/kvm/kvm.c | 20 +++----------------- + target/s390x/tcg/misc_helper.c | 5 +++-- + 4 files changed, 20 insertions(+), 31 deletions(-) + +diff --git a/hw/s390x/s390-virtio-hcall.c b/hw/s390x/s390-virtio-hcall.c +index ca49e3cd22..5fb78a719e 100644 +--- a/hw/s390x/s390-virtio-hcall.c ++++ b/hw/s390x/s390-virtio-hcall.c +@@ -1,5 +1,5 @@ + /* +- * Support for virtio hypercalls on s390 ++ * Support for QEMU/KVM hypercalls on s390 + * + * Copyright 2012 IBM Corp. + * Author(s): Cornelia Huck +@@ -57,18 +57,19 @@ static int handle_virtio_ccw_notify(uint64_t subch_id, uint64_t data) + return 0; + } + +-int s390_virtio_hypercall(CPUS390XState *env) ++void handle_diag_500(S390CPU *cpu, uintptr_t ra) + { ++ CPUS390XState *env = &cpu->env; + const uint64_t subcode = env->regs[1]; + + switch (subcode) { +- case KVM_S390_VIRTIO_NOTIFY: ++ case DIAG500_VIRTIO_NOTIFY: + env->regs[2] = handle_virtio_notify(env->regs[2]); +- return 0; +- case KVM_S390_VIRTIO_CCW_NOTIFY: ++ break; ++ case DIAG500_VIRTIO_CCW_NOTIFY: + env->regs[2] = handle_virtio_ccw_notify(env->regs[2], env->regs[3]); +- return 0; ++ break; + default: +- return -EINVAL; ++ s390_program_interrupt(env, PGM_SPECIFICATION, ra); + } + } +diff --git a/hw/s390x/s390-virtio-hcall.h b/hw/s390x/s390-virtio-hcall.h +index 3d9fe147d2..dca456b926 100644 +--- a/hw/s390x/s390-virtio-hcall.h ++++ b/hw/s390x/s390-virtio-hcall.h +@@ -1,5 +1,5 @@ + /* +- * Support for virtio hypercalls on s390x ++ * Support for QEMU/KVM hypercalls on s390x + * + * Copyright IBM Corp. 2012, 2017 + * Author(s): Cornelia Huck +@@ -12,12 +12,13 @@ + #ifndef HW_S390_VIRTIO_HCALL_H + #define HW_S390_VIRTIO_HCALL_H + +-#include "standard-headers/asm-s390/virtio-ccw.h" + #include "cpu.h" + +-/* The only thing that we need from the old kvm_virtio.h file */ +-#define KVM_S390_VIRTIO_NOTIFY 0 ++#define DIAG500_VIRTIO_NOTIFY 0 /* legacy, implemented as a NOP */ ++#define DIAG500_VIRTIO_RESET 1 /* legacy */ ++#define DIAG500_VIRTIO_SET_STATUS 2 /* legacy */ ++#define DIAG500_VIRTIO_CCW_NOTIFY 3 /* KVM_S390_VIRTIO_CCW_NOTIFY */ + +-int s390_virtio_hypercall(CPUS390XState *env); ++void handle_diag_500(S390CPU *cpu, uintptr_t ra); + + #endif /* HW_S390_VIRTIO_HCALL_H */ +diff --git a/target/s390x/kvm/kvm.c b/target/s390x/kvm/kvm.c +index 5947dda829..42d6a54126 100644 +--- a/target/s390x/kvm/kvm.c ++++ b/target/s390x/kvm/kvm.c +@@ -1492,22 +1492,6 @@ static int handle_e3(S390CPU *cpu, struct kvm_run *run, uint8_t ipbl) + return r; + } + +-static int handle_hypercall(S390CPU *cpu, struct kvm_run *run) +-{ +- CPUS390XState *env = &cpu->env; +- int ret = -EINVAL; +- +-#ifdef CONFIG_S390_CCW_VIRTIO +- ret = s390_virtio_hypercall(env); +-#endif /* CONFIG_S390_CCW_VIRTIO */ +- if (ret == -EINVAL) { +- kvm_s390_program_interrupt(cpu, PGM_SPECIFICATION); +- return 0; +- } +- +- return ret; +-} +- + static void kvm_handle_diag_288(S390CPU *cpu, struct kvm_run *run) + { + uint64_t r1, r3; +@@ -1603,9 +1587,11 @@ static int handle_diag(S390CPU *cpu, struct kvm_run *run, uint32_t ipb) + case DIAG_SET_CONTROL_PROGRAM_CODES: + handle_diag_318(cpu, run); + break; ++#ifdef CONFIG_S390_CCW_VIRTIO + case DIAG_KVM_HYPERCALL: +- r = handle_hypercall(cpu, run); ++ handle_diag_500(cpu, RA_IGNORED); + break; ++#endif /* CONFIG_S390_CCW_VIRTIO */ + case DIAG_KVM_BREAKPOINT: + r = handle_sw_breakpoint(cpu, run); + break; +diff --git a/target/s390x/tcg/misc_helper.c b/target/s390x/tcg/misc_helper.c +index f44136a568..2b4310003b 100644 +--- a/target/s390x/tcg/misc_helper.c ++++ b/target/s390x/tcg/misc_helper.c +@@ -119,10 +119,11 @@ void HELPER(diag)(CPUS390XState *env, uint32_t r1, uint32_t r3, uint32_t num) + switch (num) { + #ifdef CONFIG_S390_CCW_VIRTIO + case 0x500: +- /* KVM hypercall */ ++ /* QEMU/KVM hypercall */ + bql_lock(); +- r = s390_virtio_hypercall(env); ++ handle_diag_500(env_archcpu(env), GETPC()); + bql_unlock(); ++ r = 0; + break; + #endif /* CONFIG_S390_CCW_VIRTIO */ + case 0x44: +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-s390-virtio-hcall-remove-hypercall-registratio.patch b/SOURCES/kvm-s390x-s390-virtio-hcall-remove-hypercall-registratio.patch new file mode 100644 index 0000000..14d03de --- /dev/null +++ b/SOURCES/kvm-s390x-s390-virtio-hcall-remove-hypercall-registratio.patch @@ -0,0 +1,296 @@ +From 16ccb16d393a3e63936dc993c30c67fdecb1f120 Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Thu, 19 Dec 2024 15:41:03 +0100 +Subject: [PATCH 11/26] s390x/s390-virtio-hcall: remove hypercall registration + mechanism + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [11/26] 5e8d2720fe9fd6e6e24487d71988821f1cf27f17 (thuth/qemu-kvm-cs) + +Nowadays, we only have a single machine type in QEMU, everything is based +on virtio-ccw and the traditional virtio machine does no longer exist. No +need to dynamically register diag500 handlers. Move the two existing +handlers into s390-virtio-hcall.c. + +Message-ID: <20241219144115.2820241-3-david@redhat.com> +Acked-by: Michael S. Tsirkin +Reviewed-by: Thomas Huth +Acked-by: Christian Borntraeger +Signed-off-by: David Hildenbrand +(cherry picked from commit 4be0fce498d0a08f18b3a9accdb9ded79484d30a) +Signed-off-by: Thomas Huth +--- + hw/s390x/meson.build | 6 ++-- + hw/s390x/s390-virtio-ccw.c | 58 ------------------------------ + hw/s390x/s390-virtio-hcall.c | 65 +++++++++++++++++++++++++--------- + hw/s390x/s390-virtio-hcall.h | 2 -- + target/s390x/kvm/kvm.c | 5 ++- + target/s390x/tcg/misc_helper.c | 3 ++ + 6 files changed, 60 insertions(+), 79 deletions(-) + +diff --git a/hw/s390x/meson.build b/hw/s390x/meson.build +index 482fd13420..d6c8c33915 100644 +--- a/hw/s390x/meson.build ++++ b/hw/s390x/meson.build +@@ -12,7 +12,6 @@ s390x_ss.add(files( + 's390-pci-inst.c', + 's390-skeys.c', + 's390-stattrib.c', +- 's390-virtio-hcall.c', + 'sclp.c', + 'sclpcpu.c', + 'sclpquiesce.c', +@@ -28,7 +27,10 @@ s390x_ss.add(when: 'CONFIG_KVM', if_true: files( + s390x_ss.add(when: 'CONFIG_TCG', if_true: files( + 'tod-tcg.c', + )) +-s390x_ss.add(when: 'CONFIG_S390_CCW_VIRTIO', if_true: files('s390-virtio-ccw.c')) ++s390x_ss.add(when: 'CONFIG_S390_CCW_VIRTIO', if_true: files( ++ 's390-virtio-ccw.c', ++ 's390-virtio-hcall.c', ++)) + s390x_ss.add(when: 'CONFIG_TERMINAL3270', if_true: files('3270-ccw.c')) + s390x_ss.add(when: 'CONFIG_VFIO', if_true: files('s390-pci-vfio.c')) + +diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c +index 82ded9666c..d47e99028e 100644 +--- a/hw/s390x/s390-virtio-ccw.c ++++ b/hw/s390x/s390-virtio-ccw.c +@@ -16,11 +16,8 @@ + #include "exec/ram_addr.h" + #include "exec/confidential-guest-support.h" + #include "hw/boards.h" +-#include "hw/s390x/s390-virtio-hcall.h" + #include "hw/s390x/sclp.h" + #include "hw/s390x/s390_flic.h" +-#include "hw/s390x/ioinst.h" +-#include "hw/s390x/css.h" + #include "virtio-ccw.h" + #include "qemu/config-file.h" + #include "qemu/ctype.h" +@@ -124,58 +121,6 @@ static void subsystem_reset(void) + } + } + +-static int virtio_ccw_hcall_notify(const uint64_t *args) +-{ +- uint64_t subch_id = args[0]; +- uint64_t data = args[1]; +- SubchDev *sch; +- VirtIODevice *vdev; +- int cssid, ssid, schid, m; +- uint16_t vq_idx = data; +- +- if (ioinst_disassemble_sch_ident(subch_id, &m, &cssid, &ssid, &schid)) { +- return -EINVAL; +- } +- sch = css_find_subch(m, cssid, ssid, schid); +- if (!sch || !css_subch_visible(sch)) { +- return -EINVAL; +- } +- +- vdev = virtio_ccw_get_vdev(sch); +- if (vq_idx >= VIRTIO_QUEUE_MAX || !virtio_queue_get_num(vdev, vq_idx)) { +- return -EINVAL; +- } +- +- if (virtio_vdev_has_feature(vdev, VIRTIO_F_NOTIFICATION_DATA)) { +- virtio_queue_set_shadow_avail_idx(virtio_get_queue(vdev, vq_idx), +- (data >> 16) & 0xFFFF); +- } +- +- virtio_queue_notify(vdev, vq_idx); +- return 0; +-} +- +-static int virtio_ccw_hcall_early_printk(const uint64_t *args) +-{ +- uint64_t mem = args[0]; +- MachineState *ms = MACHINE(qdev_get_machine()); +- +- if (mem < ms->ram_size) { +- /* Early printk */ +- return 0; +- } +- return -EINVAL; +-} +- +-static void virtio_ccw_register_hcalls(void) +-{ +- s390_register_virtio_hypercall(KVM_S390_VIRTIO_CCW_NOTIFY, +- virtio_ccw_hcall_notify); +- /* Tolerate early printk. */ +- s390_register_virtio_hypercall(KVM_S390_VIRTIO_NOTIFY, +- virtio_ccw_hcall_early_printk); +-} +- + static void s390_memory_init(MemoryRegion *ram) + { + MemoryRegion *sysmem = get_system_memory(); +@@ -296,9 +241,6 @@ static void ccw_init(MachineState *machine) + OBJECT(dev)); + sysbus_realize_and_unref(SYS_BUS_DEVICE(dev), &error_fatal); + +- /* register hypercalls */ +- virtio_ccw_register_hcalls(); +- + s390_enable_css_support(s390_cpu_addr2state(0)); + + ret = css_create_css_image(VIRTUAL_CSSID, true); +diff --git a/hw/s390x/s390-virtio-hcall.c b/hw/s390x/s390-virtio-hcall.c +index ec7cf8beb3..ca49e3cd22 100644 +--- a/hw/s390x/s390-virtio-hcall.c ++++ b/hw/s390x/s390-virtio-hcall.c +@@ -11,31 +11,64 @@ + + #include "qemu/osdep.h" + #include "cpu.h" ++#include "hw/boards.h" + #include "hw/s390x/s390-virtio-hcall.h" ++#include "hw/s390x/ioinst.h" ++#include "hw/s390x/css.h" ++#include "virtio-ccw.h" + +-#define MAX_DIAG_SUBCODES 255 ++static int handle_virtio_notify(uint64_t mem) ++{ ++ MachineState *ms = MACHINE(qdev_get_machine()); + +-static s390_virtio_fn s390_diag500_table[MAX_DIAG_SUBCODES]; ++ if (mem < ms->ram_size) { ++ /* Early printk */ ++ return 0; ++ } ++ return -EINVAL; ++} + +-void s390_register_virtio_hypercall(uint64_t code, s390_virtio_fn fn) ++static int handle_virtio_ccw_notify(uint64_t subch_id, uint64_t data) + { +- assert(code < MAX_DIAG_SUBCODES); +- assert(!s390_diag500_table[code]); ++ SubchDev *sch; ++ VirtIODevice *vdev; ++ int cssid, ssid, schid, m; ++ uint16_t vq_idx = data; ++ ++ if (ioinst_disassemble_sch_ident(subch_id, &m, &cssid, &ssid, &schid)) { ++ return -EINVAL; ++ } ++ sch = css_find_subch(m, cssid, ssid, schid); ++ if (!sch || !css_subch_visible(sch)) { ++ return -EINVAL; ++ } + +- s390_diag500_table[code] = fn; ++ vdev = virtio_ccw_get_vdev(sch); ++ if (vq_idx >= VIRTIO_QUEUE_MAX || !virtio_queue_get_num(vdev, vq_idx)) { ++ return -EINVAL; ++ } ++ ++ if (virtio_vdev_has_feature(vdev, VIRTIO_F_NOTIFICATION_DATA)) { ++ virtio_queue_set_shadow_avail_idx(virtio_get_queue(vdev, vq_idx), ++ (data >> 16) & 0xFFFF); ++ } ++ ++ virtio_queue_notify(vdev, vq_idx); ++ return 0; + } + + int s390_virtio_hypercall(CPUS390XState *env) + { +- s390_virtio_fn fn; +- +- if (env->regs[1] < MAX_DIAG_SUBCODES) { +- fn = s390_diag500_table[env->regs[1]]; +- if (fn) { +- env->regs[2] = fn(&env->regs[2]); +- return 0; +- } +- } ++ const uint64_t subcode = env->regs[1]; + +- return -EINVAL; ++ switch (subcode) { ++ case KVM_S390_VIRTIO_NOTIFY: ++ env->regs[2] = handle_virtio_notify(env->regs[2]); ++ return 0; ++ case KVM_S390_VIRTIO_CCW_NOTIFY: ++ env->regs[2] = handle_virtio_ccw_notify(env->regs[2], env->regs[3]); ++ return 0; ++ default: ++ return -EINVAL; ++ } + } +diff --git a/hw/s390x/s390-virtio-hcall.h b/hw/s390x/s390-virtio-hcall.h +index 3ae6d6ae3a..3d9fe147d2 100644 +--- a/hw/s390x/s390-virtio-hcall.h ++++ b/hw/s390x/s390-virtio-hcall.h +@@ -18,8 +18,6 @@ + /* The only thing that we need from the old kvm_virtio.h file */ + #define KVM_S390_VIRTIO_NOTIFY 0 + +-typedef int (*s390_virtio_fn)(const uint64_t *args); +-void s390_register_virtio_hypercall(uint64_t code, s390_virtio_fn fn); + int s390_virtio_hypercall(CPUS390XState *env); + + #endif /* HW_S390_VIRTIO_HCALL_H */ +diff --git a/target/s390x/kvm/kvm.c b/target/s390x/kvm/kvm.c +index 7a0ca5570f..5947dda829 100644 +--- a/target/s390x/kvm/kvm.c ++++ b/target/s390x/kvm/kvm.c +@@ -51,6 +51,7 @@ + #include "hw/s390x/s390-virtio-ccw.h" + #include "hw/s390x/s390-virtio-hcall.h" + #include "target/s390x/kvm/pv.h" ++#include CONFIG_DEVICES + + #define kvm_vm_check_mem_attr(s, attr) \ + kvm_vm_check_attr(s, KVM_S390_VM_MEM_CTRL, attr) +@@ -1494,9 +1495,11 @@ static int handle_e3(S390CPU *cpu, struct kvm_run *run, uint8_t ipbl) + static int handle_hypercall(S390CPU *cpu, struct kvm_run *run) + { + CPUS390XState *env = &cpu->env; +- int ret; ++ int ret = -EINVAL; + ++#ifdef CONFIG_S390_CCW_VIRTIO + ret = s390_virtio_hypercall(env); ++#endif /* CONFIG_S390_CCW_VIRTIO */ + if (ret == -EINVAL) { + kvm_s390_program_interrupt(cpu, PGM_SPECIFICATION); + return 0; +diff --git a/target/s390x/tcg/misc_helper.c b/target/s390x/tcg/misc_helper.c +index 303f86d363..f44136a568 100644 +--- a/target/s390x/tcg/misc_helper.c ++++ b/target/s390x/tcg/misc_helper.c +@@ -43,6 +43,7 @@ + #include "hw/s390x/s390-pci-inst.h" + #include "hw/boards.h" + #include "hw/s390x/tod.h" ++#include CONFIG_DEVICES + #endif + + /* #define DEBUG_HELPER */ +@@ -116,12 +117,14 @@ void HELPER(diag)(CPUS390XState *env, uint32_t r1, uint32_t r3, uint32_t num) + uint64_t r; + + switch (num) { ++#ifdef CONFIG_S390_CCW_VIRTIO + case 0x500: + /* KVM hypercall */ + bql_lock(); + r = s390_virtio_hypercall(env); + bql_unlock(); + break; ++#endif /* CONFIG_S390_CCW_VIRTIO */ + case 0x44: + /* yield */ + r = 0; +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-virtio-ccw-add-support-for-virtio-based-memory.patch b/SOURCES/kvm-s390x-virtio-ccw-add-support-for-virtio-based-memory.patch new file mode 100644 index 0000000..808f95b --- /dev/null +++ b/SOURCES/kvm-s390x-virtio-ccw-add-support-for-virtio-based-memory.patch @@ -0,0 +1,423 @@ +From 6b82fca2ecac0c7b30780ebb71ce5bad0421b9b4 Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Thu, 19 Dec 2024 15:41:14 +0100 +Subject: [PATCH 22/26] s390x/virtio-ccw: add support for virtio based memory + devices + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [22/26] 270a9fbe7e5bacfa6c9377815a01da26c4d26097 (thuth/qemu-kvm-cs) + +Let's implement support for abstract virtio based memory devices, using +the virtio-pci implementation as an orientation. Wire them up in the +machine hotplug handler, taking care of s390x page size limitations. + +As we neither support virtio-mem or virtio-pmem yet, the code is +effectively unused. We'll implement support for virtio-mem based on this +next. + +Note that we won't wire up the virtio-pci variant (should currently be +impossible due to lack of support for MSI-X), but we'll add a safety net +to reject plugging them in the pre-plug handler. + +Message-ID: <20241219144115.2820241-14-david@redhat.com> +Acked-by: Michael S. Tsirkin +Signed-off-by: David Hildenbrand +(cherry picked from commit 88d86f6f1e36741ba9e1625da19a7ccf1a343d39) +Signed-off-by: Thomas Huth +--- + MAINTAINERS | 3 + + hw/s390x/meson.build | 3 + + hw/s390x/s390-virtio-ccw.c | 47 +++++++++- + hw/s390x/virtio-ccw-md-stubs.c | 24 ++++++ + hw/s390x/virtio-ccw-md.c | 153 +++++++++++++++++++++++++++++++++ + hw/s390x/virtio-ccw-md.h | 44 ++++++++++ + hw/virtio/Kconfig | 1 + + 7 files changed, 274 insertions(+), 1 deletion(-) + create mode 100644 hw/s390x/virtio-ccw-md-stubs.c + create mode 100644 hw/s390x/virtio-ccw-md.c + create mode 100644 hw/s390x/virtio-ccw-md.h + +diff --git a/MAINTAINERS b/MAINTAINERS +index 3584d6a6c6..f21dc3fa75 100644 +--- a/MAINTAINERS ++++ b/MAINTAINERS +@@ -2387,6 +2387,9 @@ F: include/hw/virtio/virtio-crypto.h + virtio based memory device + M: David Hildenbrand + S: Supported ++F: hw/s390x/virtio-ccw-md.c ++F: hw/s390x/virtio-ccw-md.h ++F: hw/s390x/virtio-ccw-md-stubs.c + F: hw/virtio/virtio-md-pci.c + F: include/hw/virtio/virtio-md-pci.h + F: stubs/virtio-md-pci.c +diff --git a/hw/s390x/meson.build b/hw/s390x/meson.build +index e344a3bd8c..4431868408 100644 +--- a/hw/s390x/meson.build ++++ b/hw/s390x/meson.build +@@ -50,8 +50,11 @@ endif + virtio_ss.add(when: 'CONFIG_VHOST_SCSI', if_true: files('vhost-scsi-ccw.c')) + virtio_ss.add(when: 'CONFIG_VHOST_VSOCK', if_true: files('vhost-vsock-ccw.c')) + virtio_ss.add(when: 'CONFIG_VHOST_USER_FS', if_true: files('vhost-user-fs-ccw.c')) ++virtio_ss.add(when: 'CONFIG_VIRTIO_MD', if_true: files('virtio-ccw-md.c')) + s390x_ss.add_all(when: 'CONFIG_VIRTIO_CCW', if_true: virtio_ss) + ++s390x_ss.add(when: 'CONFIG_VIRTIO_MD', if_false: files('virtio-ccw-md-stubs.c')) ++ + hw_arch += {'s390x': s390x_ss} + + hw_s390x_modules = {} +diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c +index bd05a22b4e..9f4ad01789 100644 +--- a/hw/s390x/s390-virtio-ccw.c ++++ b/hw/s390x/s390-virtio-ccw.c +@@ -46,6 +46,8 @@ + #include "qapi/visitor.h" + #include "hw/s390x/cpu-topology.h" + #include "kvm/kvm_s390x.h" ++#include "hw/virtio/virtio-md-pci.h" ++#include "hw/s390x/virtio-ccw-md.h" + #include CONFIG_DEVICES + + static Error *pv_mig_blocker; +@@ -546,11 +548,39 @@ static void s390_machine_reset(MachineState *machine, ResetType type) + s390_ipl_clear_reset_request(); + } + ++static void s390_machine_device_pre_plug(HotplugHandler *hotplug_dev, ++ DeviceState *dev, Error **errp) ++{ ++ if (object_dynamic_cast(OBJECT(dev), TYPE_VIRTIO_MD_CCW)) { ++ virtio_ccw_md_pre_plug(VIRTIO_MD_CCW(dev), MACHINE(hotplug_dev), errp); ++ } else if (object_dynamic_cast(OBJECT(dev), TYPE_VIRTIO_MD_PCI)) { ++ error_setg(errp, ++ "PCI-attached virtio based memory devices not supported"); ++ } ++} ++ + static void s390_machine_device_plug(HotplugHandler *hotplug_dev, + DeviceState *dev, Error **errp) + { ++ S390CcwMachineState *s390ms = S390_CCW_MACHINE(hotplug_dev); ++ + if (object_dynamic_cast(OBJECT(dev), TYPE_CPU)) { + s390_cpu_plug(hotplug_dev, dev, errp); ++ } else if (object_dynamic_cast(OBJECT(dev), TYPE_VIRTIO_MD_CCW)) { ++ /* ++ * At this point, the device is realized and set all memdevs mapped, so ++ * qemu_maxrampagesize() will pick up the page sizes of these memdevs ++ * as well. Before we plug the device and expose any RAM memory regions ++ * to the system, make sure we don't exceed the previously set max page ++ * size. While only relevant for KVM, there is not really any use case ++ * for this with TCG, so we'll unconditionally reject it. ++ */ ++ if (qemu_maxrampagesize() != s390ms->max_pagesize) { ++ error_setg(errp, "Memory device uses a bigger page size than" ++ " initial memory"); ++ return; ++ } ++ virtio_ccw_md_plug(VIRTIO_MD_CCW(dev), MACHINE(hotplug_dev), errp); + } + } + +@@ -560,9 +590,20 @@ static void s390_machine_device_unplug_request(HotplugHandler *hotplug_dev, + if (object_dynamic_cast(OBJECT(dev), TYPE_CPU)) { + error_setg(errp, "CPU hot unplug not supported on this machine"); + return; ++ } else if (object_dynamic_cast(OBJECT(dev), TYPE_VIRTIO_MD_CCW)) { ++ virtio_ccw_md_unplug_request(VIRTIO_MD_CCW(dev), MACHINE(hotplug_dev), ++ errp); + } + } + ++static void s390_machine_device_unplug(HotplugHandler *hotplug_dev, ++ DeviceState *dev, Error **errp) ++{ ++ if (object_dynamic_cast(OBJECT(dev), TYPE_VIRTIO_MD_CCW)) { ++ virtio_ccw_md_unplug(VIRTIO_MD_CCW(dev), MACHINE(hotplug_dev), errp); ++ } ++ } ++ + static CpuInstanceProperties s390_cpu_index_to_props(MachineState *ms, + unsigned cpu_index) + { +@@ -609,7 +650,9 @@ static const CPUArchIdList *s390_possible_cpu_arch_ids(MachineState *ms) + static HotplugHandler *s390_get_hotplug_handler(MachineState *machine, + DeviceState *dev) + { +- if (object_dynamic_cast(OBJECT(dev), TYPE_CPU)) { ++ if (object_dynamic_cast(OBJECT(dev), TYPE_CPU) || ++ object_dynamic_cast(OBJECT(dev), TYPE_VIRTIO_MD_CCW) || ++ object_dynamic_cast(OBJECT(dev), TYPE_VIRTIO_MD_PCI)) { + return HOTPLUG_HANDLER(machine); + } + return NULL; +@@ -769,8 +812,10 @@ static void ccw_machine_class_init(ObjectClass *oc, void *data) + mc->possible_cpu_arch_ids = s390_possible_cpu_arch_ids; + /* it is overridden with 'host' cpu *in kvm_arch_init* */ + mc->default_cpu_type = S390_CPU_TYPE_NAME("qemu"); ++ hc->pre_plug = s390_machine_device_pre_plug; + hc->plug = s390_machine_device_plug; + hc->unplug_request = s390_machine_device_unplug_request; ++ hc->unplug = s390_machine_device_unplug; + nc->nmi_monitor_handler = s390_nmi; + mc->default_ram_id = "s390.ram"; + mc->default_nic = "virtio-net-ccw"; +diff --git a/hw/s390x/virtio-ccw-md-stubs.c b/hw/s390x/virtio-ccw-md-stubs.c +new file mode 100644 +index 0000000000..e937865550 +--- /dev/null ++++ b/hw/s390x/virtio-ccw-md-stubs.c +@@ -0,0 +1,24 @@ ++#include "qemu/osdep.h" ++#include "qapi/error.h" ++#include "hw/s390x/virtio-ccw-md.h" ++ ++void virtio_ccw_md_pre_plug(VirtIOMDCcw *vmd, MachineState *ms, Error **errp) ++{ ++ error_setg(errp, "virtio based memory devices not supported"); ++} ++ ++void virtio_ccw_md_plug(VirtIOMDCcw *vmd, MachineState *ms, Error **errp) ++{ ++ error_setg(errp, "virtio based memory devices not supported"); ++} ++ ++void virtio_ccw_md_unplug_request(VirtIOMDCcw *vmd, MachineState *ms, ++ Error **errp) ++{ ++ error_setg(errp, "virtio based memory devices not supported"); ++} ++ ++void virtio_ccw_md_unplug(VirtIOMDCcw *vmd, MachineState *ms, Error **errp) ++{ ++ error_setg(errp, "virtio based memory devices not supported"); ++} +diff --git a/hw/s390x/virtio-ccw-md.c b/hw/s390x/virtio-ccw-md.c +new file mode 100644 +index 0000000000..de333282df +--- /dev/null ++++ b/hw/s390x/virtio-ccw-md.c +@@ -0,0 +1,153 @@ ++/* ++ * Virtio CCW support for abstract virtio based memory device ++ * ++ * Copyright (C) 2024 Red Hat, Inc. ++ * ++ * Authors: ++ * David Hildenbrand ++ * ++ * This work is licensed under the terms of the GNU GPL, version 2. ++ * See the COPYING file in the top-level directory. ++ */ ++ ++#include "qemu/osdep.h" ++#include "hw/s390x/virtio-ccw-md.h" ++#include "hw/mem/memory-device.h" ++#include "qapi/error.h" ++#include "qemu/error-report.h" ++ ++void virtio_ccw_md_pre_plug(VirtIOMDCcw *vmd, MachineState *ms, Error **errp) ++{ ++ DeviceState *dev = DEVICE(vmd); ++ HotplugHandler *bus_handler = qdev_get_bus_hotplug_handler(dev); ++ MemoryDeviceState *md = MEMORY_DEVICE(vmd); ++ Error *local_err = NULL; ++ ++ if (!bus_handler && dev->hotplugged) { ++ /* ++ * Without a bus hotplug handler, we cannot control the plug/unplug ++ * order. We should never reach this point when hotplugging, but ++ * better add a safety net. ++ */ ++ error_setg(errp, "hotplug of virtio based memory devices not supported" ++ " on this bus."); ++ return; ++ } ++ ++ /* ++ * First, see if we can plug this memory device at all. If that ++ * succeeds, branch of to the actual hotplug handler. ++ */ ++ memory_device_pre_plug(md, ms, &local_err); ++ if (!local_err && bus_handler) { ++ hotplug_handler_pre_plug(bus_handler, dev, &local_err); ++ } ++ error_propagate(errp, local_err); ++} ++ ++void virtio_ccw_md_plug(VirtIOMDCcw *vmd, MachineState *ms, Error **errp) ++{ ++ DeviceState *dev = DEVICE(vmd); ++ HotplugHandler *bus_handler = qdev_get_bus_hotplug_handler(dev); ++ MemoryDeviceState *md = MEMORY_DEVICE(vmd); ++ Error *local_err = NULL; ++ ++ /* ++ * Plug the memory device first and then branch off to the actual ++ * hotplug handler. If that one fails, we can easily undo the memory ++ * device bits. ++ */ ++ memory_device_plug(md, ms); ++ if (bus_handler) { ++ hotplug_handler_plug(bus_handler, dev, &local_err); ++ if (local_err) { ++ memory_device_unplug(md, ms); ++ } ++ } ++ error_propagate(errp, local_err); ++} ++ ++void virtio_ccw_md_unplug_request(VirtIOMDCcw *vmd, MachineState *ms, ++ Error **errp) ++{ ++ VirtIOMDCcwClass *vmdc = VIRTIO_MD_CCW_GET_CLASS(vmd); ++ DeviceState *dev = DEVICE(vmd); ++ HotplugHandler *bus_handler = qdev_get_bus_hotplug_handler(dev); ++ HotplugHandlerClass *hdc; ++ Error *local_err = NULL; ++ ++ if (!vmdc->unplug_request_check) { ++ error_setg(errp, ++ "this virtio based memory devices cannot be unplugged"); ++ return; ++ } ++ ++ if (!bus_handler) { ++ error_setg(errp, "hotunplug of virtio based memory devices not" ++ "supported on this bus"); ++ return; ++ } ++ ++ vmdc->unplug_request_check(vmd, &local_err); ++ if (local_err) { ++ error_propagate(errp, local_err); ++ return; ++ } ++ ++ /* ++ * Forward the async request or turn it into a sync request (handling it ++ * like qdev_unplug()). ++ */ ++ hdc = HOTPLUG_HANDLER_GET_CLASS(bus_handler); ++ if (hdc->unplug_request) { ++ hotplug_handler_unplug_request(bus_handler, dev, &local_err); ++ } else { ++ virtio_ccw_md_unplug(vmd, ms, &local_err); ++ if (!local_err) { ++ object_unparent(OBJECT(dev)); ++ } ++ } ++} ++ ++void virtio_ccw_md_unplug(VirtIOMDCcw *vmd, MachineState *ms, Error **errp) ++{ ++ DeviceState *dev = DEVICE(vmd); ++ HotplugHandler *bus_handler = qdev_get_bus_hotplug_handler(dev); ++ MemoryDeviceState *md = MEMORY_DEVICE(vmd); ++ Error *local_err = NULL; ++ ++ /* Unplug the memory device while it is still realized. */ ++ memory_device_unplug(md, ms); ++ ++ if (bus_handler) { ++ hotplug_handler_unplug(bus_handler, dev, &local_err); ++ if (local_err) { ++ /* Not expected to fail ... but still try to recover. */ ++ memory_device_plug(md, ms); ++ error_propagate(errp, local_err); ++ return; ++ } ++ } else { ++ /* Very unexpected, but let's just try to do the right thing. */ ++ warn_report("Unexpected unplug of virtio based memory device"); ++ qdev_unrealize(dev); ++ } ++} ++ ++static const TypeInfo virtio_ccw_md_info = { ++ .name = TYPE_VIRTIO_MD_CCW, ++ .parent = TYPE_VIRTIO_CCW_DEVICE, ++ .instance_size = sizeof(VirtIOMDCcw), ++ .class_size = sizeof(VirtIOMDCcwClass), ++ .abstract = true, ++ .interfaces = (InterfaceInfo[]) { ++ { TYPE_MEMORY_DEVICE }, ++ { } ++ }, ++}; ++ ++static void virtio_ccw_md_register(void) ++{ ++ type_register_static(&virtio_ccw_md_info); ++} ++type_init(virtio_ccw_md_register) +diff --git a/hw/s390x/virtio-ccw-md.h b/hw/s390x/virtio-ccw-md.h +new file mode 100644 +index 0000000000..39ba864c92 +--- /dev/null ++++ b/hw/s390x/virtio-ccw-md.h +@@ -0,0 +1,44 @@ ++/* ++ * Virtio CCW support for abstract virtio based memory device ++ * ++ * Copyright (C) 2024 Red Hat, Inc. ++ * ++ * Authors: ++ * David Hildenbrand ++ * ++ * This work is licensed under the terms of the GNU GPL, version 2. ++ * See the COPYING file in the top-level directory. ++ */ ++ ++#ifndef HW_S390X_VIRTIO_CCW_MD_H ++#define HW_S390X_VIRTIO_CCW_MD_H ++ ++#include "virtio-ccw.h" ++#include "qom/object.h" ++ ++/* ++ * virtio-md-ccw: This extends VirtioCcwDevice. ++ */ ++#define TYPE_VIRTIO_MD_CCW "virtio-md-ccw" ++ ++OBJECT_DECLARE_TYPE(VirtIOMDCcw, VirtIOMDCcwClass, VIRTIO_MD_CCW) ++ ++struct VirtIOMDCcwClass { ++ /* private */ ++ VirtIOCCWDeviceClass parent; ++ ++ /* public */ ++ void (*unplug_request_check)(VirtIOMDCcw *vmd, Error **errp); ++}; ++ ++struct VirtIOMDCcw { ++ VirtioCcwDevice parent_obj; ++}; ++ ++void virtio_ccw_md_pre_plug(VirtIOMDCcw *vmd, MachineState *ms, Error **errp); ++void virtio_ccw_md_plug(VirtIOMDCcw *vmd, MachineState *ms, Error **errp); ++void virtio_ccw_md_unplug_request(VirtIOMDCcw *vmd, MachineState *ms, ++ Error **errp); ++void virtio_ccw_md_unplug(VirtIOMDCcw *vmd, MachineState *ms, Error **errp); ++ ++#endif /* HW_S390X_VIRTIO_CCW_MD_H */ +diff --git a/hw/virtio/Kconfig b/hw/virtio/Kconfig +index 0afec2ae92..f4b14e1a44 100644 +--- a/hw/virtio/Kconfig ++++ b/hw/virtio/Kconfig +@@ -25,6 +25,7 @@ config VIRTIO_MMIO + config VIRTIO_CCW + bool + select VIRTIO ++ select VIRTIO_MD_SUPPORTED + + config VIRTIO_BALLOON + bool +-- +2.48.1 + diff --git a/SOURCES/kvm-s390x-virtio-mem-support.patch b/SOURCES/kvm-s390x-virtio-mem-support.patch new file mode 100644 index 0000000..3c7313f --- /dev/null +++ b/SOURCES/kvm-s390x-virtio-mem-support.patch @@ -0,0 +1,459 @@ +From fa68427f55bee8d18d846e03ebf9f1eeb80f274d Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Thu, 19 Dec 2024 15:41:15 +0100 +Subject: [PATCH 23/26] s390x: virtio-mem support + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [23/26] 4c59ba9025ce5ba7686a7f3e01bb70e8c580709f (thuth/qemu-kvm-cs) + +Let's add our virtio-mem-ccw proxy device and wire it up. We should +be supporting everything (e.g., device unplug, "dynamic-memslots") that +we already support for the virtio-pci variant. + +With a Linux guest that supports virtio-mem (and has automatic memory +onlining properly configured) the following example will work: + +1. Start a VM with 4G initial memory and a virtio-mem device with a maximum + capacity of 16GB: + + qemu/build/qemu-system-s390x \ + --enable-kvm \ + -m 4G,maxmem=20G \ + -nographic \ + -smp 8 \ + -hda Fedora-Server-KVM-40-1.14.s390x.qcow2 \ + -chardev socket,id=monitor,path=/var/tmp/monitor,server,nowait \ + -mon chardev=monitor,mode=readline \ + -object memory-backend-ram,id=mem0,size=16G,reserve=off \ + -device virtio-mem-ccw,id=vmem0,memdev=mem0,dynamic-memslots=on + +2. Query the current size of virtio-mem device: + + (qemu) info memory-devices + Memory device [virtio-mem]: "vmem0" + memaddr: 0x100000000 + node: 0 + requested-size: 0 + size: 0 + max-size: 17179869184 + block-size: 1048576 + memdev: /objects/mem0 + +3. Request to grow it to 8GB (hotplug 8GB): + + (qemu) qom-set vmem0 requested-size 8G + (qemu) info memory-devices + Memory device [virtio-mem]: "vmem0" + memaddr: 0x100000000 + node: 0 + requested-size: 8589934592 + size: 8589934592 + max-size: 17179869184 + block-size: 1048576 + memdev: /objects/mem0 + +4. Request to grow to 16GB (hotplug another 8GB): + + (qemu) qom-set vmem0 requested-size 16G + (qemu) info memory-devices + Memory device [virtio-mem]: "vmem0" + memaddr: 0x100000000 + node: 0 + requested-size: 17179869184 + size: 17179869184 + max-size: 17179869184 + block-size: 1048576 + memdev: /objects/mem0 + +5. Try to hotunplug all memory again, shrinking to 0GB: + + (qemu) qom-set vmem0 requested-size 0G + (qemu) info memory-devices + Memory device [virtio-mem]: "vmem0" + memaddr: 0x100000000 + node: 0 + requested-size: 0 + size: 0 + max-size: 17179869184 + block-size: 1048576 + memdev: /objects/mem0 + +6. If it worked, unplug the device + + (qemu) device_del vmem0 + (qemu) info memory-devices + (qemu) object_del mem0 + +7. Hotplug a new device with a smaller capacity and directly size it to 1GB + + (qemu) object_add memory-backend-ram,id=mem0,size=8G,reserve=off + (qemu) device_add virtio-mem-ccw,id=vmem0,memdev=mem0,\ + dynamic-memslots=on,requested-size=1G + (qemu) info memory-devices + Memory device [virtio-mem]: "vmem0" + memaddr: 0x100000000 + node: 0 + requested-size: 1073741824 + size: 1073741824 + max-size: 8589934592 + block-size: 1048576 + memdev: /objects/mem0 + +Trying to use a virtio-mem device backed by hugetlb into a !hugetlb VM +correctly results in the error: + ... Memory device uses a bigger page size than initial memory + +Note that the virtio-mem driver in Linux will supports 1 MiB (pageblock) +granularity. + +Message-ID: <20241219144115.2820241-15-david@redhat.com> +Acked-by: Michael S. Tsirkin +Signed-off-by: David Hildenbrand +(cherry picked from commit aa910c20ec5f3b10551da19e441b3e2b54406e25) +Signed-off-by: Thomas Huth +--- + MAINTAINERS | 2 + + hw/s390x/Kconfig | 1 + + hw/s390x/meson.build | 1 + + hw/s390x/virtio-ccw-mem.c | 226 ++++++++++++++++++++++++++++++++++++++ + hw/s390x/virtio-ccw-mem.h | 34 ++++++ + hw/virtio/virtio-mem.c | 4 +- + 6 files changed, 267 insertions(+), 1 deletion(-) + create mode 100644 hw/s390x/virtio-ccw-mem.c + create mode 100644 hw/s390x/virtio-ccw-mem.h + +diff --git a/MAINTAINERS b/MAINTAINERS +index f21dc3fa75..f7b7ceffc4 100644 +--- a/MAINTAINERS ++++ b/MAINTAINERS +@@ -2401,6 +2401,8 @@ W: https://virtio-mem.gitlab.io/ + F: hw/virtio/virtio-mem.c + F: hw/virtio/virtio-mem-pci.h + F: hw/virtio/virtio-mem-pci.c ++F: hw/s390x/virtio-ccw-mem.c ++F: hw/s390x/virtio-ccw-mem.h + F: include/hw/virtio/virtio-mem.h + + virtio-snd +diff --git a/hw/s390x/Kconfig b/hw/s390x/Kconfig +index 3bbf4ae56e..5d57daff77 100644 +--- a/hw/s390x/Kconfig ++++ b/hw/s390x/Kconfig +@@ -15,3 +15,4 @@ config S390_CCW_VIRTIO + select SCLPCONSOLE + select VIRTIO_CCW + select MSI_NONBROKEN ++ select VIRTIO_MEM_SUPPORTED +diff --git a/hw/s390x/meson.build b/hw/s390x/meson.build +index 4431868408..3bbebfd817 100644 +--- a/hw/s390x/meson.build ++++ b/hw/s390x/meson.build +@@ -51,6 +51,7 @@ virtio_ss.add(when: 'CONFIG_VHOST_SCSI', if_true: files('vhost-scsi-ccw.c')) + virtio_ss.add(when: 'CONFIG_VHOST_VSOCK', if_true: files('vhost-vsock-ccw.c')) + virtio_ss.add(when: 'CONFIG_VHOST_USER_FS', if_true: files('vhost-user-fs-ccw.c')) + virtio_ss.add(when: 'CONFIG_VIRTIO_MD', if_true: files('virtio-ccw-md.c')) ++virtio_ss.add(when: 'CONFIG_VIRTIO_MEM', if_true: files('virtio-ccw-mem.c')) + s390x_ss.add_all(when: 'CONFIG_VIRTIO_CCW', if_true: virtio_ss) + + s390x_ss.add(when: 'CONFIG_VIRTIO_MD', if_false: files('virtio-ccw-md-stubs.c')) +diff --git a/hw/s390x/virtio-ccw-mem.c b/hw/s390x/virtio-ccw-mem.c +new file mode 100644 +index 0000000000..bee0d560cb +--- /dev/null ++++ b/hw/s390x/virtio-ccw-mem.c +@@ -0,0 +1,226 @@ ++/* ++ * virtio-mem CCW implementation ++ * ++ * Copyright (C) 2024 Red Hat, Inc. ++ * ++ * Authors: ++ * David Hildenbrand ++ * ++ * This work is licensed under the terms of the GNU GPL, version 2. ++ * See the COPYING file in the top-level directory. ++ */ ++ ++#include "qemu/osdep.h" ++#include "hw/qdev-properties.h" ++#include "qapi/error.h" ++#include "qemu/module.h" ++#include "virtio-ccw-mem.h" ++#include "hw/mem/memory-device.h" ++#include "qapi/qapi-events-machine.h" ++#include "qapi/qapi-events-misc.h" ++ ++static void virtio_ccw_mem_realize(VirtioCcwDevice *ccw_dev, Error **errp) ++{ ++ VirtIOMEMCcw *dev = VIRTIO_MEM_CCW(ccw_dev); ++ DeviceState *vdev = DEVICE(&dev->vdev); ++ ++ qdev_realize(vdev, BUS(&ccw_dev->bus), errp); ++} ++ ++static void virtio_ccw_mem_set_addr(MemoryDeviceState *md, uint64_t addr, ++ Error **errp) ++{ ++ object_property_set_uint(OBJECT(md), VIRTIO_MEM_ADDR_PROP, addr, errp); ++} ++ ++static uint64_t virtio_ccw_mem_get_addr(const MemoryDeviceState *md) ++{ ++ return object_property_get_uint(OBJECT(md), VIRTIO_MEM_ADDR_PROP, ++ &error_abort); ++} ++ ++static MemoryRegion *virtio_ccw_mem_get_memory_region(MemoryDeviceState *md, ++ Error **errp) ++{ ++ VirtIOMEMCcw *dev = VIRTIO_MEM_CCW(md); ++ VirtIOMEM *vmem = &dev->vdev; ++ VirtIOMEMClass *vmc = VIRTIO_MEM_GET_CLASS(vmem); ++ ++ return vmc->get_memory_region(vmem, errp); ++} ++ ++static void virtio_ccw_mem_decide_memslots(MemoryDeviceState *md, ++ unsigned int limit) ++{ ++ VirtIOMEMCcw *dev = VIRTIO_MEM_CCW(md); ++ VirtIOMEM *vmem = VIRTIO_MEM(&dev->vdev); ++ VirtIOMEMClass *vmc = VIRTIO_MEM_GET_CLASS(vmem); ++ ++ vmc->decide_memslots(vmem, limit); ++} ++ ++static unsigned int virtio_ccw_mem_get_memslots(MemoryDeviceState *md) ++{ ++ VirtIOMEMCcw *dev = VIRTIO_MEM_CCW(md); ++ VirtIOMEM *vmem = VIRTIO_MEM(&dev->vdev); ++ VirtIOMEMClass *vmc = VIRTIO_MEM_GET_CLASS(vmem); ++ ++ return vmc->get_memslots(vmem); ++} ++ ++static uint64_t virtio_ccw_mem_get_plugged_size(const MemoryDeviceState *md, ++ Error **errp) ++{ ++ return object_property_get_uint(OBJECT(md), VIRTIO_MEM_SIZE_PROP, ++ errp); ++} ++ ++static void virtio_ccw_mem_fill_device_info(const MemoryDeviceState *md, ++ MemoryDeviceInfo *info) ++{ ++ VirtioMEMDeviceInfo *vi = g_new0(VirtioMEMDeviceInfo, 1); ++ VirtIOMEMCcw *dev = VIRTIO_MEM_CCW(md); ++ VirtIOMEM *vmem = &dev->vdev; ++ VirtIOMEMClass *vpc = VIRTIO_MEM_GET_CLASS(vmem); ++ DeviceState *vdev = DEVICE(md); ++ ++ if (vdev->id) { ++ vi->id = g_strdup(vdev->id); ++ } ++ ++ /* let the real device handle everything else */ ++ vpc->fill_device_info(vmem, vi); ++ ++ info->u.virtio_mem.data = vi; ++ info->type = MEMORY_DEVICE_INFO_KIND_VIRTIO_MEM; ++} ++ ++static uint64_t virtio_ccw_mem_get_min_alignment(const MemoryDeviceState *md) ++{ ++ return object_property_get_uint(OBJECT(md), VIRTIO_MEM_BLOCK_SIZE_PROP, ++ &error_abort); ++} ++ ++static void virtio_ccw_mem_size_change_notify(Notifier *notifier, void *data) ++{ ++ VirtIOMEMCcw *dev = container_of(notifier, VirtIOMEMCcw, ++ size_change_notifier); ++ DeviceState *vdev = DEVICE(dev); ++ char *qom_path = object_get_canonical_path(OBJECT(dev)); ++ const uint64_t * const size_p = data; ++ ++ qapi_event_send_memory_device_size_change(vdev->id, *size_p, qom_path); ++ g_free(qom_path); ++} ++ ++static void virtio_ccw_mem_unplug_request_check(VirtIOMDCcw *vmd, Error **errp) ++{ ++ VirtIOMEMCcw *dev = VIRTIO_MEM_CCW(vmd); ++ VirtIOMEM *vmem = &dev->vdev; ++ VirtIOMEMClass *vpc = VIRTIO_MEM_GET_CLASS(vmem); ++ ++ vpc->unplug_request_check(vmem, errp); ++} ++ ++static void virtio_ccw_mem_get_requested_size(Object *obj, Visitor *v, ++ const char *name, void *opaque, ++ Error **errp) ++{ ++ VirtIOMEMCcw *dev = VIRTIO_MEM_CCW(obj); ++ ++ object_property_get(OBJECT(&dev->vdev), name, v, errp); ++} ++ ++static void virtio_ccw_mem_set_requested_size(Object *obj, Visitor *v, ++ const char *name, void *opaque, ++ Error **errp) ++{ ++ VirtIOMEMCcw *dev = VIRTIO_MEM_CCW(obj); ++ DeviceState *vdev = DEVICE(obj); ++ ++ /* ++ * If we passed virtio_ccw_mem_unplug_request_check(), making sure that ++ * the requested size is 0, don't allow modifying the requested size ++ * anymore, otherwise the VM might end up hotplugging memory before ++ * handling the unplug request. ++ */ ++ if (vdev->pending_deleted_event) { ++ error_setg(errp, "'%s' cannot be changed if the device is in the" ++ " process of unplug", name); ++ return; ++ } ++ ++ object_property_set(OBJECT(&dev->vdev), name, v, errp); ++} ++ ++static Property virtio_ccw_mem_properties[] = { ++ DEFINE_PROP_BIT("ioeventfd", VirtioCcwDevice, flags, ++ VIRTIO_CCW_FLAG_USE_IOEVENTFD_BIT, true), ++ DEFINE_PROP_UINT32("max_revision", VirtioCcwDevice, max_rev, ++ VIRTIO_CCW_MAX_REV), ++ DEFINE_PROP_END_OF_LIST(), ++}; ++ ++static void virtio_ccw_mem_class_init(ObjectClass *klass, void *data) ++{ ++ DeviceClass *dc = DEVICE_CLASS(klass); ++ VirtIOCCWDeviceClass *k = VIRTIO_CCW_DEVICE_CLASS(klass); ++ MemoryDeviceClass *mdc = MEMORY_DEVICE_CLASS(klass); ++ VirtIOMDCcwClass *vmdc = VIRTIO_MD_CCW_CLASS(klass); ++ ++ k->realize = virtio_ccw_mem_realize; ++ set_bit(DEVICE_CATEGORY_MISC, dc->categories); ++ device_class_set_props(dc, virtio_ccw_mem_properties); ++ ++ mdc->get_addr = virtio_ccw_mem_get_addr; ++ mdc->set_addr = virtio_ccw_mem_set_addr; ++ mdc->get_plugged_size = virtio_ccw_mem_get_plugged_size; ++ mdc->get_memory_region = virtio_ccw_mem_get_memory_region; ++ mdc->decide_memslots = virtio_ccw_mem_decide_memslots; ++ mdc->get_memslots = virtio_ccw_mem_get_memslots; ++ mdc->fill_device_info = virtio_ccw_mem_fill_device_info; ++ mdc->get_min_alignment = virtio_ccw_mem_get_min_alignment; ++ ++ vmdc->unplug_request_check = virtio_ccw_mem_unplug_request_check; ++} ++ ++static void virtio_ccw_mem_instance_init(Object *obj) ++{ ++ VirtIOMEMCcw *dev = VIRTIO_MEM_CCW(obj); ++ VirtIOMEMClass *vmc; ++ VirtIOMEM *vmem; ++ ++ virtio_instance_init_common(obj, &dev->vdev, sizeof(dev->vdev), ++ TYPE_VIRTIO_MEM); ++ ++ dev->size_change_notifier.notify = virtio_ccw_mem_size_change_notify; ++ vmem = &dev->vdev; ++ vmc = VIRTIO_MEM_GET_CLASS(vmem); ++ /* ++ * We never remove the notifier again, as we expect both devices to ++ * disappear at the same time. ++ */ ++ vmc->add_size_change_notifier(vmem, &dev->size_change_notifier); ++ ++ object_property_add_alias(obj, VIRTIO_MEM_BLOCK_SIZE_PROP, ++ OBJECT(&dev->vdev), VIRTIO_MEM_BLOCK_SIZE_PROP); ++ object_property_add_alias(obj, VIRTIO_MEM_SIZE_PROP, OBJECT(&dev->vdev), ++ VIRTIO_MEM_SIZE_PROP); ++ object_property_add(obj, VIRTIO_MEM_REQUESTED_SIZE_PROP, "size", ++ virtio_ccw_mem_get_requested_size, ++ virtio_ccw_mem_set_requested_size, NULL, NULL); ++} ++ ++static const TypeInfo virtio_ccw_mem = { ++ .name = TYPE_VIRTIO_MEM_CCW, ++ .parent = TYPE_VIRTIO_MD_CCW, ++ .instance_size = sizeof(VirtIOMEMCcw), ++ .instance_init = virtio_ccw_mem_instance_init, ++ .class_init = virtio_ccw_mem_class_init, ++}; ++ ++static void virtio_ccw_mem_register_types(void) ++{ ++ type_register_static(&virtio_ccw_mem); ++} ++type_init(virtio_ccw_mem_register_types) +diff --git a/hw/s390x/virtio-ccw-mem.h b/hw/s390x/virtio-ccw-mem.h +new file mode 100644 +index 0000000000..738ab2c744 +--- /dev/null ++++ b/hw/s390x/virtio-ccw-mem.h +@@ -0,0 +1,34 @@ ++/* ++ * Virtio MEM CCW device ++ * ++ * Copyright (C) 2024 Red Hat, Inc. ++ * ++ * Authors: ++ * David Hildenbrand ++ * ++ * This work is licensed under the terms of the GNU GPL, version 2. ++ * See the COPYING file in the top-level directory. ++ */ ++ ++#ifndef HW_S390X_VIRTIO_CCW_MEM_H ++#define HW_S390X_VIRTIO_CCW_MEM_H ++ ++#include "virtio-ccw-md.h" ++#include "hw/virtio/virtio-mem.h" ++#include "qom/object.h" ++ ++typedef struct VirtIOMEMCcw VirtIOMEMCcw; ++ ++/* ++ * virtio-mem-ccw: This extends VirtIOMDCcw ++ */ ++#define TYPE_VIRTIO_MEM_CCW "virtio-mem-ccw" ++DECLARE_INSTANCE_CHECKER(VirtIOMEMCcw, VIRTIO_MEM_CCW, TYPE_VIRTIO_MEM_CCW) ++ ++struct VirtIOMEMCcw { ++ VirtIOMDCcw parent_obj; ++ VirtIOMEM vdev; ++ Notifier size_change_notifier; ++}; ++ ++#endif /* HW_S390X_VIRTIO_CCW_MEM_H */ +diff --git a/hw/virtio/virtio-mem.c b/hw/virtio/virtio-mem.c +index 00da98b6e1..c9f8a23bbc 100644 +--- a/hw/virtio/virtio-mem.c ++++ b/hw/virtio/virtio-mem.c +@@ -61,6 +61,8 @@ static uint32_t virtio_mem_default_thp_size(void) + } else if (qemu_real_host_page_size() == 64 * KiB) { + default_thp_size = 512 * MiB; + } ++#elif defined(__s390x__) ++ default_thp_size = 1 * MiB; + #endif + + return default_thp_size; +@@ -161,7 +163,7 @@ static bool virtio_mem_has_shared_zeropage(RAMBlock *rb) + * necessary (as the section size can change). But it's more likely that the + * section size will rather get smaller and not bigger over time. + */ +-#if defined(TARGET_X86_64) || defined(TARGET_I386) ++#if defined(TARGET_X86_64) || defined(TARGET_I386) || defined(TARGET_S390X) + #define VIRTIO_MEM_USABLE_EXTENT (2 * (128 * MiB)) + #elif defined(TARGET_ARM) + #define VIRTIO_MEM_USABLE_EXTENT (2 * (512 * MiB)) +-- +2.48.1 + diff --git a/SOURCES/kvm-scripts-improve-error-from-qemu-trace-stap-on-missin.patch b/SOURCES/kvm-scripts-improve-error-from-qemu-trace-stap-on-missin.patch new file mode 100644 index 0000000..a6c8257 --- /dev/null +++ b/SOURCES/kvm-scripts-improve-error-from-qemu-trace-stap-on-missin.patch @@ -0,0 +1,90 @@ +From 314804fa4be6d653a7809b64076d4f3133a0ff59 Mon Sep 17 00:00:00 2001 +From: =?UTF-8?q?Daniel=20P=2E=20Berrang=C3=A9?= +Date: Fri, 6 Dec 2024 11:45:24 +0000 +Subject: [PATCH 8/9] scripts: improve error from qemu-trace-stap on missing + 'stap' +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Daniel P. Berrangé +RH-MergeRequest: 345: scripts: improve error from qemu-trace-stap on missing 'stap' +RH-Jira: RHEL-47340 +RH-Acked-by: Gerd Hoffmann +RH-Acked-by: Stefan Hajnoczi +RH-Commit: [1/2] c90635123f40e683488d83b59c71a5236c6d4659 (berrange/centos-src-qemu) + +If the 'stap' binary is missing in $PATH, a huge trace is thrown + + $ qemu-trace-stap list /usr/bin/qemu-system-x86_64 + Traceback (most recent call last): + File "/usr/bin/qemu-trace-stap", line 169, in + main() + File "/usr/bin/qemu-trace-stap", line 165, in main + args.func(args) + File "/usr/bin/qemu-trace-stap", line 83, in cmd_run + subprocess.call(stapargs) + File "/usr/lib64/python3.12/subprocess.py", line 389, in call + with Popen(*popenargs, **kwargs) as p: + ^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/lib64/python3.12/subprocess.py", line 1026, in {}init{} + self._execute_child(args, executable, preexec_fn, close_fds, + File "/usr/lib64/python3.12/subprocess.py", line 1955, in _execute_child + raise child_exception_type(errno_num, err_msg, err_filename) + FileNotFoundError: [Errno 2] No such file or directory: 'stap' + +With this change the user now gets + + $ qemu-trace-stap list /usr/bin/qemu-system-x86_64 + Unable to find 'stap' in $PATH + +Signed-off-by: Daniel P. Berrangé +Reviewed-by: Philippe Mathieu-Daudé +Message-id: 20241206114524.1666664-1-berrange@redhat.com +Signed-off-by: Stefan Hajnoczi +(cherry picked from commit 9976be3911a2d0503f026ae37c17077273bf30ee) +--- + scripts/qemu-trace-stap | 6 ++++-- + 1 file changed, 4 insertions(+), 2 deletions(-) + +diff --git a/scripts/qemu-trace-stap b/scripts/qemu-trace-stap +index eb6e951ff2..e983460ee7 100755 +--- a/scripts/qemu-trace-stap ++++ b/scripts/qemu-trace-stap +@@ -56,6 +56,7 @@ def tapset_dir(binary): + + + def cmd_run(args): ++ stap = which("stap") + prefix = probe_prefix(args.binary) + tapsets = tapset_dir(args.binary) + +@@ -76,7 +77,7 @@ def cmd_run(args): + + # We request an 8MB buffer, since the stap default 1MB buffer + # can be easily overflowed by frequently firing QEMU traces +- stapargs = ["stap", "-s", "8", "-I", tapsets ] ++ stapargs = [stap, "-s", "8", "-I", tapsets ] + if args.pid is not None: + stapargs.extend(["-x", args.pid]) + stapargs.extend(["-e", script]) +@@ -84,6 +85,7 @@ def cmd_run(args): + + + def cmd_list(args): ++ stap = which("stap") + tapsets = tapset_dir(args.binary) + + if args.verbose: +@@ -96,7 +98,7 @@ def cmd_list(args): + + if verbose: + print("Listing probes with name '%s'" % script) +- proc = subprocess.Popen(["stap", "-I", tapsets, "-l", script], ++ proc = subprocess.Popen([stap, "-I", tapsets, "-l", script], + stdout=subprocess.PIPE, + universal_newlines=True) + out, err = proc.communicate() +-- +2.48.1 + diff --git a/SOURCES/kvm-target-i386-Add-PerfMonV2-feature-bit.patch b/SOURCES/kvm-target-i386-Add-PerfMonV2-feature-bit.patch new file mode 100644 index 0000000..217987d --- /dev/null +++ b/SOURCES/kvm-target-i386-Add-PerfMonV2-feature-bit.patch @@ -0,0 +1,105 @@ +From 1587da0703e72cca8325a20b709280b8df85d066 Mon Sep 17 00:00:00 2001 +From: Sandipan Das +Date: Thu, 24 Oct 2024 17:18:21 -0500 +Subject: [PATCH 16/57] target/i386: Add PerfMonV2 feature bit + +RH-Author: John Allen +RH-MergeRequest: 378: Update EPYC Models and Feature Bits +RH-Jira: RHEL-52649 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [2/8] ec365cf4ac558c6c83f7a957e8df937cb6fbfa27 (johnalle/qemu-kvm-fork) + +CPUID leaf 0x80000022, i.e. ExtPerfMonAndDbg, advertises new performance +monitoring features for AMD processors. Bit 0 of EAX indicates support +for Performance Monitoring Version 2 (PerfMonV2) features. If found to +be set during PMU initialization, the EBX bits can be used to determine +the number of available counters for different PMUs. It also denotes the +availability of global control and status registers. + +Add the required CPUID feature word and feature bit to allow guests to +make use of the PerfMonV2 features. + +Signed-off-by: Sandipan Das +Signed-off-by: Babu Moger +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/a96f00ee2637674c63c61e9fc4dee343ea818053.1729807947.git.babu.moger@amd.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 209b0ac12074341d0093985eb9ad3e7edb252ce5) + +JIRA: https://issues.redhat.com/browse/RHEL-52649 + +Signed-off-by: John Allen +--- + target/i386/cpu.c | 26 ++++++++++++++++++++++++++ + target/i386/cpu.h | 4 ++++ + 2 files changed, 30 insertions(+) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 53069a460c..4546369836 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -1246,6 +1246,22 @@ FeatureWordInfo feature_word_info[FEATURE_WORDS] = { + .tcg_features = 0, + .unmigratable_flags = 0, + }, ++ [FEAT_8000_0022_EAX] = { ++ .type = CPUID_FEATURE_WORD, ++ .feat_names = { ++ "perfmon-v2", NULL, NULL, NULL, ++ NULL, NULL, NULL, NULL, ++ NULL, NULL, NULL, NULL, ++ NULL, NULL, NULL, NULL, ++ NULL, NULL, NULL, NULL, ++ NULL, NULL, NULL, NULL, ++ NULL, NULL, NULL, NULL, ++ NULL, NULL, NULL, NULL, ++ }, ++ .cpuid = { .eax = 0x80000022, .reg = R_EAX, }, ++ .tcg_features = 0, ++ .unmigratable_flags = 0, ++ }, + [FEAT_XSAVE] = { + .type = CPUID_FEATURE_WORD, + .feat_names = { +@@ -7096,6 +7112,16 @@ void cpu_x86_cpuid(CPUX86State *env, uint32_t index, uint32_t count, + *edx = 0; + } + break; ++ case 0x80000022: ++ *eax = *ebx = *ecx = *edx = 0; ++ /* AMD Extended Performance Monitoring and Debug */ ++ if (kvm_enabled() && cpu->enable_pmu && ++ (env->features[FEAT_8000_0022_EAX] & CPUID_8000_0022_EAX_PERFMON_V2)) { ++ *eax |= CPUID_8000_0022_EAX_PERFMON_V2; ++ *ebx |= kvm_arch_get_supported_cpuid(cs->kvm_state, index, count, ++ R_EBX) & 0xf; ++ } ++ break; + case 0xC0000000: + *eax = env->cpuid_xlevel2; + *ebx = 0; +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index 9a16239b8e..cf92a4972c 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -638,6 +638,7 @@ typedef enum FeatureWord { + FEAT_8000_0007_EDX, /* CPUID[8000_0007].EDX */ + FEAT_8000_0008_EBX, /* CPUID[8000_0008].EBX */ + FEAT_8000_0021_EAX, /* CPUID[8000_0021].EAX */ ++ FEAT_8000_0022_EAX, /* CPUID[8000_0022].EAX */ + FEAT_C000_0001_EDX, /* CPUID[C000_0001].EDX */ + FEAT_KVM, /* CPUID[4000_0001].EAX (KVM_CPUID_FEATURES) */ + FEAT_KVM_HINTS, /* CPUID[4000_0001].EDX */ +@@ -1044,6 +1045,9 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w); + /* Not vulnerable to SRSO at the user-kernel boundary */ + #define CPUID_8000_0021_EAX_SRSO_USER_KERNEL_NO (1U << 30) + ++/* Performance Monitoring Version 2 */ ++#define CPUID_8000_0022_EAX_PERFMON_V2 (1U << 0) ++ + #define CPUID_XSAVE_XSAVEOPT (1U << 0) + #define CPUID_XSAVE_XSAVEC (1U << 1) + #define CPUID_XSAVE_XGETBV1 (1U << 2) +-- +2.39.3 + diff --git a/SOURCES/kvm-target-i386-Add-couple-of-feature-bits-in-CPUID_Fn80.patch b/SOURCES/kvm-target-i386-Add-couple-of-feature-bits-in-CPUID_Fn80.patch new file mode 100644 index 0000000..6214232 --- /dev/null +++ b/SOURCES/kvm-target-i386-Add-couple-of-feature-bits-in-CPUID_Fn80.patch @@ -0,0 +1,83 @@ +From 79ac76edecdbbe253ad42385730aac18cdc40bd7 Mon Sep 17 00:00:00 2001 +From: Babu Moger +Date: Fri, 20 Jun 2025 14:54:53 -0500 +Subject: [PATCH 20/57] target/i386: Add couple of feature bits in + CPUID_Fn80000021_EAX + +RH-Author: John Allen +RH-MergeRequest: 378: Update EPYC Models and Feature Bits +RH-Jira: RHEL-52649 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [6/8] c6507eb24fcef271fdd6a234d2c255ef38c4e691 (johnalle/qemu-kvm-fork) + +Add CPUID bit indicates that a WRMSR to MSR_FS_BASE, MSR_GS_BASE, or +MSR_KERNEL_GS_BASE is non-serializing amd PREFETCHI that the +cates +support for IC prefetch. + +CPUID_Fn80000021_EAX +Bit Feature description +20 Indicates support for IC prefetch. +1 FsGsKernelGsBaseNonSerializing. + WRMSR to FS_BASE, GS_BASE and KernelGSbase are +serializing. + +Link: https://www.amd.com/content/dam/amd/en/documents/epyc-technical-docs/programmer-references/57238.zip +Signed-off-by: Babu Moger +Reviewed-by: Maksim Davydov +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/a5f6283a59579b09ac345b3f21ecb3b3b2d92451.1746734284.git.babu.moger@amd.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit dfd5b456108a75588ab094358ba5754787146d3d) + +JIRA: https://issues.redhat.com/browse/RHEL-52649 + +Signed-off-by: John Allen +--- + target/i386/cpu.c | 4 ++-- + target/i386/cpu.h | 4 ++++ + 2 files changed, 6 insertions(+), 2 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 7d48c51767..2218071fca 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -1233,12 +1233,12 @@ FeatureWordInfo feature_word_info[FEATURE_WORDS] = { + [FEAT_8000_0021_EAX] = { + .type = CPUID_FEATURE_WORD, + .feat_names = { +- "no-nested-data-bp", NULL, "lfence-always-serializing", NULL, ++ "no-nested-data-bp", "fs-gs-base-ns", "lfence-always-serializing", NULL, + NULL, NULL, "null-sel-clr-base", NULL, + "auto-ibrs", NULL, NULL, NULL, + NULL, NULL, NULL, NULL, + NULL, NULL, NULL, NULL, +- NULL, NULL, NULL, NULL, ++ "prefetchi", NULL, NULL, NULL, + NULL, NULL, NULL, NULL, + "ibpb-brtype", "srso-no", "srso-user-kernel-no", NULL, + }, +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index cf92a4972c..e513e5f62d 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -1030,12 +1030,16 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w); + + /* Processor ignores nested data breakpoints */ + #define CPUID_8000_0021_EAX_NO_NESTED_DATA_BP (1U << 0) ++/* WRMSR to FS_BASE, GS_BASE, or KERNEL_GS_BASE is non-serializing */ ++#define CPUID_8000_0021_EAX_FS_GS_BASE_NS (1U << 1) + /* LFENCE is always serializing */ + #define CPUID_8000_0021_EAX_LFENCE_ALWAYS_SERIALIZING (1U << 2) + /* Null Selector Clears Base */ + #define CPUID_8000_0021_EAX_NULL_SEL_CLR_BASE (1U << 6) + /* Automatic IBRS */ + #define CPUID_8000_0021_EAX_AUTO_IBRS (1U << 8) ++/* Indicates support for IC prefetch */ ++#define CPUID_8000_0021_EAX_PREFETCHI (1U << 20) + /* Selective Branch Predictor Barrier */ + #define CPUID_8000_0021_EAX_SBPB (1U << 27) + /* IBPB includes branch type prediction flushing */ +-- +2.39.3 + diff --git a/SOURCES/kvm-target-i386-Add-support-for-EPYC-Turin-model.patch b/SOURCES/kvm-target-i386-Add-support-for-EPYC-Turin-model.patch new file mode 100644 index 0000000..6293b4c --- /dev/null +++ b/SOURCES/kvm-target-i386-Add-support-for-EPYC-Turin-model.patch @@ -0,0 +1,200 @@ +From e0b59a57883faac254cd75cc243fed784ad4975b Mon Sep 17 00:00:00 2001 +From: Babu Moger +Date: Thu, 8 May 2025 14:58:04 -0500 +Subject: [PATCH 22/57] target/i386: Add support for EPYC-Turin model + +RH-Author: John Allen +RH-MergeRequest: 378: Update EPYC Models and Feature Bits +RH-Jira: RHEL-52649 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [8/8] 42e90c7fc6bf858f98ae3a6be35d70b824a7d6bf (johnalle/qemu-kvm-fork) + +Add the support for AMD EPYC zen 5 processors (EPYC-Turin). + +Add the following new feature bits on top of the feature bits from +the previous generation EPYC models. + +movdiri : Move Doubleword as Direct Store Instruction +movdir64b : Move 64 Bytes as Direct Store Instruction +avx512-vp2intersect : AVX512 Vector Pair Intersection to a Pair + of Mask Register +avx-vnni : AVX VNNI Instruction +prefetchi : Indicates support for IC prefetch +sbpb : Selective Branch Predictor Barrier +ibpb-brtype : IBPB includes branch type prediction flushing +srso-user-kernel-no : Not vulnerable to SRSO at the user-kernel boundary + +Link: https://www.amd.com/content/dam/amd/en/documents/epyc-technical-docs/programmer-references/57238.zip +Link: https://www.amd.com/content/dam/amd/en/documents/corporate/cr/speculative-return-stack-overflow-whitepaper.pdf +Signed-off-by: Babu Moger +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/b4fa7708a0e1453d2e9b8ec3dc881feb92eeca0b.1746734284.git.babu.moger@amd.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 3771a4daa273ba17cb27309984413790d1df5651) + +JIRA: https://issues.redhat.com/browse/RHEL-52649 + +Signed-off-by: John Allen +--- + target/i386/cpu.c | 138 ++++++++++++++++++++++++++++++++++++++++++++++ + 1 file changed, 138 insertions(+) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 2bc2d41259..fdfa183f4d 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -2651,6 +2651,61 @@ static const CPUCaches epyc_genoa_v2_cache_info = { + .share_level = CPU_TOPO_LEVEL_DIE, + }, + }; ++ ++static const CPUCaches epyc_turin_cache_info = { ++ .l1d_cache = &(CPUCacheInfo) { ++ .type = DATA_CACHE, ++ .level = 1, ++ .size = 48 * KiB, ++ .line_size = 64, ++ .associativity = 12, ++ .partitions = 1, ++ .sets = 64, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l1i_cache = &(CPUCacheInfo) { ++ .type = INSTRUCTION_CACHE, ++ .level = 1, ++ .size = 32 * KiB, ++ .line_size = 64, ++ .associativity = 8, ++ .partitions = 1, ++ .sets = 64, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l2_cache = &(CPUCacheInfo) { ++ .type = UNIFIED_CACHE, ++ .level = 2, ++ .size = 1 * MiB, ++ .line_size = 64, ++ .associativity = 16, ++ .partitions = 1, ++ .sets = 1024, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .inclusive = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l3_cache = &(CPUCacheInfo) { ++ .type = UNIFIED_CACHE, ++ .level = 3, ++ .size = 32 * MiB, ++ .line_size = 64, ++ .associativity = 16, ++ .partitions = 1, ++ .sets = 32768, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .no_invd_sharing = true, ++ .complex_indexing = false, ++ .share_level = CPU_TOPO_LEVEL_DIE, ++ }, ++}; ++ + /* The following VMX features are not supported by KVM and are left out in the + * CPU definitions: + * +@@ -5644,6 +5699,89 @@ static const X86CPUDefinition builtin_x86_defs[] = { + { /* end of list */ } + } + }, ++ { ++ .name = "EPYC-Turin", ++ .level = 0xd, ++ .vendor = CPUID_VENDOR_AMD, ++ .family = 26, ++ .model = 0, ++ .stepping = 0, ++ .features[FEAT_1_ECX] = ++ CPUID_EXT_RDRAND | CPUID_EXT_F16C | CPUID_EXT_AVX | ++ CPUID_EXT_XSAVE | CPUID_EXT_AES | CPUID_EXT_POPCNT | ++ CPUID_EXT_MOVBE | CPUID_EXT_SSE42 | CPUID_EXT_SSE41 | ++ CPUID_EXT_PCID | CPUID_EXT_CX16 | CPUID_EXT_FMA | ++ CPUID_EXT_SSSE3 | CPUID_EXT_MONITOR | CPUID_EXT_PCLMULQDQ | ++ CPUID_EXT_SSE3, ++ .features[FEAT_1_EDX] = ++ CPUID_SSE2 | CPUID_SSE | CPUID_FXSR | CPUID_MMX | CPUID_CLFLUSH | ++ CPUID_PSE36 | CPUID_PAT | CPUID_CMOV | CPUID_MCA | CPUID_PGE | ++ CPUID_MTRR | CPUID_SEP | CPUID_APIC | CPUID_CX8 | CPUID_MCE | ++ CPUID_PAE | CPUID_MSR | CPUID_TSC | CPUID_PSE | CPUID_DE | ++ CPUID_VME | CPUID_FP87, ++ .features[FEAT_6_EAX] = ++ CPUID_6_EAX_ARAT, ++ .features[FEAT_7_0_EBX] = ++ CPUID_7_0_EBX_FSGSBASE | CPUID_7_0_EBX_BMI1 | CPUID_7_0_EBX_AVX2 | ++ CPUID_7_0_EBX_SMEP | CPUID_7_0_EBX_BMI2 | CPUID_7_0_EBX_ERMS | ++ CPUID_7_0_EBX_INVPCID | CPUID_7_0_EBX_AVX512F | ++ CPUID_7_0_EBX_AVX512DQ | CPUID_7_0_EBX_RDSEED | CPUID_7_0_EBX_ADX | ++ CPUID_7_0_EBX_SMAP | CPUID_7_0_EBX_AVX512IFMA | ++ CPUID_7_0_EBX_CLFLUSHOPT | CPUID_7_0_EBX_CLWB | ++ CPUID_7_0_EBX_AVX512CD | CPUID_7_0_EBX_SHA_NI | ++ CPUID_7_0_EBX_AVX512BW | CPUID_7_0_EBX_AVX512VL, ++ .features[FEAT_7_0_ECX] = ++ CPUID_7_0_ECX_AVX512_VBMI | CPUID_7_0_ECX_UMIP | CPUID_7_0_ECX_PKU | ++ CPUID_7_0_ECX_AVX512_VBMI2 | CPUID_7_0_ECX_GFNI | ++ CPUID_7_0_ECX_VAES | CPUID_7_0_ECX_VPCLMULQDQ | ++ CPUID_7_0_ECX_AVX512VNNI | CPUID_7_0_ECX_AVX512BITALG | ++ CPUID_7_0_ECX_AVX512_VPOPCNTDQ | CPUID_7_0_ECX_LA57 | ++ CPUID_7_0_ECX_RDPID | CPUID_7_0_ECX_MOVDIRI | ++ CPUID_7_0_ECX_MOVDIR64B, ++ .features[FEAT_7_0_EDX] = ++ CPUID_7_0_EDX_FSRM | CPUID_7_0_EDX_AVX512_VP2INTERSECT, ++ .features[FEAT_7_1_EAX] = ++ CPUID_7_1_EAX_AVX_VNNI | CPUID_7_1_EAX_AVX512_BF16, ++ .features[FEAT_8000_0001_ECX] = ++ CPUID_EXT3_OSVW | CPUID_EXT3_3DNOWPREFETCH | ++ CPUID_EXT3_MISALIGNSSE | CPUID_EXT3_SSE4A | CPUID_EXT3_ABM | ++ CPUID_EXT3_CR8LEG | CPUID_EXT3_SVM | CPUID_EXT3_LAHF_LM | ++ CPUID_EXT3_TOPOEXT | CPUID_EXT3_PERFCORE, ++ .features[FEAT_8000_0001_EDX] = ++ CPUID_EXT2_LM | CPUID_EXT2_RDTSCP | CPUID_EXT2_PDPE1GB | ++ CPUID_EXT2_FFXSR | CPUID_EXT2_MMXEXT | CPUID_EXT2_NX | ++ CPUID_EXT2_SYSCALL, ++ .features[FEAT_8000_0007_EBX] = ++ CPUID_8000_0007_EBX_OVERFLOW_RECOV | CPUID_8000_0007_EBX_SUCCOR, ++ .features[FEAT_8000_0008_EBX] = ++ CPUID_8000_0008_EBX_CLZERO | CPUID_8000_0008_EBX_XSAVEERPTR | ++ CPUID_8000_0008_EBX_WBNOINVD | CPUID_8000_0008_EBX_IBPB | ++ CPUID_8000_0008_EBX_IBRS | CPUID_8000_0008_EBX_STIBP | ++ CPUID_8000_0008_EBX_STIBP_ALWAYS_ON | ++ CPUID_8000_0008_EBX_AMD_SSBD | CPUID_8000_0008_EBX_AMD_PSFD, ++ .features[FEAT_8000_0021_EAX] = ++ CPUID_8000_0021_EAX_NO_NESTED_DATA_BP | ++ CPUID_8000_0021_EAX_FS_GS_BASE_NS | ++ CPUID_8000_0021_EAX_LFENCE_ALWAYS_SERIALIZING | ++ CPUID_8000_0021_EAX_NULL_SEL_CLR_BASE | ++ CPUID_8000_0021_EAX_AUTO_IBRS | CPUID_8000_0021_EAX_PREFETCHI | ++ CPUID_8000_0021_EAX_SBPB | CPUID_8000_0021_EAX_IBPB_BRTYPE | ++ CPUID_8000_0021_EAX_SRSO_USER_KERNEL_NO, ++ .features[FEAT_8000_0022_EAX] = ++ CPUID_8000_0022_EAX_PERFMON_V2, ++ .features[FEAT_XSAVE] = ++ CPUID_XSAVE_XSAVEOPT | CPUID_XSAVE_XSAVEC | ++ CPUID_XSAVE_XGETBV1 | CPUID_XSAVE_XSAVES, ++ .features[FEAT_SVM] = ++ CPUID_SVM_NPT | CPUID_SVM_LBRV | CPUID_SVM_NRIPSAVE | ++ CPUID_SVM_TSCSCALE | CPUID_SVM_VMCBCLEAN | CPUID_SVM_FLUSHASID | ++ CPUID_SVM_PAUSEFILTER | CPUID_SVM_PFTHRESHOLD | ++ CPUID_SVM_V_VMSAVE_VMLOAD | CPUID_SVM_VGIF | ++ CPUID_SVM_VNMI | CPUID_SVM_SVME_ADDR_CHK, ++ .xlevel = 0x80000022, ++ .model_id = "AMD EPYC-Turin Processor", ++ .cache_info = &epyc_turin_cache_info, ++ }, + }; + + /* +-- +2.39.3 + diff --git a/SOURCES/kvm-target-i386-Enable-fdp-excptn-only-and-zero-fcs-fds.patch b/SOURCES/kvm-target-i386-Enable-fdp-excptn-only-and-zero-fcs-fds.patch new file mode 100644 index 0000000..1fac08e --- /dev/null +++ b/SOURCES/kvm-target-i386-Enable-fdp-excptn-only-and-zero-fcs-fds.patch @@ -0,0 +1,75 @@ +From d6c70ced910aede72d11b6a698bfda9648a3c959 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:43 +0200 +Subject: [PATCH 002/115] target/i386: Enable fdp-excptn-only and zero-fcs-fds + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [2/115] 58b986eab9ccc86b3f50ca480ae184779cd85bee (bonzini/rhel-qemu-kvm) + +- CPUID.(EAX=07H,ECX=0H):EBX[bit 6]: x87 FPU Data Pointer updated only + on x87 exceptions if 1. + +- CPUID.(EAX=07H,ECX=0H):EBX[bit 13]: Deprecates FPU CS and FPU DS + values if 1. i.e., X87 FCS and FDS are always zero. + +Define names for them so that they can be exposed to guest with -cpu host. + +Also define the bit field MACROs so that named cpu models can add it as +well in the future. + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20240814075431.339209-3-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 7dddc3bb875e7141ab25931d0f30a1c319bc8457) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 4 ++-- + target/i386/cpu.h | 4 ++++ + 2 files changed, 6 insertions(+), 2 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 0ac6cd8ad7..1fe492f33d 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -1058,9 +1058,9 @@ FeatureWordInfo feature_word_info[FEATURE_WORDS] = { + .type = CPUID_FEATURE_WORD, + .feat_names = { + "fsgsbase", "tsc-adjust", "sgx", "bmi1", +- "hle", "avx2", NULL, "smep", ++ "hle", "avx2", "fdp-excptn-only", "smep", + "bmi2", "erms", "invpcid", "rtm", +- NULL, NULL, "mpx", NULL, ++ NULL, "zero-fcs-fds", "mpx", NULL, + "avx512f", "avx512dq", "rdseed", "adx", + "smap", "avx512ifma", "pcommit", "clflushopt", + "clwb", "intel-pt", "avx512pf", "avx512er", +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index e513e5f62d..5924761551 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -828,6 +828,8 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w); + #define CPUID_7_0_EBX_HLE (1U << 4) + /* Intel Advanced Vector Extensions 2 */ + #define CPUID_7_0_EBX_AVX2 (1U << 5) ++/* FPU data pointer updated only on x87 exceptions */ ++#define CPUID_7_0_EBX_FDP_EXCPTN_ONLY (1u << 6) + /* Supervisor-mode Execution Prevention */ + #define CPUID_7_0_EBX_SMEP (1U << 7) + /* 2nd Group of Advanced Bit Manipulation Extensions */ +@@ -838,6 +840,8 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w); + #define CPUID_7_0_EBX_INVPCID (1U << 10) + /* Restricted Transactional Memory */ + #define CPUID_7_0_EBX_RTM (1U << 11) ++/* Zero out FPU CS and FPU DS */ ++#define CPUID_7_0_EBX_ZERO_FCS_FDS (1U << 13) + /* Memory Protection Extension */ + #define CPUID_7_0_EBX_MPX (1U << 14) + /* AVX-512 Foundation */ +-- +2.50.1 + diff --git a/SOURCES/kvm-target-i386-Exclude-hv-syndbg-from-hv-passthrough.patch b/SOURCES/kvm-target-i386-Exclude-hv-syndbg-from-hv-passthrough.patch new file mode 100644 index 0000000..df4e5e3 --- /dev/null +++ b/SOURCES/kvm-target-i386-Exclude-hv-syndbg-from-hv-passthrough.patch @@ -0,0 +1,102 @@ +From 0288537593cd4452a2523b686b297dad3735f7f8 Mon Sep 17 00:00:00 2001 +From: Vitaly Kuznetsov +Date: Thu, 17 Apr 2025 15:30:50 +0200 +Subject: [PATCH 2/2] target/i386: Exclude 'hv-syndbg' from 'hv-passthrough' + +RH-Author: Vitaly Kuznetsov +RH-MergeRequest: 352: hyper-v: exclude 'hv-syndbg' from 'hv-passthrough' set +RH-Jira: RHEL-7130 +RH-Acked-by: Maxim Levitsky +RH-Acked-by: Ani Sinha +RH-Acked-by: Emanuele Giuseppe Esposito +RH-Commit: [2/2] bf276ad5b340139f71b92e656a0c7756a55dec0b (vkuznets/qemu-kvm) + +Windows with Hyper-V role enabled doesn't boot with 'hv-passthrough' when +no debugger is configured, this significantly limits the usefulness of the +feature as there's no support for subtracting Hyper-V features from CPU +flags at this moment (e.g. "-cpu host,hv-passthrough,-hv-syndbg" does not +work). While this is also theoretically fixable, 'hv-syndbg' is likely +very special and unneeded in the default set. Genuine Hyper-V doesn't seem +to enable it either. + +Introduce 'skip_passthrough' flag to 'kvm_hyperv_properties' and use it as +one-off to skip 'hv-syndbg' when enabling features in 'hv-passthrough' +mode. Note, "-cpu host,hv-passthrough,hv-syndbg" can still be used if +needed. + +As both 'hv-passthrough' and 'hv-syndbg' are debug features, the change +should not have any effect on production environments. + +Signed-off-by: Vitaly Kuznetsov +Link: https://lore.kernel.org/r/20240917160051.2637594-3-vkuznets@redhat.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 7d7b9c7655a26e09c800ef40373078a80e90d9f3) +Signed-off-by: Vitaly Kuznetsov +--- + docs/system/i386/hyperv.rst | 13 +++++++++---- + target/i386/kvm/kvm.c | 7 +++++-- + 2 files changed, 14 insertions(+), 6 deletions(-) + +diff --git a/docs/system/i386/hyperv.rst b/docs/system/i386/hyperv.rst +index 2505dc4c86..009947e391 100644 +--- a/docs/system/i386/hyperv.rst ++++ b/docs/system/i386/hyperv.rst +@@ -262,14 +262,19 @@ Supplementary features + ``hv-passthrough`` + In some cases (e.g. during development) it may make sense to use QEMU in + 'pass-through' mode and give Windows guests all enlightenments currently +- supported by KVM. This pass-through mode is enabled by "hv-passthrough" CPU +- flag. ++ supported by KVM. + + Note: ``hv-passthrough`` flag only enables enlightenments which are known to QEMU + (have corresponding 'hv-' flag) and copies ``hv-spinlocks`` and ``hv-vendor-id`` + values from KVM to QEMU. ``hv-passthrough`` overrides all other 'hv-' settings on +- the command line. Also, enabling this flag effectively prevents migration as the +- list of enabled enlightenments may differ between target and destination hosts. ++ the command line. ++ ++ Note: ``hv-passthrough`` does not enable ``hv-syndbg`` which can prevent certain ++ Windows guests from booting when used without proper configuration. If needed, ++ ``hv-syndbg`` can be enabled additionally. ++ ++ Note: ``hv-passthrough`` effectively prevents migration as the list of enabled ++ enlightenments may differ between target and destination hosts. + + ``hv-enforce-cpuid`` + By default, KVM allows the guest to use all currently supported Hyper-V +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index 5bf77d761f..94b678e9e3 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -913,6 +913,7 @@ static struct { + uint32_t bits; + } flags[2]; + uint64_t dependencies; ++ bool skip_passthrough; + } kvm_hyperv_properties[] = { + [HYPERV_FEAT_RELAXED] = { + .desc = "relaxed timing (hv-relaxed)", +@@ -1041,7 +1042,8 @@ static struct { + {.func = HV_CPUID_FEATURES, .reg = R_EDX, + .bits = HV_FEATURE_DEBUG_MSRS_AVAILABLE} + }, +- .dependencies = BIT(HYPERV_FEAT_SYNIC) | BIT(HYPERV_FEAT_RELAXED) ++ .dependencies = BIT(HYPERV_FEAT_SYNIC) | BIT(HYPERV_FEAT_RELAXED), ++ .skip_passthrough = true, + }, + [HYPERV_FEAT_MSR_BITMAP] = { + .desc = "enlightened MSR-Bitmap (hv-emsr-bitmap)", +@@ -1450,7 +1452,8 @@ bool kvm_hyperv_expand_features(X86CPU *cpu, Error **errp) + * hv_build_cpuid_leaf() uses this info to build guest CPUIDs. + */ + for (feat = 0; feat < ARRAY_SIZE(kvm_hyperv_properties); feat++) { +- if (hyperv_feature_supported(cs, feat)) { ++ if (hyperv_feature_supported(cs, feat) && ++ !kvm_hyperv_properties[feat].skip_passthrough) { + cpu->hyperv_features |= BIT(feat); + } + } +-- +2.48.1 + diff --git a/SOURCES/kvm-target-i386-Expose-IBPB-BRTYPE-and-SBPB-CPUID-bits-t.patch b/SOURCES/kvm-target-i386-Expose-IBPB-BRTYPE-and-SBPB-CPUID-bits-t.patch new file mode 100644 index 0000000..0b889e4 --- /dev/null +++ b/SOURCES/kvm-target-i386-Expose-IBPB-BRTYPE-and-SBPB-CPUID-bits-t.patch @@ -0,0 +1,70 @@ +From dd03cf49fbf6a961a726506cb5264768d814d2c4 Mon Sep 17 00:00:00 2001 +From: Igor Mammedov +Date: Mon, 5 Aug 2024 17:20:41 -0300 +Subject: [PATCH] target/i386: Expose IBPB-BRTYPE and SBPB CPUID bits to the + guest + +RH-Author: Igor Mammedov +RH-MergeRequest: 401: target/i386: Expose IBPB-BRTYPE and SBPB CPUID bits to the guest +RH-Jira: RHEL-17614 +RH-Acked-by: Ani Sinha +RH-Acked-by: Jon Maloy +RH-Commit: [1/1] aa904a1ea0552fc37b61f79fda8a471928ea5d81 (imammedo/qemu-kvm-cs) + +According to AMD's Speculative Return Stack Overflow whitepaper (link +below), the hypervisor should synthesize the value of IBPB_BRTYPE and +SBPB CPUID bits to the guest. + +Support for this is already present in the kernel with commit +e47d86083c66 ("KVM: x86: Add SBPB support") and commit 6f0f23ef76be +("KVM: x86: Add IBPB_BRTYPE support"). + +Add support in QEMU to expose the bits to the guest OS. + +host: + # cat /sys/devices/system/cpu/vulnerabilities/spec_rstack_overflow + Mitigation: Safe RET + +before (guest): + $ cpuid -l 0x80000021 -1 -r + 0x80000021 0x00: eax=0x00000045 ebx=0x00000000 ecx=0x00000000 edx=0x00000000 + ^ + $ cat /sys/devices/system/cpu/vulnerabilities/spec_rstack_overflow + Vulnerable: Safe RET, no microcode + +after (guest): + $ cpuid -l 0x80000021 -1 -r + 0x80000021 0x00: eax=0x18000045 ebx=0x00000000 ecx=0x00000000 edx=0x00000000 + ^ + $ cat /sys/devices/system/cpu/vulnerabilities/spec_rstack_overflow + Mitigation: Safe RET + +Reported-by: Fabian Vogt +Link: https://www.amd.com/content/dam/amd/en/documents/corporate/cr/speculative-return-stack-overflow-whitepaper.pdf +Signed-off-by: Fabiano Rosas +Link: https://lore.kernel.org/r/20240805202041.5936-1-farosas@suse.de +Signed-off-by: Paolo Bonzini + +(cherry picked from commit 0701abbf9880b5ab1cf44e0caa6ad173aec840e7) +JIRA: https://issues.redhat.com/browse/RHEL-17614 +Signed-off-by: Igor Mammedov +--- + target/i386/cpu.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index ee753351fc..f75cc04cd3 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -1241,7 +1241,7 @@ FeatureWordInfo feature_word_info[FEATURE_WORDS] = { + NULL, NULL, NULL, NULL, + NULL, NULL, NULL, NULL, + "prefetchi", NULL, NULL, NULL, +- NULL, NULL, NULL, NULL, ++ NULL, NULL, NULL, "sbpb", + "ibpb-brtype", "srso-no", "srso-user-kernel-no", NULL, + }, + .cpuid = { .eax = 0x80000021, .reg = R_EAX, }, +-- +2.50.1 + diff --git a/SOURCES/kvm-target-i386-Expose-bits-related-to-SRSO-vulnerabilit.patch b/SOURCES/kvm-target-i386-Expose-bits-related-to-SRSO-vulnerabilit.patch new file mode 100644 index 0000000..7c666cf --- /dev/null +++ b/SOURCES/kvm-target-i386-Expose-bits-related-to-SRSO-vulnerabilit.patch @@ -0,0 +1,84 @@ +From 1d667a354613385b1552fdbae91799882776f908 Mon Sep 17 00:00:00 2001 +From: Babu Moger +Date: Thu, 24 Oct 2024 17:18:23 -0500 +Subject: [PATCH 15/57] target/i386: Expose bits related to SRSO vulnerability + +RH-Author: John Allen +RH-MergeRequest: 378: Update EPYC Models and Feature Bits +RH-Jira: RHEL-52649 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [1/8] 9a6f4126ab023269e8afb3537aaa94ae60228382 (johnalle/qemu-kvm-fork) + +Add following bits related Speculative Return Stack Overflow (SRSO). +Guests can make use of these bits if supported. + +These bits are reported via CPUID Fn8000_0021_EAX. +=================================================================== +Bit Feature Description +=================================================================== +27 SBPB Indicates support for the Selective Branch Predictor Barrier. +28 IBPB_BRTYPE MSR_PRED_CMD[IBPB] flushes all branch type predictions. +29 SRSO_NO Not vulnerable to SRSO. +30 SRSO_USER_KERNEL_NO Not vulnerable to SRSO at the user-kernel boundary. +=================================================================== + +Link: https://www.amd.com/content/dam/amd/en/documents/corporate/cr/speculative-return-stack-overflow-whitepaper.pdf +Link: https://www.amd.com/content/dam/amd/en/documents/epyc-technical-docs/programmer-references/57238.zip +Signed-off-by: Babu Moger +Link: https://lore.kernel.org/r/dadbd70c38f4e165418d193918a3747bd715c5f4.1729807947.git.babu.moger@amd.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 2ec282b8eaaddf5c136f7566b5f61d80288a2065) + +JIRA: https://issues.redhat.com/browse/RHEL-52649 + +Signed-off-by: John Allen +--- + target/i386/cpu.c | 2 +- + target/i386/cpu.h | 14 +++++++++++--- + 2 files changed, 12 insertions(+), 4 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 0a955b1c45..53069a460c 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -1240,7 +1240,7 @@ FeatureWordInfo feature_word_info[FEATURE_WORDS] = { + NULL, NULL, NULL, NULL, + NULL, NULL, NULL, NULL, + NULL, NULL, NULL, NULL, +- NULL, NULL, NULL, NULL, ++ "ibpb-brtype", "srso-no", "srso-user-kernel-no", NULL, + }, + .cpuid = { .eax = 0x80000021, .reg = R_EAX, }, + .tcg_features = 0, +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index 4da9ed5930..9a16239b8e 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -1028,13 +1028,21 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w); + #define CPUID_8000_0008_EBX_AMD_PSFD (1U << 28) + + /* Processor ignores nested data breakpoints */ +-#define CPUID_8000_0021_EAX_No_NESTED_DATA_BP (1U << 0) ++#define CPUID_8000_0021_EAX_NO_NESTED_DATA_BP (1U << 0) + /* LFENCE is always serializing */ + #define CPUID_8000_0021_EAX_LFENCE_ALWAYS_SERIALIZING (1U << 2) + /* Null Selector Clears Base */ +-#define CPUID_8000_0021_EAX_NULL_SEL_CLR_BASE (1U << 6) ++#define CPUID_8000_0021_EAX_NULL_SEL_CLR_BASE (1U << 6) + /* Automatic IBRS */ +-#define CPUID_8000_0021_EAX_AUTO_IBRS (1U << 8) ++#define CPUID_8000_0021_EAX_AUTO_IBRS (1U << 8) ++/* Selective Branch Predictor Barrier */ ++#define CPUID_8000_0021_EAX_SBPB (1U << 27) ++/* IBPB includes branch type prediction flushing */ ++#define CPUID_8000_0021_EAX_IBPB_BRTYPE (1U << 28) ++/* Not vulnerable to Speculative Return Stack Overflow */ ++#define CPUID_8000_0021_EAX_SRSO_NO (1U << 29) ++/* Not vulnerable to SRSO at the user-kernel boundary */ ++#define CPUID_8000_0021_EAX_SRSO_USER_KERNEL_NO (1U << 30) + + #define CPUID_XSAVE_XSAVEOPT (1U << 0) + #define CPUID_XSAVE_XSAVEC (1U << 1) +-- +2.39.3 + diff --git a/SOURCES/kvm-target-i386-Fix-conditional-CONFIG_SYNDBG-enablement.patch b/SOURCES/kvm-target-i386-Fix-conditional-CONFIG_SYNDBG-enablement.patch new file mode 100644 index 0000000..049f3fe --- /dev/null +++ b/SOURCES/kvm-target-i386-Fix-conditional-CONFIG_SYNDBG-enablement.patch @@ -0,0 +1,108 @@ +From 26d5561f7a07c9bc6f8ea9a602c53bfa5daddd13 Mon Sep 17 00:00:00 2001 +From: Vitaly Kuznetsov +Date: Thu, 17 Apr 2025 15:30:42 +0200 +Subject: [PATCH 1/2] target/i386: Fix conditional CONFIG_SYNDBG enablement + +RH-Author: Vitaly Kuznetsov +RH-MergeRequest: 352: hyper-v: exclude 'hv-syndbg' from 'hv-passthrough' set +RH-Jira: RHEL-7130 +RH-Acked-by: Maxim Levitsky +RH-Acked-by: Ani Sinha +RH-Acked-by: Emanuele Giuseppe Esposito +RH-Commit: [1/2] 0446b6202fb3dbae865da0dc7e08092399661f7a (vkuznets/qemu-kvm) + +Putting HYPERV_FEAT_SYNDBG entry under "#ifdef CONFIG_SYNDBG" in +'kvm_hyperv_properties' array is wrong: as HYPERV_FEAT_SYNDBG is not +the highest feature number, the result is an empty (zeroed) entry in +the array (and not a skipped entry!). hyperv_feature_supported() is +designed to check that all CPUID bits are set but for a zeroed +feature in 'kvm_hyperv_properties' it returns 'true' so QEMU considers +HYPERV_FEAT_SYNDBG as always supported, regardless of whether KVM host +actually supports it. + +To fix the issue, leave HYPERV_FEAT_SYNDBG's definition in +'kvm_hyperv_properties' array, there's nothing wrong in having it defined +even when 'CONFIG_SYNDBG' is not set. Instead, put "hv-syndbg" CPU property +under '#ifdef CONFIG_SYNDBG' to alter the existing behavior when the flag +is silently skipped in !CONFIG_SYNDBG builds. + +Leave an 'assert' sentinel in hyperv_feature_supported() making sure there +are no 'holes' or improperly defined features in 'kvm_hyperv_properties'. + +Fixes: d8701185f40c ("hw: hyperv: Initial commit for Synthetic Debugging device") +Signed-off-by: Vitaly Kuznetsov +Link: https://lore.kernel.org/r/20240917160051.2637594-2-vkuznets@redhat.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit bbf3810f2c4f97bd7a1982d3e0ff0f00295b8169) +Signed-off-by: Vitaly Kuznetsov +--- + target/i386/cpu.c | 2 ++ + target/i386/kvm/kvm.c | 11 +++++++---- + 2 files changed, 9 insertions(+), 4 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index a70a3aa670..0a955b1c45 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -8450,8 +8450,10 @@ static Property x86_cpu_properties[] = { + HYPERV_FEAT_TLBFLUSH_DIRECT, 0), + DEFINE_PROP_ON_OFF_AUTO("hv-no-nonarch-coresharing", X86CPU, + hyperv_no_nonarch_cs, ON_OFF_AUTO_OFF), ++#ifdef CONFIG_SYNDBG + DEFINE_PROP_BIT64("hv-syndbg", X86CPU, hyperv_features, + HYPERV_FEAT_SYNDBG, 0), ++#endif + DEFINE_PROP_BOOL("hv-passthrough", X86CPU, hyperv_passthrough, false), + DEFINE_PROP_BOOL("hv-enforce-cpuid", X86CPU, hyperv_enforce_cpuid, false), + +diff --git a/target/i386/kvm/kvm.c b/target/i386/kvm/kvm.c +index d0329a4ed7..5bf77d761f 100644 +--- a/target/i386/kvm/kvm.c ++++ b/target/i386/kvm/kvm.c +@@ -1035,7 +1035,6 @@ static struct { + .bits = HV_DEPRECATING_AEOI_RECOMMENDED} + } + }, +-#ifdef CONFIG_SYNDBG + [HYPERV_FEAT_SYNDBG] = { + .desc = "Enable synthetic kernel debugger channel (hv-syndbg)", + .flags = { +@@ -1044,7 +1043,6 @@ static struct { + }, + .dependencies = BIT(HYPERV_FEAT_SYNIC) | BIT(HYPERV_FEAT_RELAXED) + }, +-#endif + [HYPERV_FEAT_MSR_BITMAP] = { + .desc = "enlightened MSR-Bitmap (hv-emsr-bitmap)", + .flags = { +@@ -1296,6 +1294,13 @@ static bool hyperv_feature_supported(CPUState *cs, int feature) + uint32_t func, bits; + int i, reg; + ++ /* ++ * kvm_hyperv_properties needs to define at least one CPUID flag which ++ * must be used to detect the feature, it's hard to say whether it is ++ * supported or not otherwise. ++ */ ++ assert(kvm_hyperv_properties[feature].flags[0].func); ++ + for (i = 0; i < ARRAY_SIZE(kvm_hyperv_properties[feature].flags); i++) { + + func = kvm_hyperv_properties[feature].flags[i].func; +@@ -3925,13 +3930,11 @@ static int kvm_put_msrs(X86CPU *cpu, int level) + kvm_msr_entry_add(cpu, HV_X64_MSR_TSC_EMULATION_STATUS, + env->msr_hv_tsc_emulation_status); + } +-#ifdef CONFIG_SYNDBG + if (hyperv_feat_enabled(cpu, HYPERV_FEAT_SYNDBG) && + has_msr_hv_syndbg_options) { + kvm_msr_entry_add(cpu, HV_X64_MSR_SYNDBG_OPTIONS, + hyperv_syndbg_query_options()); + } +-#endif + } + if (hyperv_feat_enabled(cpu, HYPERV_FEAT_VAPIC)) { + kvm_msr_entry_add(cpu, HV_X64_MSR_APIC_ASSIST_PAGE, +-- +2.48.1 + diff --git a/SOURCES/kvm-target-i386-Make-invtsc-migratable-when-user-sets-ts.patch b/SOURCES/kvm-target-i386-Make-invtsc-migratable-when-user-sets-ts.patch new file mode 100644 index 0000000..d6a6c44 --- /dev/null +++ b/SOURCES/kvm-target-i386-Make-invtsc-migratable-when-user-sets-ts.patch @@ -0,0 +1,73 @@ +From 5c3ab9a195310186be83501c20c95033a2fff594 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:43 +0200 +Subject: [PATCH 001/115] target/i386: Make invtsc migratable when user sets + tsc-khz explicitly + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [1/115] 14556e7f141ab24aa977392eb30819df061b608a (bonzini/rhel-qemu-kvm) + +When user sets tsc-frequency explicitly, the invtsc feature is actually +migratable because the tsc-frequency is supposed to be fixed during the +migration. + +See commit d99569d9d856 ("kvm: Allow invtsc migration if tsc-khz +is set explicitly") for referrence. + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20240814075431.339209-10-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 87c88db3143e91076d167a62dd7febf49afca8a2) +Signed-off-by: Paolo Bonzini +(cherry picked from commit 3a7e5d481be248bf0b81f23fd84da43663597504) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 11 +++++++++-- + 1 file changed, 9 insertions(+), 2 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index fdfa183f4d..0ac6cd8ad7 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -1917,9 +1917,10 @@ static inline uint64_t x86_cpu_xsave_xss_components(X86CPU *cpu) + * Returns the set of feature flags that are supported and migratable by + * QEMU, for a given FeatureWord. + */ +-static uint64_t x86_cpu_get_migratable_flags(FeatureWord w) ++static uint64_t x86_cpu_get_migratable_flags(X86CPU *cpu, FeatureWord w) + { + FeatureWordInfo *wi = &feature_word_info[w]; ++ CPUX86State *env = &cpu->env; + uint64_t r = 0; + int i; + +@@ -1933,6 +1934,12 @@ static uint64_t x86_cpu_get_migratable_flags(FeatureWord w) + r |= f; + } + } ++ ++ /* when tsc-khz is set explicitly, invtsc is migratable */ ++ if ((w == FEAT_8000_0007_EDX) && env->user_tsc_khz) { ++ r |= CPUID_APM_INVTSC; ++ } ++ + return r; + } + +@@ -6650,7 +6657,7 @@ uint64_t x86_cpu_get_supported_feature_word(X86CPU *cpu, FeatureWord w) + + r &= ~unavail; + if (cpu && cpu->migratable) { +- r &= x86_cpu_get_migratable_flags(w); ++ r &= x86_cpu_get_migratable_flags(cpu, w); + } + return r; + } +-- +2.50.1 + diff --git a/SOURCES/kvm-target-i386-Print-CPUID-subleaf-info-for-unsupported.patch b/SOURCES/kvm-target-i386-Print-CPUID-subleaf-info-for-unsupported.patch new file mode 100644 index 0000000..56cf96d --- /dev/null +++ b/SOURCES/kvm-target-i386-Print-CPUID-subleaf-info-for-unsupported.patch @@ -0,0 +1,47 @@ +From 532fe3ba3d7c05387884bf3460894f9d2e0e8a91 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 018/115] target/i386: Print CPUID subleaf info for unsupported + feature + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [18/115] 7a9640191aa0a03348b8c0162ba1957d749cd454 (bonzini/rhel-qemu-kvm) + +Some CPUID leaves have meaningful subleaf index. Print the subleaf info +in feature_word_description for CPUID features. + +Signed-off-by: Xiaoyao Li +Reviewed-by: Eduardo Habkost +Reviewed-by: Zhao Liu +Message-ID: <20241217123932.948789-3-xiaoyao.li@intel.com> +Signed-off-by: Paolo Bonzini +(cherry picked from commit 22a2a701090b83d8dd06765edc1b61f6208c76b1) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 5 +++-- + 1 file changed, 3 insertions(+), 2 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 32e89f1a5c..816285facd 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -5906,8 +5906,9 @@ static char *feature_word_description(FeatureWordInfo *f) + { + const char *reg = get_register_name_32(f->cpuid.reg); + assert(reg); +- return g_strdup_printf("CPUID.%02XH:%s", +- f->cpuid.eax, reg); ++ return g_strdup_printf("CPUID.%02XH_%02XH:%s", ++ f->cpuid.eax, ++ f->cpuid.needs_ecx ? f->cpuid.ecx : 0, reg); + } + case MSR_FEATURE_WORD: + return g_strdup_printf("MSR(%02XH)", +-- +2.50.1 + diff --git a/SOURCES/kvm-target-i386-Remove-AccelCPUClass-cpu_class_init-need.patch b/SOURCES/kvm-target-i386-Remove-AccelCPUClass-cpu_class_init-need.patch new file mode 100644 index 0000000..027d184 --- /dev/null +++ b/SOURCES/kvm-target-i386-Remove-AccelCPUClass-cpu_class_init-need.patch @@ -0,0 +1,122 @@ +From 5be7728369a8f109a86b2d5aea89c6dc4014c559 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:45 +0200 +Subject: [PATCH 023/115] target/i386: Remove AccelCPUClass::cpu_class_init + need +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [23/115] a327127b0c6a96c8198420732006770480b18e69 (bonzini/rhel-qemu-kvm) + +Expose x86_tcg_ops symbol, then directly set it as +CPUClass::tcg_ops in TYPE_X86_CPU's class_init(), +using CONFIG_TCG #ifdef'ry. No need for the +AccelCPUClass::cpu_class_init() handler anymore. + +Signed-off-by: Philippe Mathieu-Daudé +Message-ID: <20250405161320.76854-3-philmd@linaro.org> +Reviewed-by: Richard Henderson +Signed-off-by: Richard Henderson +(cherry picked from commit a522b04bb9cf67789116ad7a6165946d4b214bac) +Signed-off-by: Paolo Bonzini + +Conflicts: missing one member of x86_tcg_ops +--- + target/i386/cpu.c | 4 ++++ + target/i386/tcg/tcg-cpu.c | 14 +------------- + target/i386/tcg/tcg-cpu.h | 4 ++++ + 3 files changed, 9 insertions(+), 13 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 816285facd..4eef3d1dbd 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -42,6 +42,7 @@ + #include "hw/boards.h" + #include "hw/i386/sgx-epc.h" + #endif ++#include "tcg/tcg-cpu.h" + + #include "disas/capstone.h" + #include "cpu-internal.h" +@@ -9034,6 +9035,9 @@ static void x86_cpu_common_class_init(ObjectClass *oc, void *data) + #ifndef CONFIG_USER_ONLY + cc->sysemu_ops = &i386_sysemu_ops; + #endif /* !CONFIG_USER_ONLY */ ++#ifdef CONFIG_TCG ++ cc->tcg_ops = &x86_tcg_ops; ++#endif /* CONFIG_TCG */ + + cc->gdb_arch_name = x86_gdb_arch_name; + #ifdef TARGET_X86_64 +diff --git a/target/i386/tcg/tcg-cpu.c b/target/i386/tcg/tcg-cpu.c +index cca19cd40e..0160f3f70d 100644 +--- a/target/i386/tcg/tcg-cpu.c ++++ b/target/i386/tcg/tcg-cpu.c +@@ -106,7 +106,7 @@ static bool x86_debug_check_breakpoint(CPUState *cs) + + #include "hw/core/tcg-cpu-ops.h" + +-static const TCGCPUOps x86_tcg_ops = { ++const TCGCPUOps x86_tcg_ops = { + .initialize = tcg_x86_init, + .synchronize_from_tb = x86_cpu_synchronize_from_tb, + .restore_state_to_opc = x86_restore_state_to_opc, +@@ -128,17 +128,6 @@ static const TCGCPUOps x86_tcg_ops = { + #endif /* !CONFIG_USER_ONLY */ + }; + +-static void x86_tcg_cpu_init_ops(AccelCPUClass *accel_cpu, CPUClass *cc) +-{ +- /* for x86, all cpus use the same set of operations */ +- cc->tcg_ops = &x86_tcg_ops; +-} +- +-static void x86_tcg_cpu_class_init(CPUClass *cc) +-{ +- cc->init_accel_cpu = x86_tcg_cpu_init_ops; +-} +- + static void x86_tcg_cpu_xsave_init(void) + { + #define XO(bit, field) \ +@@ -187,7 +176,6 @@ static void x86_tcg_cpu_accel_class_init(ObjectClass *oc, void *data) + acc->cpu_target_realize = tcg_cpu_realizefn; + #endif /* CONFIG_USER_ONLY */ + +- acc->cpu_class_init = x86_tcg_cpu_class_init; + acc->cpu_instance_init = x86_tcg_cpu_instance_init; + } + static const TypeInfo x86_tcg_cpu_accel_type_info = { +diff --git a/target/i386/tcg/tcg-cpu.h b/target/i386/tcg/tcg-cpu.h +index 53a8494455..9bbf0cb875 100644 +--- a/target/i386/tcg/tcg-cpu.h ++++ b/target/i386/tcg/tcg-cpu.h +@@ -19,6 +19,8 @@ + #ifndef TCG_CPU_H + #define TCG_CPU_H + ++#include "cpu.h" ++ + #define XSAVE_FCW_FSW_OFFSET 0x000 + #define XSAVE_FTW_FOP_OFFSET 0x004 + #define XSAVE_CWD_RIP_OFFSET 0x008 +@@ -76,6 +78,8 @@ QEMU_BUILD_BUG_ON(offsetof(X86XSaveArea, zmm_hi256_state) != XSAVE_ZMM_HI256_OFF + QEMU_BUILD_BUG_ON(offsetof(X86XSaveArea, hi16_zmm_state) != XSAVE_HI16_ZMM_OFFSET); + QEMU_BUILD_BUG_ON(offsetof(X86XSaveArea, pkru_state) != XSAVE_PKRU_OFFSET); + ++extern const TCGCPUOps x86_tcg_ops; ++ + bool tcg_cpu_realizefn(CPUState *cs, Error **errp); + + #endif /* TCG_CPU_H */ +-- +2.50.1 + diff --git a/SOURCES/kvm-target-i386-Update-EPYC-CPU-model-for-Cache-property.patch b/SOURCES/kvm-target-i386-Update-EPYC-CPU-model-for-Cache-property.patch new file mode 100644 index 0000000..f5cfccc --- /dev/null +++ b/SOURCES/kvm-target-i386-Update-EPYC-CPU-model-for-Cache-property.patch @@ -0,0 +1,147 @@ +From 4091d13096918dfff5f3a292b43e613ea888ddc1 Mon Sep 17 00:00:00 2001 +From: Babu Moger +Date: Thu, 8 May 2025 14:57:59 -0500 +Subject: [PATCH 17/57] target/i386: Update EPYC CPU model for Cache property, + RAS, SVM feature bits + +RH-Author: John Allen +RH-MergeRequest: 378: Update EPYC Models and Feature Bits +RH-Jira: RHEL-52649 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [3/8] afc52d066ad5f66732b0bb04142e210c52896708 (johnalle/qemu-kvm-fork) + +Found that some of the cache properties are not set correctly for EPYC models. + +l1d_cache.no_invd_sharing should not be true. +l1i_cache.no_invd_sharing should not be true. + +L2.self_init should be true. +L2.inclusive should be true. + +L3.inclusive should not be true. +L3.no_invd_sharing should be true. + +Fix the cache properties. + +Also add the missing RAS and SVM features bits on AMD +EPYC CPU models. The SVM feature bits are used in nested guests. + +succor : Software uncorrectable error containment and recovery capability. +overflow-recov : MCA overflow recovery support. +lbrv : LBR virtualization +tsc-scale : MSR based TSC rate control +vmcb-clean : VMCB clean bits +flushbyasid : Flush by ASID +pause-filter : Pause intercept filter +pfthreshold : PAUSE filter threshold +v-vmsave-vmload : Virtualized VMLOAD and VMSAVE +vgif : Virtualized GIF + +Signed-off-by: Babu Moger +Reviewed-by: Maksim Davydov +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/515941861700d7066186c9600bc5d96a1741ef0c.1746734284.git.babu.moger@amd.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 397db937e85d7b9f5a6f0b30764786cef09d1ff3) + +JIRA: https://issues.redhat.com/browse/RHEL-52649 + +Signed-off-by: John Allen +--- + target/i386/cpu.c | 73 +++++++++++++++++++++++++++++++++++++++++++++++ + 1 file changed, 73 insertions(+) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 4546369836..32c575f63b 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -2166,6 +2166,60 @@ static CPUCaches epyc_v4_cache_info = { + }, + }; + ++static CPUCaches epyc_v5_cache_info = { ++ .l1d_cache = &(CPUCacheInfo) { ++ .type = DATA_CACHE, ++ .level = 1, ++ .size = 32 * KiB, ++ .line_size = 64, ++ .associativity = 8, ++ .partitions = 1, ++ .sets = 64, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l1i_cache = &(CPUCacheInfo) { ++ .type = INSTRUCTION_CACHE, ++ .level = 1, ++ .size = 64 * KiB, ++ .line_size = 64, ++ .associativity = 4, ++ .partitions = 1, ++ .sets = 256, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l2_cache = &(CPUCacheInfo) { ++ .type = UNIFIED_CACHE, ++ .level = 2, ++ .size = 512 * KiB, ++ .line_size = 64, ++ .associativity = 8, ++ .partitions = 1, ++ .sets = 1024, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .inclusive = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l3_cache = &(CPUCacheInfo) { ++ .type = UNIFIED_CACHE, ++ .level = 3, ++ .size = 8 * MiB, ++ .line_size = 64, ++ .associativity = 16, ++ .partitions = 1, ++ .sets = 8192, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .no_invd_sharing = true, ++ .complex_indexing = false, ++ .share_level = CPU_TOPO_LEVEL_DIE, ++ }, ++}; ++ + static const CPUCaches epyc_rome_cache_info = { + .l1d_cache = &(CPUCacheInfo) { + .type = DATA_CACHE, +@@ -5059,6 +5113,25 @@ static const X86CPUDefinition builtin_x86_defs[] = { + }, + .cache_info = &epyc_v4_cache_info + }, ++ { ++ .version = 5, ++ .props = (PropValue[]) { ++ { "overflow-recov", "on" }, ++ { "succor", "on" }, ++ { "lbrv", "on" }, ++ { "tsc-scale", "on" }, ++ { "vmcb-clean", "on" }, ++ { "flushbyasid", "on" }, ++ { "pause-filter", "on" }, ++ { "pfthreshold", "on" }, ++ { "v-vmsave-vmload", "on" }, ++ { "vgif", "on" }, ++ { "model-id", ++ "AMD EPYC-v5 Processor" }, ++ { /* end of list */ } ++ }, ++ .cache_info = &epyc_v5_cache_info ++ }, + { /* end of list */ } + } + }, +-- +2.39.3 + diff --git a/SOURCES/kvm-target-i386-Update-EPYC-Genoa-for-Cache-property-per.patch b/SOURCES/kvm-target-i386-Update-EPYC-Genoa-for-Cache-property-per.patch new file mode 100644 index 0000000..59260d8 --- /dev/null +++ b/SOURCES/kvm-target-i386-Update-EPYC-Genoa-for-Cache-property-per.patch @@ -0,0 +1,167 @@ +From 768e39f40b394eb4524a83857b86e8f7497f4414 Mon Sep 17 00:00:00 2001 +From: Babu Moger +Date: Thu, 8 May 2025 14:58:03 -0500 +Subject: [PATCH 21/57] target/i386: Update EPYC-Genoa for Cache property, + perfmon-v2, RAS and SVM feature bits + +RH-Author: John Allen +RH-MergeRequest: 378: Update EPYC Models and Feature Bits +RH-Jira: RHEL-52649 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [7/8] b144233a1115385a1f792c4454f4511173f753d8 (johnalle/qemu-kvm-fork) + +Found that some of the cache properties are not set correctly for EPYC models. +l1d_cache.no_invd_sharing should not be true. +l1i_cache.no_invd_sharing should not be true. + +L2.self_init should be true. +L2.inclusive should be true. + +L3.inclusive should not be true. +L3.no_invd_sharing should be true. + +Fix these cache properties. + +Also add the missing RAS and SVM features bits on AMD EPYC-Genoa model. +The SVM feature bits are used in nested guests. + +perfmon-v2 : Allow guests to make use of the PerfMonV2 features. +succor : Software uncorrectable error containment and recovery capability. +overflow-recov : MCA overflow recovery support. +lbrv : LBR virtualization +tsc-scale : MSR based TSC rate control +vmcb-clean : VMCB clean bits +flushbyasid : Flush by ASID +pause-filter : Pause intercept filter +pfthreshold : PAUSE filter threshold +v-vmsave-vmload: Virtualized VMLOAD and VMSAVE +vgif : Virtualized GIF +fs-gs-base-ns : WRMSR to {FS,GS,KERNEL_GS}_BASE is non-serializing + +The feature details are available in APM listed below [1]. +[1] AMD64 Architecture Programmer's Manual Volume 2: System Programming +Publication # 24593 Revision 3.41. + +Link: https://bugzilla.kernel.org/show_bug.cgi?id=206537 +Signed-off-by: Babu Moger +Reviewed-by: Maksim Davydov +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/afe3f05d4116124fd5795f28fc23d7b396140313.1746734284.git.babu.moger@amd.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit abc92cc8488b5dbcc403b5be24d8092180605101) + +JIRA: https://issues.redhat.com/browse/RHEL-52649 + +Signed-off-by: John Allen +--- + target/i386/cpu.c | 80 ++++++++++++++++++++++++++++++++++++++++++++++- + 1 file changed, 79 insertions(+), 1 deletion(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 2218071fca..2bc2d41259 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -2598,6 +2598,59 @@ static const CPUCaches epyc_genoa_cache_info = { + }, + }; + ++static const CPUCaches epyc_genoa_v2_cache_info = { ++ .l1d_cache = &(CPUCacheInfo) { ++ .type = DATA_CACHE, ++ .level = 1, ++ .size = 32 * KiB, ++ .line_size = 64, ++ .associativity = 8, ++ .partitions = 1, ++ .sets = 64, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l1i_cache = &(CPUCacheInfo) { ++ .type = INSTRUCTION_CACHE, ++ .level = 1, ++ .size = 32 * KiB, ++ .line_size = 64, ++ .associativity = 8, ++ .partitions = 1, ++ .sets = 64, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l2_cache = &(CPUCacheInfo) { ++ .type = UNIFIED_CACHE, ++ .level = 2, ++ .size = 1 * MiB, ++ .line_size = 64, ++ .associativity = 8, ++ .partitions = 1, ++ .sets = 2048, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .inclusive = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l3_cache = &(CPUCacheInfo) { ++ .type = UNIFIED_CACHE, ++ .level = 3, ++ .size = 32 * MiB, ++ .line_size = 64, ++ .associativity = 16, ++ .partitions = 1, ++ .sets = 32768, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .no_invd_sharing = true, ++ .complex_indexing = false, ++ .share_level = CPU_TOPO_LEVEL_DIE, ++ }, ++}; + /* The following VMX features are not supported by KVM and are left out in the + * CPU definitions: + * +@@ -5530,7 +5583,7 @@ static const X86CPUDefinition builtin_x86_defs[] = { + CPUID_8000_0008_EBX_STIBP_ALWAYS_ON | + CPUID_8000_0008_EBX_AMD_SSBD | CPUID_8000_0008_EBX_AMD_PSFD, + .features[FEAT_8000_0021_EAX] = +- CPUID_8000_0021_EAX_No_NESTED_DATA_BP | ++ CPUID_8000_0021_EAX_NO_NESTED_DATA_BP | + CPUID_8000_0021_EAX_LFENCE_ALWAYS_SERIALIZING | + CPUID_8000_0021_EAX_NULL_SEL_CLR_BASE | + CPUID_8000_0021_EAX_AUTO_IBRS, +@@ -5565,6 +5618,31 @@ static const X86CPUDefinition builtin_x86_defs[] = { + .xlevel = 0x80000022, + .model_id = "AMD EPYC-Genoa Processor", + .cache_info = &epyc_genoa_cache_info, ++ .versions = (X86CPUVersionDefinition[]) { ++ { .version = 1 }, ++ { ++ .version = 2, ++ .props = (PropValue[]) { ++ { "overflow-recov", "on" }, ++ { "succor", "on" }, ++ { "lbrv", "on" }, ++ { "tsc-scale", "on" }, ++ { "vmcb-clean", "on" }, ++ { "flushbyasid", "on" }, ++ { "pause-filter", "on" }, ++ { "pfthreshold", "on" }, ++ { "v-vmsave-vmload", "on" }, ++ { "vgif", "on" }, ++ { "fs-gs-base-ns", "on" }, ++ { "perfmon-v2", "on" }, ++ { "model-id", ++ "AMD EPYC-Genoa-v2 Processor" }, ++ { /* end of list */ } ++ }, ++ .cache_info = &epyc_genoa_v2_cache_info ++ }, ++ { /* end of list */ } ++ } + }, + }; + +-- +2.39.3 + diff --git a/SOURCES/kvm-target-i386-Update-EPYC-Milan-CPU-model-for-Cache-pr.patch b/SOURCES/kvm-target-i386-Update-EPYC-Milan-CPU-model-for-Cache-pr.patch new file mode 100644 index 0000000..ba143ba --- /dev/null +++ b/SOURCES/kvm-target-i386-Update-EPYC-Milan-CPU-model-for-Cache-pr.patch @@ -0,0 +1,146 @@ +From a2cd6a5aac0ba2bbb50d2ff22b83c8b9d7761028 Mon Sep 17 00:00:00 2001 +From: Babu Moger +Date: Thu, 8 May 2025 14:58:01 -0500 +Subject: [PATCH 19/57] target/i386: Update EPYC-Milan CPU model for Cache + property, RAS, SVM feature bits + +RH-Author: John Allen +RH-MergeRequest: 378: Update EPYC Models and Feature Bits +RH-Jira: RHEL-52649 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [5/8] e9e34ade25cb7be05d40745e1d074c0356d1923f (johnalle/qemu-kvm-fork) + +Found that some of the cache properties are not set correctly for EPYC models. +l1d_cache.no_invd_sharing should not be true. +l1i_cache.no_invd_sharing should not be true. + +L2.self_init should be true. +L2.inclusive should be true. + +L3.inclusive should not be true. +L3.no_invd_sharing should be true. + +Fix these cache properties. + +Also add the missing RAS and SVM features bits on AMD EPYC-Milan model. +The SVM feature bits are used in nested guests. + +succor : Software uncorrectable error containment and recovery capability. +overflow-recov : MCA overflow recovery support. +lbrv : LBR virtualization +tsc-scale : MSR based TSC rate control +vmcb-clean : VMCB clean bits +flushbyasid : Flush by ASID +pause-filter : Pause intercept filter +pfthreshold : PAUSE filter threshold +v-vmsave-vmload : Virtualized VMLOAD and VMSAVE +vgif : Virtualized GIF + +Signed-off-by: Babu Moger +Reviewed-by: Maksim Davydov +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/c619c0e09a9d5d496819ed48d69181d65f416891.1746734284.git.babu.moger@amd.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit fc014d9ba5b26b27401e0e88a4e1ef827c68fe64) + +JIRA: https://issues.redhat.com/browse/RHEL-52649 + +Signed-off-by: John Allen +--- + target/i386/cpu.c | 73 +++++++++++++++++++++++++++++++++++++++++++++++ + 1 file changed, 73 insertions(+) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index a73b5bfca4..7d48c51767 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -2490,6 +2490,60 @@ static const CPUCaches epyc_milan_v2_cache_info = { + }, + }; + ++static const CPUCaches epyc_milan_v3_cache_info = { ++ .l1d_cache = &(CPUCacheInfo) { ++ .type = DATA_CACHE, ++ .level = 1, ++ .size = 32 * KiB, ++ .line_size = 64, ++ .associativity = 8, ++ .partitions = 1, ++ .sets = 64, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l1i_cache = &(CPUCacheInfo) { ++ .type = INSTRUCTION_CACHE, ++ .level = 1, ++ .size = 32 * KiB, ++ .line_size = 64, ++ .associativity = 8, ++ .partitions = 1, ++ .sets = 64, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l2_cache = &(CPUCacheInfo) { ++ .type = UNIFIED_CACHE, ++ .level = 2, ++ .size = 512 * KiB, ++ .line_size = 64, ++ .associativity = 8, ++ .partitions = 1, ++ .sets = 1024, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .inclusive = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l3_cache = &(CPUCacheInfo) { ++ .type = UNIFIED_CACHE, ++ .level = 3, ++ .size = 32 * MiB, ++ .line_size = 64, ++ .associativity = 16, ++ .partitions = 1, ++ .sets = 32768, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .no_invd_sharing = true, ++ .complex_indexing = false, ++ .share_level = CPU_TOPO_LEVEL_DIE, ++ }, ++}; ++ + static const CPUCaches epyc_genoa_cache_info = { + .l1d_cache = &(CPUCacheInfo) { + .type = DATA_CACHE, +@@ -5418,6 +5472,25 @@ static const X86CPUDefinition builtin_x86_defs[] = { + }, + .cache_info = &epyc_milan_v2_cache_info + }, ++ { ++ .version = 3, ++ .props = (PropValue[]) { ++ { "overflow-recov", "on" }, ++ { "succor", "on" }, ++ { "lbrv", "on" }, ++ { "tsc-scale", "on" }, ++ { "vmcb-clean", "on" }, ++ { "flushbyasid", "on" }, ++ { "pause-filter", "on" }, ++ { "pfthreshold", "on" }, ++ { "v-vmsave-vmload", "on" }, ++ { "vgif", "on" }, ++ { "model-id", ++ "AMD EPYC-Milan-v3 Processor" }, ++ { /* end of list */ } ++ }, ++ .cache_info = &epyc_milan_v3_cache_info ++ }, + { /* end of list */ } + } + }, +-- +2.39.3 + diff --git a/SOURCES/kvm-target-i386-Update-EPYC-Rome-CPU-model-for-Cache-pro.patch b/SOURCES/kvm-target-i386-Update-EPYC-Rome-CPU-model-for-Cache-pro.patch new file mode 100644 index 0000000..82c2cb4 --- /dev/null +++ b/SOURCES/kvm-target-i386-Update-EPYC-Rome-CPU-model-for-Cache-pro.patch @@ -0,0 +1,147 @@ +From dc86ee01fb27b174871ff8be9095ed1a20513772 Mon Sep 17 00:00:00 2001 +From: Babu Moger +Date: Thu, 8 May 2025 14:58:00 -0500 +Subject: [PATCH 18/57] target/i386: Update EPYC-Rome CPU model for Cache + property, RAS, SVM feature bits + +RH-Author: John Allen +RH-MergeRequest: 378: Update EPYC Models and Feature Bits +RH-Jira: RHEL-52649 +RH-Acked-by: Miroslav Rezanina +RH-Commit: [4/8] f23618215eee3c54d9ba52c5b74f5d574c522649 (johnalle/qemu-kvm-fork) + +Found that some of the cache properties are not set correctly for EPYC models. + +l1d_cache.no_invd_sharing should not be true. +l1i_cache.no_invd_sharing should not be true. + +L2.self_init should be true. +L2.inclusive should be true. + +L3.inclusive should not be true. +L3.no_invd_sharing should be true. + +Fix these cache properties. + +Also add the missing RAS and SVM features bits on AMD EPYC-Rome. The SVM +feature bits are used in nested guests. + +succor : Software uncorrectable error containment and recovery capability. +overflow-recov : MCA overflow recovery support. +lbrv : LBR virtualization +tsc-scale : MSR based TSC rate control +vmcb-clean : VMCB clean bits +flushbyasid : Flush by ASID +pause-filter : Pause intercept filter +pfthreshold : PAUSE filter threshold +v-vmsave-vmload : Virtualized VMLOAD and VMSAVE +vgif : Virtualized GIF + +Signed-off-by: Babu Moger +Reviewed-by: Maksim Davydov +Reviewed-by: Zhao Liu +Link: https://lore.kernel.org/r/8265af72057b84c99ac3a02a5487e32759cc69b1.1746734284.git.babu.moger@amd.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 83d940e9700527ff080416ce2fa52ee1f4771d72) + +JIRA: https://issues.redhat.com/browse/RHEL-52649 + +Signed-off-by: John Allen +--- + target/i386/cpu.c | 73 +++++++++++++++++++++++++++++++++++++++++++++++ + 1 file changed, 73 insertions(+) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 32c575f63b..a73b5bfca4 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -2328,6 +2328,60 @@ static const CPUCaches epyc_rome_v3_cache_info = { + }, + }; + ++static const CPUCaches epyc_rome_v5_cache_info = { ++ .l1d_cache = &(CPUCacheInfo) { ++ .type = DATA_CACHE, ++ .level = 1, ++ .size = 32 * KiB, ++ .line_size = 64, ++ .associativity = 8, ++ .partitions = 1, ++ .sets = 64, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l1i_cache = &(CPUCacheInfo) { ++ .type = INSTRUCTION_CACHE, ++ .level = 1, ++ .size = 32 * KiB, ++ .line_size = 64, ++ .associativity = 8, ++ .partitions = 1, ++ .sets = 64, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l2_cache = &(CPUCacheInfo) { ++ .type = UNIFIED_CACHE, ++ .level = 2, ++ .size = 512 * KiB, ++ .line_size = 64, ++ .associativity = 8, ++ .partitions = 1, ++ .sets = 1024, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .inclusive = true, ++ .share_level = CPU_TOPO_LEVEL_CORE, ++ }, ++ .l3_cache = &(CPUCacheInfo) { ++ .type = UNIFIED_CACHE, ++ .level = 3, ++ .size = 16 * MiB, ++ .line_size = 64, ++ .associativity = 16, ++ .partitions = 1, ++ .sets = 16384, ++ .lines_per_tag = 1, ++ .self_init = true, ++ .no_invd_sharing = true, ++ .complex_indexing = false, ++ .share_level = CPU_TOPO_LEVEL_DIE, ++ }, ++}; ++ + static const CPUCaches epyc_milan_cache_info = { + .l1d_cache = &(CPUCacheInfo) { + .type = DATA_CACHE, +@@ -5270,6 +5324,25 @@ static const X86CPUDefinition builtin_x86_defs[] = { + { /* end of list */ } + }, + }, ++ { ++ .version = 5, ++ .props = (PropValue[]) { ++ { "overflow-recov", "on" }, ++ { "succor", "on" }, ++ { "lbrv", "on" }, ++ { "tsc-scale", "on" }, ++ { "vmcb-clean", "on" }, ++ { "flushbyasid", "on" }, ++ { "pause-filter", "on" }, ++ { "pfthreshold", "on" }, ++ { "v-vmsave-vmload", "on" }, ++ { "vgif", "on" }, ++ { "model-id", ++ "AMD EPYC-Rome-v5 Processor" }, ++ { /* end of list */ } ++ }, ++ .cache_info = &epyc_rome_v5_cache_info ++ }, + { /* end of list */ } + } + }, +-- +2.39.3 + diff --git a/SOURCES/kvm-target-i386-allow-reordering-max_x86_cpu_initfn-vs-a.patch b/SOURCES/kvm-target-i386-allow-reordering-max_x86_cpu_initfn-vs-a.patch new file mode 100644 index 0000000..66406c5 --- /dev/null +++ b/SOURCES/kvm-target-i386-allow-reordering-max_x86_cpu_initfn-vs-a.patch @@ -0,0 +1,84 @@ +From 12c5f0bbef0aed7da1b6afa1cd8303aef4f5caf1 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:49 +0200 +Subject: [PATCH 096/115] target/i386: allow reordering max_x86_cpu_initfn vs + accel CPU init + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [96/115] 3cd04c30a233f88bf0e2747bf5d5ad2a06ffe056 (bonzini/rhel-qemu-kvm) + +The PMU feature is only supported by KVM, so move it there. And since +all accelerators other than TCG overwrite the vendor, set it in +max_x86_cpu_initfn only if it has not been initialized by the +superclass. This makes it possible to run max_x86_cpu_initfn +after accelerator init. + +Reviewed-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Signed-off-by: Paolo Bonzini +(cherry picked from commit 810fcc41fc572d90b3c05af3f06f451626ee6b10) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 24 ++++++++++++------------ + target/i386/kvm/kvm-cpu.c | 2 ++ + 2 files changed, 14 insertions(+), 12 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 4df98838a3..dd2180fc3a 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -5880,21 +5880,21 @@ static void max_x86_cpu_class_init(ObjectClass *oc, void *data) + static void max_x86_cpu_initfn(Object *obj) + { + X86CPU *cpu = X86_CPU(obj); +- +- /* We can't fill the features array here because we don't know yet if +- * "migratable" is true or false. +- */ +- object_property_set_bool(OBJECT(cpu), "pmu", true, &error_abort); ++ CPUX86State *env = &cpu->env; + + /* +- * these defaults are used for TCG and all other accelerators +- * besides KVM and HVF, which overwrite these values ++ * these defaults are used for TCG, other accelerators overwrite these ++ * values + */ +- object_property_set_str(OBJECT(cpu), "vendor", CPUID_VENDOR_AMD, +- &error_abort); +- object_property_set_str(OBJECT(cpu), "model-id", +- "QEMU TCG CPU version " QEMU_HW_VERSION, +- &error_abort); ++ if (!env->cpuid_vendor1) { ++ object_property_set_str(OBJECT(cpu), "vendor", CPUID_VENDOR_AMD, ++ &error_abort); ++ } ++ if (!env->cpuid_model[0]) { ++ object_property_set_str(OBJECT(cpu), "model-id", ++ "QEMU TCG CPU version " QEMU_HW_VERSION, ++ &error_abort); ++ } + } + + static const TypeInfo max_x86_cpu_type_info = { +diff --git a/target/i386/kvm/kvm-cpu.c b/target/i386/kvm/kvm-cpu.c +index 49e820b69e..660ccb70f8 100644 +--- a/target/i386/kvm/kvm-cpu.c ++++ b/target/i386/kvm/kvm-cpu.c +@@ -111,6 +111,8 @@ static void kvm_cpu_max_instance_init(X86CPU *cpu) + + host_cpu_max_instance_init(cpu); + ++ object_property_set_bool(OBJECT(cpu), "pmu", true, &error_abort); ++ + if (lmce_supported()) { + object_property_set_bool(OBJECT(cpu), "lmce", true, &error_abort); + } +-- +2.50.1 + diff --git a/SOURCES/kvm-target-i386-merge-host_cpu_instance_init-and-host_cp.patch b/SOURCES/kvm-target-i386-merge-host_cpu_instance_init-and-host_cp.patch new file mode 100644 index 0000000..2ffc422 --- /dev/null +++ b/SOURCES/kvm-target-i386-merge-host_cpu_instance_init-and-host_cp.patch @@ -0,0 +1,103 @@ +From 3a75b23ebb7bb11ed3f25c291df996d7aef21656 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:49 +0200 +Subject: [PATCH 098/115] target/i386: merge host_cpu_instance_init() and + host_cpu_max_instance_init() + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [98/115] 5b4cfd0432cc108ecf67ff80b7573e601a5bea42 (bonzini/rhel-qemu-kvm) + +Simplify the accelerators' cpu_instance_init callbacks by doing all +host-cpu setup in a single function. + +Based-on: <20250711000603.438312-1-pbonzini@redhat.com> +Cc: Xiaoyao Li +Signed-off-by: Paolo Bonzini +(cherry picked from commit 29f1ba338baf60a9e455b6fdc37489ca1efe25aa) +Signed-off-by: Paolo Bonzini +--- + target/i386/host-cpu.c | 28 ++++++++++++++-------------- + target/i386/hvf/hvf-cpu.c | 2 -- + target/i386/kvm/kvm-cpu.c | 2 -- + 3 files changed, 14 insertions(+), 18 deletions(-) + +diff --git a/target/i386/host-cpu.c b/target/i386/host-cpu.c +index 4ab536ab80..52b13daee7 100644 +--- a/target/i386/host-cpu.c ++++ b/target/i386/host-cpu.c +@@ -132,27 +132,27 @@ void host_cpu_instance_init(X86CPU *cpu) + { + X86CPUClass *xcc = X86_CPU_GET_CLASS(cpu); + +- if (xcc->model) { +- char vendor[CPUID_VENDOR_SZ + 1]; +- +- host_cpu_vendor_fms(vendor, NULL, NULL, NULL); +- object_property_set_str(OBJECT(cpu), "vendor", vendor, &error_abort); +- } +-} +- +-void host_cpu_max_instance_init(X86CPU *cpu) +-{ + char vendor[CPUID_VENDOR_SZ + 1] = { 0 }; + char model_id[CPUID_MODEL_ID_SZ + 1] = { 0 }; + int family, model, stepping; + +- /* Use max host physical address bits if -cpu max option is applied */ +- object_property_set_bool(OBJECT(cpu), "host-phys-bits", true, &error_abort); +- ++ /* ++ * setting vendor applies to both max/host and builtin_x86_defs CPU. ++ * FIXME: this probably should warn or should be skipped if vendors do ++ * not match, because family numbers are incompatible between Intel and AMD. ++ */ + host_cpu_vendor_fms(vendor, &family, &model, &stepping); ++ object_property_set_str(OBJECT(cpu), "vendor", vendor, &error_abort); ++ ++ if (!xcc->max_features) { ++ return; ++ } ++ + host_cpu_fill_model_id(model_id); + +- object_property_set_str(OBJECT(cpu), "vendor", vendor, &error_abort); ++ /* Use max host physical address bits if -cpu max option is applied */ ++ object_property_set_bool(OBJECT(cpu), "host-phys-bits", true, &error_abort); ++ + object_property_set_int(OBJECT(cpu), "family", family, &error_abort); + object_property_set_int(OBJECT(cpu), "model", model, &error_abort); + object_property_set_int(OBJECT(cpu), "stepping", stepping, +diff --git a/target/i386/hvf/hvf-cpu.c b/target/i386/hvf/hvf-cpu.c +index 35fc642b93..12cc9502ba 100644 +--- a/target/i386/hvf/hvf-cpu.c ++++ b/target/i386/hvf/hvf-cpu.c +@@ -21,8 +21,6 @@ static void hvf_cpu_max_instance_init(X86CPU *cpu) + { + CPUX86State *env = &cpu->env; + +- host_cpu_max_instance_init(cpu); +- + env->cpuid_min_level = + hvf_get_supported_cpuid(0x0, 0, R_EAX); + env->cpuid_min_xlevel = +diff --git a/target/i386/kvm/kvm-cpu.c b/target/i386/kvm/kvm-cpu.c +index 660ccb70f8..62ee570763 100644 +--- a/target/i386/kvm/kvm-cpu.c ++++ b/target/i386/kvm/kvm-cpu.c +@@ -109,8 +109,6 @@ static void kvm_cpu_max_instance_init(X86CPU *cpu) + CPUX86State *env = &cpu->env; + KVMState *s = kvm_state; + +- host_cpu_max_instance_init(cpu); +- + object_property_set_bool(OBJECT(cpu), "pmu", true, &error_abort); + + if (lmce_supported()) { +-- +2.50.1 + diff --git a/SOURCES/kvm-target-i386-move-accel_cpu_instance_init-to-.instanc.patch b/SOURCES/kvm-target-i386-move-accel_cpu_instance_init-to-.instanc.patch new file mode 100644 index 0000000..793c170 --- /dev/null +++ b/SOURCES/kvm-target-i386-move-accel_cpu_instance_init-to-.instanc.patch @@ -0,0 +1,68 @@ +From 10a611c8c607b7c320eb7c01c444944d8ba318f6 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:49 +0200 +Subject: [PATCH 097/115] target/i386: move accel_cpu_instance_init to + .instance_init + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [97/115] b2725f7b55754f8ec6a54428eb1d70c588eb09ef (bonzini/rhel-qemu-kvm) + +With the reordering of instance_post_init callbacks that is new in 10.1 +accel_cpu_instance_init must execute in .instance_init as is already +the case for RISC-V. Otherwise, for example, setting the vendor +property is broken when using KVM or Hypervisor.framework, because +KVM sets it *after* the user's value is set by DeviceState's +intance_post_init callback. + +Reported-by: Like Xu +Reported-by: Dongli Zhang +Reviewed-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Signed-off-by: Paolo Bonzini +(cherry picked from commit 5f158abef44c7e0945fc5f76715ef135a9bf9bd2) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 8 ++++---- + 1 file changed, 4 insertions(+), 4 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index dd2180fc3a..2160754869 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -5883,8 +5883,8 @@ static void max_x86_cpu_initfn(Object *obj) + CPUX86State *env = &cpu->env; + + /* +- * these defaults are used for TCG, other accelerators overwrite these +- * values ++ * these defaults are used for TCG, other accelerators have overwritten ++ * these values + */ + if (!env->cpuid_vendor1) { + object_property_set_str(OBJECT(cpu), "vendor", CPUID_VENDOR_AMD, +@@ -8636,8 +8636,6 @@ static void x86_cpu_post_initfn(Object *obj) + } + } + +- accel_cpu_instance_init(CPU(obj)); +- + #ifndef CONFIG_USER_ONLY + if (current_machine && current_machine->cgs) { + x86_confidential_guest_cpu_instance_init( +@@ -8712,6 +8710,8 @@ static void x86_cpu_initfn(Object *obj) + if (xcc->model) { + x86_cpu_load_model(cpu, xcc->model); + } ++ ++ accel_cpu_instance_init(CPU(obj)); + } + + static int64_t x86_cpu_get_arch_id(CPUState *cs) +-- +2.50.1 + diff --git a/SOURCES/kvm-target-i386-move-max_features-to-class.patch b/SOURCES/kvm-target-i386-move-max_features-to-class.patch new file mode 100644 index 0000000..0b34c97 --- /dev/null +++ b/SOURCES/kvm-target-i386-move-max_features-to-class.patch @@ -0,0 +1,145 @@ +From 62aee66f496828bec5574d7790501788ad602966 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:49 +0200 +Subject: [PATCH 094/115] target/i386: move max_features to class + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [94/115] f1ce43d8d481176eed2bf0f51c4e1bb423480858 (bonzini/rhel-qemu-kvm) + +max_features is always set to true for instances created by -cpu max or +-cpu host; it's always false for other classes. Therefore it can be +turned into a field in the X86CPUClass. + +Reviewed-by: Xiaoyao Li +Reviewed-by: Zhao Liu +Signed-off-by: Paolo Bonzini +(cherry picked from commit cb2273edf5df57cb270bc19d7e92d8be5870c17a) +Signed-off-by: Paolo Bonzini +--- + target/i386/cpu.c | 7 ++++--- + target/i386/cpu.h | 2 +- + target/i386/hvf/hvf-cpu.c | 3 ++- + target/i386/kvm/kvm-cpu.c | 5 +++-- + 4 files changed, 10 insertions(+), 7 deletions(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index a9d6811032..8685b9d998 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -5868,6 +5868,7 @@ static void max_x86_cpu_class_init(ObjectClass *oc, void *data) + + xcc->ordering = 9; + ++ xcc->max_features = true; + xcc->model_description = + "Enables all features supported by the accelerator in the current host"; + +@@ -5882,7 +5883,6 @@ static void max_x86_cpu_initfn(Object *obj) + /* We can't fill the features array here because we don't know yet if + * "migratable" is true or false. + */ +- cpu->max_features = true; + object_property_set_bool(OBJECT(cpu), "pmu", true, &error_abort); + + /* +@@ -7954,6 +7954,7 @@ static void x86_cpu_enable_xsave_components(X86CPU *cpu) + */ + void x86_cpu_expand_features(X86CPU *cpu, Error **errp) + { ++ X86CPUClass *xcc = X86_CPU_GET_CLASS(cpu); + CPUX86State *env = &cpu->env; + FeatureWord w; + int i; +@@ -7973,12 +7974,12 @@ void x86_cpu_expand_features(X86CPU *cpu, Error **errp) + } + } + +- /*TODO: Now cpu->max_features doesn't overwrite features ++ /* TODO: Now xcc->max_features doesn't overwrite features + * set using QOM properties, and we can convert + * plus_features & minus_features to global properties + * inside x86_cpu_parse_featurestr() too. + */ +- if (cpu->max_features) { ++ if (xcc->max_features) { + for (w = 0; w < FEATURE_WORDS; w++) { + /* Override only features that weren't set explicitly + * by the user. +diff --git a/target/i386/cpu.h b/target/i386/cpu.h +index a08931f969..a419eb64a5 100644 +--- a/target/i386/cpu.h ++++ b/target/i386/cpu.h +@@ -2122,7 +2122,6 @@ struct ArchCPU { + bool expose_tcg; + bool migratable; + bool migrate_smi_count; +- bool max_features; /* Enable all supported features automatically */ + uint32_t apic_id; + + /* Enables publishing of TSC increment and Local APIC bus frequencies to +@@ -2275,6 +2274,7 @@ struct X86CPUClass { + */ + X86CPUModel *model; + ++ bool max_features; /* Enable all supported features automatically */ + bool host_cpuid_required; + int ordering; + bool migration_safe; +diff --git a/target/i386/hvf/hvf-cpu.c b/target/i386/hvf/hvf-cpu.c +index ac617f17e7..35fc642b93 100644 +--- a/target/i386/hvf/hvf-cpu.c ++++ b/target/i386/hvf/hvf-cpu.c +@@ -61,13 +61,14 @@ static void hvf_cpu_xsave_init(void) + static void hvf_cpu_instance_init(CPUState *cs) + { + X86CPU *cpu = X86_CPU(cs); ++ X86CPUClass *xcc = X86_CPU_GET_CLASS(cpu); + + host_cpu_instance_init(cpu); + + /* Special cases not set in the X86CPUDefinition structs: */ + /* TODO: in-kernel irqchip for hvf */ + +- if (cpu->max_features) { ++ if (xcc->max_features) { + hvf_cpu_max_instance_init(cpu); + } + +diff --git a/target/i386/kvm/kvm-cpu.c b/target/i386/kvm/kvm-cpu.c +index 961b87e98e..49e820b69e 100644 +--- a/target/i386/kvm/kvm-cpu.c ++++ b/target/i386/kvm/kvm-cpu.c +@@ -41,6 +41,7 @@ static void kvm_set_guest_phys_bits(CPUState *cs) + static bool kvm_cpu_realizefn(CPUState *cs, Error **errp) + { + X86CPU *cpu = X86_CPU(cs); ++ X86CPUClass *xcc = X86_CPU_GET_CLASS(cpu); + CPUX86State *env = &cpu->env; + bool ret; + +@@ -63,7 +64,7 @@ static bool kvm_cpu_realizefn(CPUState *cs, Error **errp) + * check/update ucode_rev, phys_bits, guest_phys_bits, mwait + * cpu_common_realizefn() (via xcc->parent_realize) + */ +- if (cpu->max_features) { ++ if (xcc->max_features) { + if (enable_cpu_pm) { + if (kvm_has_waitpkg()) { + env->features[FEAT_7_0_ECX] |= CPUID_7_0_ECX_WAITPKG; +@@ -217,7 +218,7 @@ static void kvm_cpu_instance_init(CPUState *cs) + x86_cpu_apply_props(cpu, kvm_default_props); + } + +- if (cpu->max_features) { ++ if (xcc->max_features) { + kvm_cpu_max_instance_init(cpu); + } + +-- +2.50.1 + diff --git a/SOURCES/kvm-target-i386-nvmm-whpx-add-accel-CPU-class-that-sets-.patch b/SOURCES/kvm-target-i386-nvmm-whpx-add-accel-CPU-class-that-sets-.patch new file mode 100644 index 0000000..3bc3a21 --- /dev/null +++ b/SOURCES/kvm-target-i386-nvmm-whpx-add-accel-CPU-class-that-sets-.patch @@ -0,0 +1,164 @@ +From a0f80cbb64e88f7f4e222e31d2ace756ece782f7 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:49 +0200 +Subject: [PATCH 095/115] target/i386: nvmm, whpx: add accel/CPU class that + sets host vendor + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [95/115] b579400cf9a23c6405af130a5fed54fc41063bcb (bonzini/rhel-qemu-kvm) + +NVMM and WHPX are virtualizers, and therefore they need to use +(at least by default) the host vendor for the guest CPUID. +Add a cpu_instance_init implementation to these accelerators. + +Signed-off-by: Paolo Bonzini +(cherry picked from commit d93972d88b0984ed0a2090493f8d62cc188976d2) +Signed-off-by: Paolo Bonzini + +Conflicts: system/ -> sysemu/ +--- + target/i386/cpu.c | 3 ++- + target/i386/meson.build | 2 ++ + target/i386/nvmm/nvmm-all.c | 25 +++++++++++++++++++++++++ + target/i386/whpx/whpx-all.c | 25 +++++++++++++++++++++++++ + 4 files changed, 54 insertions(+), 1 deletion(-) + +diff --git a/target/i386/cpu.c b/target/i386/cpu.c +index 8685b9d998..4df98838a3 100644 +--- a/target/i386/cpu.c ++++ b/target/i386/cpu.c +@@ -43,6 +43,7 @@ + #include "hw/boards.h" + #include "hw/i386/sgx-epc.h" + #endif ++#include "sysemu/qtest.h" + #include "tcg/tcg-cpu.h" + + #include "disas/capstone.h" +@@ -1893,7 +1894,7 @@ uint32_t xsave_area_size(uint64_t mask, bool compacted) + + static inline bool accel_uses_host_cpuid(void) + { +- return kvm_enabled() || hvf_enabled(); ++ return !tcg_enabled() && !qtest_enabled(); + } + + static inline uint64_t x86_cpu_xsave_xcr0_components(X86CPU *cpu) +diff --git a/target/i386/meson.build b/target/i386/meson.build +index 075117989b..9572b31040 100644 +--- a/target/i386/meson.build ++++ b/target/i386/meson.build +@@ -11,6 +11,8 @@ i386_ss.add(when: 'CONFIG_SEV', if_true: files('host-cpu.c', 'confidential-guest + # x86 cpu type + i386_ss.add(when: 'CONFIG_KVM', if_true: files('host-cpu.c')) + i386_ss.add(when: 'CONFIG_HVF', if_true: files('host-cpu.c')) ++i386_ss.add(when: 'CONFIG_WHPX', if_true: files('host-cpu.c')) ++i386_ss.add(when: 'CONFIG_NVMM', if_true: files('host-cpu.c')) + + i386_system_ss = ss.source_set() + i386_system_ss.add(files( +diff --git a/target/i386/nvmm/nvmm-all.c b/target/i386/nvmm/nvmm-all.c +index 65768aca03..2cc84a15f9 100644 +--- a/target/i386/nvmm/nvmm-all.c ++++ b/target/i386/nvmm/nvmm-all.c +@@ -19,6 +19,8 @@ + #include "qemu/error-report.h" + #include "qapi/error.h" + #include "qemu/queue.h" ++#include "accel/accel-cpu-target.h" ++#include "host-cpu.h" + #include "migration/blocker.h" + #include "strings.h" + +@@ -1214,10 +1216,33 @@ static const TypeInfo nvmm_accel_type = { + .class_init = nvmm_accel_class_init, + }; + ++static void nvmm_cpu_instance_init(CPUState *cs) ++{ ++ X86CPU *cpu = X86_CPU(cs); ++ ++ host_cpu_instance_init(cpu); ++} ++ ++static void nvmm_cpu_accel_class_init(ObjectClass *oc, const void *data) ++{ ++ AccelCPUClass *acc = ACCEL_CPU_CLASS(oc); ++ ++ acc->cpu_instance_init = nvmm_cpu_instance_init; ++} ++ ++static const TypeInfo nvmm_cpu_accel_type = { ++ .name = ACCEL_CPU_NAME("nvmm"), ++ ++ .parent = TYPE_ACCEL_CPU, ++ .class_init = nvmm_cpu_accel_class_init, ++ .abstract = true, ++}; ++ + static void + nvmm_type_init(void) + { + type_register_static(&nvmm_accel_type); ++ type_register_static(&nvmm_cpu_accel_type); + } + + type_init(nvmm_type_init); +diff --git a/target/i386/whpx/whpx-all.c b/target/i386/whpx/whpx-all.c +index a6674a826d..ea209d50f8 100644 +--- a/target/i386/whpx/whpx-all.c ++++ b/target/i386/whpx/whpx-all.c +@@ -26,6 +26,8 @@ + #include "qapi/qapi-types-common.h" + #include "qapi/qapi-visit-common.h" + #include "migration/blocker.h" ++#include "host-cpu.h" ++#include "accel/accel-cpu-target.h" + #include + + #include "whpx-internal.h" +@@ -2512,6 +2514,28 @@ static void whpx_set_kernel_irqchip(Object *obj, Visitor *v, + } + } + ++static void whpx_cpu_instance_init(CPUState *cs) ++{ ++ X86CPU *cpu = X86_CPU(cs); ++ ++ host_cpu_instance_init(cpu); ++} ++ ++static void whpx_cpu_accel_class_init(ObjectClass *oc, const void *data) ++{ ++ AccelCPUClass *acc = ACCEL_CPU_CLASS(oc); ++ ++ acc->cpu_instance_init = whpx_cpu_instance_init; ++} ++ ++static const TypeInfo whpx_cpu_accel_type = { ++ .name = ACCEL_CPU_NAME("whpx"), ++ ++ .parent = TYPE_ACCEL_CPU, ++ .class_init = whpx_cpu_accel_class_init, ++ .abstract = true, ++}; ++ + /* + * Partition support + */ +@@ -2742,6 +2766,7 @@ static const TypeInfo whpx_accel_type = { + static void whpx_type_init(void) + { + type_register_static(&whpx_accel_type); ++ type_register_static(&whpx_cpu_accel_type); + } + + bool init_whp_dispatch(void) +-- +2.50.1 + diff --git a/SOURCES/kvm-target-i386-sev-Reduce-system-specific-declarations.patch b/SOURCES/kvm-target-i386-sev-Reduce-system-specific-declarations.patch new file mode 100644 index 0000000..cc5997a --- /dev/null +++ b/SOURCES/kvm-target-i386-sev-Reduce-system-specific-declarations.patch @@ -0,0 +1,99 @@ +From a4bc6c4fc28364e8ca9fc99344b85254268744e3 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:44 +0200 +Subject: [PATCH 019/115] target/i386/sev: Reduce system specific declarations +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [19/115] 0015a372d990c69ce81241822722bf2521571ca7 (bonzini/rhel-qemu-kvm) + +"system/confidential-guest-support.h" is not needed, +remove it. Reorder #ifdef'ry to reduce declarations +exposed on user emulation. + +Signed-off-by: Philippe Mathieu-Daudé +Reviewed-by: Thomas Huth +Reviewed-by: Zhao Liu +Message-Id: <20241218155913.72288-3-philmd@linaro.org> +(cherry picked from commit 63cda19446c5307cc05b965c203742a583fc5abf) +Signed-off-by: Paolo Bonzini +--- + hw/i386/pc_sysfw.c | 2 +- + target/i386/sev.h | 29 ++++++++++++++++------------- + 2 files changed, 17 insertions(+), 14 deletions(-) + +diff --git a/hw/i386/pc_sysfw.c b/hw/i386/pc_sysfw.c +index ef80281d28..e6271e1020 100644 +--- a/hw/i386/pc_sysfw.c ++++ b/hw/i386/pc_sysfw.c +@@ -36,7 +36,7 @@ + #include "hw/qdev-properties.h" + #include "hw/block/flash.h" + #include "sysemu/kvm.h" +-#include "sev.h" ++#include "target/i386/sev.h" + + #define FLASH_SECTOR_SIZE 4096 + +diff --git a/target/i386/sev.h b/target/i386/sev.h +index 858005a119..373669eaac 100644 +--- a/target/i386/sev.h ++++ b/target/i386/sev.h +@@ -18,7 +18,17 @@ + #include CONFIG_DEVICES /* CONFIG_SEV */ + #endif + +-#include "exec/confidential-guest-support.h" ++#if !defined(CONFIG_SEV) || defined(CONFIG_USER_ONLY) ++#define sev_enabled() 0 ++#define sev_es_enabled() 0 ++#define sev_snp_enabled() 0 ++#else ++bool sev_enabled(void); ++bool sev_es_enabled(void); ++bool sev_snp_enabled(void); ++#endif ++ ++#if !defined(CONFIG_USER_ONLY) + + #define TYPE_SEV_COMMON "sev-common" + #define TYPE_SEV_GUEST "sev-guest" +@@ -45,18 +55,6 @@ typedef struct SevKernelLoaderContext { + size_t cmdline_size; + } SevKernelLoaderContext; + +-#ifdef CONFIG_SEV +-bool sev_enabled(void); +-bool sev_es_enabled(void); +-bool sev_snp_enabled(void); +-#else +-#define sev_enabled() 0 +-#define sev_es_enabled() 0 +-#define sev_snp_enabled() 0 +-#endif +- +-uint32_t sev_get_cbit_position(void); +-uint32_t sev_get_reduced_phys_bits(void); + bool sev_add_kernel_loader_hashes(SevKernelLoaderContext *ctx, Error **errp); + + int sev_encrypt_flash(hwaddr gpa, uint8_t *ptr, uint64_t len, Error **errp); +@@ -68,4 +66,9 @@ void sev_es_set_reset_vector(CPUState *cpu); + + void pc_system_parse_sev_metadata(uint8_t *flash_ptr, size_t flash_size); + ++#endif /* !CONFIG_USER_ONLY */ ++ ++uint32_t sev_get_cbit_position(void); ++uint32_t sev_get_reduced_phys_bits(void); ++ + #endif +-- +2.50.1 + diff --git a/SOURCES/kvm-target-i386-tdx-fix-locking-for-interrupt-injection.patch b/SOURCES/kvm-target-i386-tdx-fix-locking-for-interrupt-injection.patch new file mode 100644 index 0000000..c2775a6 --- /dev/null +++ b/SOURCES/kvm-target-i386-tdx-fix-locking-for-interrupt-injection.patch @@ -0,0 +1,62 @@ +From e8d9e0b0c8b44da13be0735e67576d6d2336750a Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:50 +0200 +Subject: [PATCH 106/115] target/i386: tdx: fix locking for interrupt injection + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [106/115] 0ec3f0d2a022735ef6dbd5b0cce10ee348dde3f1 (bonzini/rhel-qemu-kvm) + +Take tdx_guest->lock when injecting the event notification interrupt into +the guest. + +Fixes CID 1612364. + +Reported-by: Peter Maydell +Cc: Xiaoyao Li +Reviewed-by: Xiaoyao Li +Signed-off-by: Paolo Bonzini +(cherry picked from commit f2b787976342a9e1d47810f3146ad74b86a5088a) +Signed-off-by: Paolo Bonzini +--- + target/i386/kvm/tdx.c | 10 +++++++--- + 1 file changed, 7 insertions(+), 3 deletions(-) + +diff --git a/target/i386/kvm/tdx.c b/target/i386/kvm/tdx.c +index 20fcd9a4c5..08eed19960 100644 +--- a/target/i386/kvm/tdx.c ++++ b/target/i386/kvm/tdx.c +@@ -1126,10 +1126,15 @@ int tdx_parse_tdvf(void *flash_ptr, int size) + return tdvf_parse_metadata(&tdx_guest->tdvf, flash_ptr, size); + } + +-static void tdx_inject_interrupt(uint32_t apicid, uint32_t vector) ++static void tdx_inject_interrupt(TdxGuest *tdx) + { + int ret; ++ uint32_t apicid, vector; + ++ qemu_mutex_lock(&tdx->lock); ++ vector = tdx->event_notify_vector; ++ apicid = tdx->event_notify_apicid; ++ qemu_mutex_unlock(&tdx->lock); + if (vector < 32 || vector > 255) { + return; + } +@@ -1179,8 +1184,7 @@ static void tdx_get_quote_completion(TdxGenerateQuoteTask *task) + error_report("TDX: get-quote: failed to update GetQuote header."); + } + +- tdx_inject_interrupt(tdx_guest->event_notify_apicid, +- tdx_guest->event_notify_vector); ++ tdx_inject_interrupt(tdx); + + g_free(task->send_data); + g_free(task->receive_buf); +-- +2.50.1 + diff --git a/SOURCES/kvm-target-s390-Convert-CPU-to-Resettable-interface.patch b/SOURCES/kvm-target-s390-Convert-CPU-to-Resettable-interface.patch new file mode 100644 index 0000000..6b93825 --- /dev/null +++ b/SOURCES/kvm-target-s390-Convert-CPU-to-Resettable-interface.patch @@ -0,0 +1,284 @@ +From 50c4cbbe0a8849dd0c720c6e706498cb0d46f5b3 Mon Sep 17 00:00:00 2001 +From: Peter Maydell +Date: Fri, 13 Sep 2024 15:31:43 +0100 +Subject: [PATCH 04/26] target/s390: Convert CPU to Resettable interface + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [4/26] 157b29ced6b92ecec5e69f8bc60d0183a0c88fa0 (thuth/qemu-kvm-cs) + +Convert the s390 CPU to the Resettable interface. This is slightly +more involved than the other CPU types were (see commits +9130cade5fc22..d66e64dd006df) because S390 has its own set of +different kinds of reset with different behaviours that it needs to +trigger. + +We handle this by adding these reset types to the Resettable +ResetType enum. Now instead of having an underlying implementation +of reset that is s390-specific and which might be called either +directly or via the DeviceClass::reset method, we can implement only +the Resettable hold phase method, and have the places that need to +trigger an s390-specific reset type do so by calling +resettable_reset(). + +The other option would have been to smuggle in the s390 reset +type via, for instance, a field in the CPU state that we set +in s390_do_cpu_initial_reset() etc and then examined in the +reset method, but doing it this way seems cleaner. + +The motivation for this change is that this is the last caller +of the legacy device_class_set_parent_reset() function, and +removing that will let us clean up some glue code that we added +for the transition to three-phase reset. + +Signed-off-by: Peter Maydell +Reviewed-by: Nina Schoetterl-Glausch +Reviewed-by: Richard Henderson +Acked-by: Thomas Huth +Message-id: 20240830145812.1967042-4-peter.maydell@linaro.org +(cherry picked from commit cf7f61d13f28f32d0b14abb70ce1bd9e41623b2e) +Signed-off-by: Thomas Huth +--- + docs/devel/reset.rst | 10 ++++++++++ + include/hw/resettable.h | 2 ++ + target/s390x/cpu.c | 38 +++++++++++++++++--------------------- + target/s390x/cpu.h | 21 ++++----------------- + target/s390x/sigp.c | 8 ++------ + 5 files changed, 35 insertions(+), 44 deletions(-) + +diff --git a/docs/devel/reset.rst b/docs/devel/reset.rst +index 24ab630465..d2799eba7a 100644 +--- a/docs/devel/reset.rst ++++ b/docs/devel/reset.rst +@@ -44,6 +44,16 @@ The Resettable interface handles reset types with an enum ``ResetType``: + value on each cold reset, such as RNG seed information, and which they + must not reinitialize on a snapshot-load reset. + ++``RESET_TYPE_S390_CPU_NORMAL`` ++ This is only used for S390 CPU objects; it clears interrupts, stops ++ processing, and clears the TLB, but does not touch register contents. ++ ++``RESET_TYPE_S390_CPU_INITIAL`` ++ This is only used for S390 CPU objects; it does everything ++ ``RESET_TYPE_S390_CPU_NORMAL`` does and also clears the PSW, prefix, ++ FPC, timer and control registers. It does not touch gprs, fprs or acrs. ++ ++ + Devices which implement reset methods must treat any unknown ``ResetType`` + as equivalent to ``RESET_TYPE_COLD``; this will reduce the amount of + existing code we need to change if we add more types in future. +diff --git a/include/hw/resettable.h b/include/hw/resettable.h +index 7e249deb8b..83b561fc83 100644 +--- a/include/hw/resettable.h ++++ b/include/hw/resettable.h +@@ -36,6 +36,8 @@ typedef struct ResettableState ResettableState; + typedef enum ResetType { + RESET_TYPE_COLD, + RESET_TYPE_SNAPSHOT_LOAD, ++ RESET_TYPE_S390_CPU_INITIAL, ++ RESET_TYPE_S390_CPU_NORMAL, + } ResetType; + + /* +diff --git a/target/s390x/cpu.c b/target/s390x/cpu.c +index 0fbfcd35d8..4e41a3dff5 100644 +--- a/target/s390x/cpu.c ++++ b/target/s390x/cpu.c +@@ -32,6 +32,7 @@ + #include "sysemu/hw_accel.h" + #include "hw/qdev-properties.h" + #include "hw/qdev-properties-system.h" ++#include "hw/resettable.h" + #include "fpu/softfloat-helpers.h" + #include "disas/capstone.h" + #include "sysemu/tcg.h" +@@ -162,23 +163,25 @@ static void s390_query_cpu_fast(CPUState *cpu, CpuInfoFast *value) + #endif + } + +-/* S390CPUClass::reset() */ +-static void s390_cpu_reset(CPUState *s, cpu_reset_type type) ++/* S390CPUClass Resettable reset_hold phase method */ ++static void s390_cpu_reset_hold(Object *obj, ResetType type) + { +- S390CPU *cpu = S390_CPU(s); ++ S390CPU *cpu = S390_CPU(obj); + S390CPUClass *scc = S390_CPU_GET_CLASS(cpu); + CPUS390XState *env = &cpu->env; +- DeviceState *dev = DEVICE(s); + +- scc->parent_reset(dev); ++ if (scc->parent_phases.hold) { ++ scc->parent_phases.hold(obj, type); ++ } + cpu->env.sigp_order = 0; + s390_cpu_set_state(S390_CPU_STATE_STOPPED, cpu); + + switch (type) { +- case S390_CPU_RESET_CLEAR: ++ default: ++ /* RESET_TYPE_COLD: power on or "clear" reset */ + memset(env, 0, offsetof(CPUS390XState, start_initial_reset_fields)); + /* fall through */ +- case S390_CPU_RESET_INITIAL: ++ case RESET_TYPE_S390_CPU_INITIAL: + /* initial reset does not clear everything! */ + memset(&env->start_initial_reset_fields, 0, + offsetof(CPUS390XState, start_normal_reset_fields) - +@@ -203,7 +206,7 @@ static void s390_cpu_reset(CPUState *s, cpu_reset_type type) + set_float_detect_tininess(float_tininess_before_rounding, + &env->fpu_status); + /* fall through */ +- case S390_CPU_RESET_NORMAL: ++ case RESET_TYPE_S390_CPU_NORMAL: + env->psw.mask &= ~PSW_MASK_RI; + memset(&env->start_normal_reset_fields, 0, + offsetof(CPUS390XState, end_reset_fields) - +@@ -212,20 +215,18 @@ static void s390_cpu_reset(CPUState *s, cpu_reset_type type) + env->pfault_token = -1UL; + env->bpbc = false; + break; +- default: +- g_assert_not_reached(); + } + + /* Reset state inside the kernel that we cannot access yet from QEMU. */ + if (kvm_enabled()) { + switch (type) { +- case S390_CPU_RESET_CLEAR: ++ default: + kvm_s390_reset_vcpu_clear(cpu); + break; +- case S390_CPU_RESET_INITIAL: ++ case RESET_TYPE_S390_CPU_INITIAL: + kvm_s390_reset_vcpu_initial(cpu); + break; +- case S390_CPU_RESET_NORMAL: ++ case RESET_TYPE_S390_CPU_NORMAL: + kvm_s390_reset_vcpu_normal(cpu); + break; + } +@@ -315,12 +316,6 @@ static Property s390x_cpu_properties[] = { + DEFINE_PROP_END_OF_LIST() + }; + +-static void s390_cpu_reset_full(DeviceState *dev) +-{ +- CPUState *s = CPU(dev); +- return s390_cpu_reset(s, S390_CPU_RESET_CLEAR); +-} +- + #ifdef CONFIG_TCG + #include "hw/core/tcg-cpu-ops.h" + +@@ -383,15 +378,16 @@ static void s390_cpu_class_init(ObjectClass *oc, void *data) + S390CPUClass *scc = S390_CPU_CLASS(oc); + CPUClass *cc = CPU_CLASS(scc); + DeviceClass *dc = DEVICE_CLASS(oc); ++ ResettableClass *rc = RESETTABLE_CLASS(oc); + + device_class_set_parent_realize(dc, s390_cpu_realizefn, + &scc->parent_realize); + device_class_set_props(dc, s390x_cpu_properties); + dc->user_creatable = true; + +- device_class_set_parent_reset(dc, s390_cpu_reset_full, &scc->parent_reset); ++ resettable_class_set_parent_phases(rc, NULL, s390_cpu_reset_hold, NULL, ++ &scc->parent_phases); + +- scc->reset = s390_cpu_reset; + cc->class_by_name = s390_cpu_class_by_name, + cc->has_work = s390_cpu_has_work; + cc->mmu_index = s390x_cpu_mmu_index; +diff --git a/target/s390x/cpu.h b/target/s390x/cpu.h +index d6b75ad0e0..6a64472403 100644 +--- a/target/s390x/cpu.h ++++ b/target/s390x/cpu.h +@@ -177,19 +177,11 @@ struct ArchCPU { + uint32_t irqstate_saved_size; + }; + +-typedef enum cpu_reset_type { +- S390_CPU_RESET_NORMAL, +- S390_CPU_RESET_INITIAL, +- S390_CPU_RESET_CLEAR, +-} cpu_reset_type; +- + /** + * S390CPUClass: + * @parent_realize: The parent class' realize handler. +- * @parent_reset: The parent class' reset handler. ++ * @parent_phases: The parent class' reset phase handlers. + * @load_normal: Performs a load normal. +- * @cpu_reset: Performs a CPU reset. +- * @initial_cpu_reset: Performs an initial CPU reset. + * + * An S/390 CPU model. + */ +@@ -203,9 +195,8 @@ struct S390CPUClass { + const char *desc; + + DeviceRealize parent_realize; +- DeviceReset parent_reset; ++ ResettablePhases parent_phases; + void (*load_normal)(CPUState *cpu); +- void (*reset)(CPUState *cpu, cpu_reset_type type); + }; + + #ifndef CONFIG_USER_ONLY +@@ -872,16 +863,12 @@ static inline void s390_do_cpu_full_reset(CPUState *cs, run_on_cpu_data arg) + + static inline void s390_do_cpu_reset(CPUState *cs, run_on_cpu_data arg) + { +- S390CPUClass *scc = S390_CPU_GET_CLASS(cs); +- +- scc->reset(cs, S390_CPU_RESET_NORMAL); ++ resettable_reset(OBJECT(cs), RESET_TYPE_S390_CPU_NORMAL); + } + + static inline void s390_do_cpu_initial_reset(CPUState *cs, run_on_cpu_data arg) + { +- S390CPUClass *scc = S390_CPU_GET_CLASS(cs); +- +- scc->reset(cs, S390_CPU_RESET_INITIAL); ++ resettable_reset(OBJECT(cs), RESET_TYPE_S390_CPU_INITIAL); + } + + static inline void s390_do_cpu_load_normal(CPUState *cs, run_on_cpu_data arg) +diff --git a/target/s390x/sigp.c b/target/s390x/sigp.c +index ad0ad61177..08aaecf12b 100644 +--- a/target/s390x/sigp.c ++++ b/target/s390x/sigp.c +@@ -251,24 +251,20 @@ static void sigp_restart(CPUState *cs, run_on_cpu_data arg) + + static void sigp_initial_cpu_reset(CPUState *cs, run_on_cpu_data arg) + { +- S390CPU *cpu = S390_CPU(cs); +- S390CPUClass *scc = S390_CPU_GET_CLASS(cpu); + SigpInfo *si = arg.host_ptr; + + cpu_synchronize_state(cs); +- scc->reset(cs, S390_CPU_RESET_INITIAL); ++ resettable_reset(OBJECT(cs), RESET_TYPE_S390_CPU_INITIAL); + cpu_synchronize_post_reset(cs); + si->cc = SIGP_CC_ORDER_CODE_ACCEPTED; + } + + static void sigp_cpu_reset(CPUState *cs, run_on_cpu_data arg) + { +- S390CPU *cpu = S390_CPU(cs); +- S390CPUClass *scc = S390_CPU_GET_CLASS(cpu); + SigpInfo *si = arg.host_ptr; + + cpu_synchronize_state(cs); +- scc->reset(cs, S390_CPU_RESET_NORMAL); ++ resettable_reset(OBJECT(cs), RESET_TYPE_S390_CPU_NORMAL); + cpu_synchronize_post_reset(cs); + si->cc = SIGP_CC_ORDER_CODE_ACCEPTED; + } +-- +2.48.1 + diff --git a/SOURCES/kvm-tests-Add-iotest-mirror-sparse-for-recent-patches.patch b/SOURCES/kvm-tests-Add-iotest-mirror-sparse-for-recent-patches.patch new file mode 100644 index 0000000..2becf51 --- /dev/null +++ b/SOURCES/kvm-tests-Add-iotest-mirror-sparse-for-recent-patches.patch @@ -0,0 +1,545 @@ +From e72aaba2efda48e083d92e6dacfe58667bdfa958 Mon Sep 17 00:00:00 2001 +From: Eric Blake +Date: Fri, 9 May 2025 15:40:30 -0500 +Subject: [PATCH 15/16] tests: Add iotest mirror-sparse for recent patches + +RH-Author: Eric Blake +RH-MergeRequest: 365: blockdev-mirror: More efficient handling of sparse mirrors +RH-Jira: RHEL-82906 RHEL-83015 +RH-Acked-by: Stefan Hajnoczi +RH-Acked-by: Jon Maloy +RH-Commit: [13/14] 6b7792e85b81e45d11c2664349db75d905e72adf (ebblake/centos-qemu-kvm) + +Prove that blockdev-mirror can now result in sparse raw destination +files, regardless of whether the source is raw or qcow2. By making +this a separate test, it was possible to test effects of individual +patches for the various pieces that all have to work together for a +sparse mirror to be successful. + +Note that ./check -file produces different job lengths than ./check +-qcow2 (the test uses a filter to normalize); that's because when +deciding how much of the image to be mirrored, the code looks at how +much of the source image was allocated (for qcow2, this is only the +written clusters; for raw, it is the entire file). But the important +part is that the destination file ends up smaller than 3M, rather than +the 20M it used to be before this patch series. + +Signed-off-by: Eric Blake +Message-ID: <20250509204341.3553601-28-eblake@redhat.com> +Reviewed-by: Stefan Hajnoczi +(cherry picked from commit c0ddcb2cbc146e64f666eaae4edc7b5db7e5814d) +Jira: https://issues.redhat.com/browse/RHEL-82906 +Jira: https://issues.redhat.com/browse/RHEL-83015 +Signed-off-by: Eric Blake +--- + tests/qemu-iotests/tests/mirror-sparse | 125 +++++++ + tests/qemu-iotests/tests/mirror-sparse.out | 365 +++++++++++++++++++++ + 2 files changed, 490 insertions(+) + create mode 100755 tests/qemu-iotests/tests/mirror-sparse + create mode 100644 tests/qemu-iotests/tests/mirror-sparse.out + +diff --git a/tests/qemu-iotests/tests/mirror-sparse b/tests/qemu-iotests/tests/mirror-sparse +new file mode 100755 +index 0000000000..8c52a4e244 +--- /dev/null ++++ b/tests/qemu-iotests/tests/mirror-sparse +@@ -0,0 +1,125 @@ ++#!/usr/bin/env bash ++# group: rw auto quick ++# ++# Test blockdev-mirror with raw sparse destination ++# ++# Copyright (C) 2025 Red Hat, Inc. ++# ++# This program is free software; you can redistribute it and/or modify ++# it under the terms of the GNU General Public License as published by ++# the Free Software Foundation; either version 2 of the License, or ++# (at your option) any later version. ++# ++# This program is distributed in the hope that it will be useful, ++# but WITHOUT ANY WARRANTY; without even the implied warranty of ++# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the ++# GNU General Public License for more details. ++# ++# You should have received a copy of the GNU General Public License ++# along with this program. If not, see . ++# ++ ++seq="$(basename $0)" ++echo "QA output created by $seq" ++ ++status=1 # failure is the default! ++ ++_cleanup() ++{ ++ _cleanup_test_img ++ _cleanup_qemu ++} ++trap "_cleanup; exit \$status" 0 1 2 3 15 ++ ++# get standard environment, filters and checks ++cd .. ++. ./common.rc ++. ./common.filter ++. ./common.qemu ++ ++_supported_fmt qcow2 raw # Format of the source. dst is always raw file ++_supported_proto file ++_supported_os Linux ++ ++echo ++echo "=== Initial image setup ===" ++echo ++ ++TEST_IMG="$TEST_IMG.base" _make_test_img 20M ++$QEMU_IO -c 'w 8M 2M' -f $IMGFMT "$TEST_IMG.base" | _filter_qemu_io ++ ++_launch_qemu \ ++ -blockdev '{"driver":"file", "cache":{"direct":true, "no-flush":false}, ++ "filename":"'"$TEST_IMG.base"'", "node-name":"src-file"}' \ ++ -blockdev '{"driver":"'$IMGFMT'", "node-name":"src", "file":"src-file"}' ++h1=$QEMU_HANDLE ++_send_qemu_cmd $h1 '{"execute": "qmp_capabilities"}' 'return' ++ ++# Check several combinations; most should result in a sparse destination; ++# the destination should only be fully allocated if pre-allocated ++# and not punching holes due to detect-zeroes ++# do_test creation discard zeroes result ++do_test() { ++ creation=$1 ++ discard=$2 ++ zeroes=$3 ++ expected=$4 ++ ++echo ++echo "=== Testing creation=$creation discard=$discard zeroes=$zeroes ===" ++echo ++ ++rm -f $TEST_IMG ++if test $creation = external; then ++ truncate --size=20M $TEST_IMG ++else ++ _send_qemu_cmd $h1 '{"execute": "blockdev-create", "arguments": ++ {"options": {"driver":"file", "filename":"'$TEST_IMG'", ++ "size":'$((20*1024*1024))', "preallocation":"'$creation'"}, ++ "job-id":"job1"}}' 'concluded' ++ _send_qemu_cmd $h1 '{"execute": "job-dismiss", "arguments": ++ {"id": "job1"}}' 'return' ++fi ++_send_qemu_cmd $h1 '{"execute": "blockdev-add", "arguments": ++ {"node-name": "dst", "driver":"file", ++ "filename":"'$TEST_IMG'", "aio":"threads", ++ "auto-read-only":true, "discard":"'$discard'", ++ "detect-zeroes":"'$zeroes'"}}' 'return' ++_send_qemu_cmd $h1 '{"execute":"blockdev-mirror", "arguments": ++ {"sync":"full", "device":"src", "target":"dst", ++ "job-id":"job2"}}' 'return' ++_timed_wait_for $h1 '"ready"' ++_send_qemu_cmd $h1 '{"execute": "job-complete", "arguments": ++ {"id":"job2"}}' 'return' \ ++ | _filter_block_job_offset | _filter_block_job_len ++_send_qemu_cmd $h1 '{"execute": "blockdev-del", "arguments": ++ {"node-name": "dst"}}' 'return' \ ++ | _filter_block_job_offset | _filter_block_job_len ++$QEMU_IMG compare -U -f $IMGFMT -F raw $TEST_IMG.base $TEST_IMG ++result=$(disk_usage $TEST_IMG) ++if test $result -lt $((3*1024*1024)); then ++ actual=sparse ++elif test $result = $((20*1024*1024)); then ++ actual=full ++else ++ actual=unknown ++fi ++echo "Destination is $actual; expected $expected" ++} ++ ++do_test external ignore off sparse ++do_test external unmap off sparse ++do_test external unmap unmap sparse ++do_test off ignore off sparse ++do_test off unmap off sparse ++do_test off unmap unmap sparse ++do_test full ignore off full ++do_test full unmap off sparse ++do_test full unmap unmap sparse ++ ++_send_qemu_cmd $h1 '{"execute":"quit"}' '' ++ ++# success, all done ++echo '*** done' ++rm -f $seq.full ++status=0 +diff --git a/tests/qemu-iotests/tests/mirror-sparse.out b/tests/qemu-iotests/tests/mirror-sparse.out +new file mode 100644 +index 0000000000..2103b891c3 +--- /dev/null ++++ b/tests/qemu-iotests/tests/mirror-sparse.out +@@ -0,0 +1,365 @@ ++QA output created by mirror-sparse ++ ++=== Initial image setup === ++ ++Formatting 'TEST_DIR/t.IMGFMT.base', fmt=IMGFMT size=20971520 ++wrote 2097152/2097152 bytes at offset 8388608 ++2 MiB, X ops; XX:XX:XX.X (XXX YYY/sec and XXX ops/sec) ++{"execute": "qmp_capabilities"} ++{"return": {}} ++ ++=== Testing creation=external discard=ignore zeroes=off === ++ ++{"execute": "blockdev-add", "arguments": ++ {"node-name": "dst", "driver":"file", ++ "filename":"TEST_DIR/t.IMGFMT", "aio":"threads", ++ "auto-read-only":true, "discard":"ignore", ++ "detect-zeroes":"off"}} ++{"return": {}} ++{"execute":"blockdev-mirror", "arguments": ++ {"sync":"full", "device":"src", "target":"dst", ++ "job-id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job2"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "ready", "id": "job2"}} ++{"execute": "job-complete", "arguments": ++ {"id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_READY", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"return": {}} ++{"execute": "blockdev-del", "arguments": ++ {"node-name": "dst"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_COMPLETED", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job2"}} ++{"return": {}} ++Images are identical. ++Destination is sparse; expected sparse ++ ++=== Testing creation=external discard=unmap zeroes=off === ++ ++{"execute": "blockdev-add", "arguments": ++ {"node-name": "dst", "driver":"file", ++ "filename":"TEST_DIR/t.IMGFMT", "aio":"threads", ++ "auto-read-only":true, "discard":"unmap", ++ "detect-zeroes":"off"}} ++{"return": {}} ++{"execute":"blockdev-mirror", "arguments": ++ {"sync":"full", "device":"src", "target":"dst", ++ "job-id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job2"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "ready", "id": "job2"}} ++{"execute": "job-complete", "arguments": ++ {"id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_READY", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"return": {}} ++{"execute": "blockdev-del", "arguments": ++ {"node-name": "dst"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_COMPLETED", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job2"}} ++{"return": {}} ++Images are identical. ++Destination is sparse; expected sparse ++ ++=== Testing creation=external discard=unmap zeroes=unmap === ++ ++{"execute": "blockdev-add", "arguments": ++ {"node-name": "dst", "driver":"file", ++ "filename":"TEST_DIR/t.IMGFMT", "aio":"threads", ++ "auto-read-only":true, "discard":"unmap", ++ "detect-zeroes":"unmap"}} ++{"return": {}} ++{"execute":"blockdev-mirror", "arguments": ++ {"sync":"full", "device":"src", "target":"dst", ++ "job-id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job2"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "ready", "id": "job2"}} ++{"execute": "job-complete", "arguments": ++ {"id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_READY", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"return": {}} ++{"execute": "blockdev-del", "arguments": ++ {"node-name": "dst"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_COMPLETED", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job2"}} ++{"return": {}} ++Images are identical. ++Destination is sparse; expected sparse ++ ++=== Testing creation=off discard=ignore zeroes=off === ++ ++{"execute": "blockdev-create", "arguments": ++ {"options": {"driver":"file", "filename":"TEST_DIR/t.IMGFMT", ++ "size":20971520, "preallocation":"off"}, ++ "job-id":"job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job1"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job1"}} ++{"execute": "job-dismiss", "arguments": ++ {"id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job1"}} ++{"return": {}} ++{"execute": "blockdev-add", "arguments": ++ {"node-name": "dst", "driver":"file", ++ "filename":"TEST_DIR/t.IMGFMT", "aio":"threads", ++ "auto-read-only":true, "discard":"ignore", ++ "detect-zeroes":"off"}} ++{"return": {}} ++{"execute":"blockdev-mirror", "arguments": ++ {"sync":"full", "device":"src", "target":"dst", ++ "job-id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job2"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "ready", "id": "job2"}} ++{"execute": "job-complete", "arguments": ++ {"id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_READY", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"return": {}} ++{"execute": "blockdev-del", "arguments": ++ {"node-name": "dst"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_COMPLETED", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job2"}} ++{"return": {}} ++Images are identical. ++Destination is sparse; expected sparse ++ ++=== Testing creation=off discard=unmap zeroes=off === ++ ++{"execute": "blockdev-create", "arguments": ++ {"options": {"driver":"file", "filename":"TEST_DIR/t.IMGFMT", ++ "size":20971520, "preallocation":"off"}, ++ "job-id":"job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job1"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job1"}} ++{"execute": "job-dismiss", "arguments": ++ {"id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job1"}} ++{"return": {}} ++{"execute": "blockdev-add", "arguments": ++ {"node-name": "dst", "driver":"file", ++ "filename":"TEST_DIR/t.IMGFMT", "aio":"threads", ++ "auto-read-only":true, "discard":"unmap", ++ "detect-zeroes":"off"}} ++{"return": {}} ++{"execute":"blockdev-mirror", "arguments": ++ {"sync":"full", "device":"src", "target":"dst", ++ "job-id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job2"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "ready", "id": "job2"}} ++{"execute": "job-complete", "arguments": ++ {"id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_READY", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"return": {}} ++{"execute": "blockdev-del", "arguments": ++ {"node-name": "dst"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_COMPLETED", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job2"}} ++{"return": {}} ++Images are identical. ++Destination is sparse; expected sparse ++ ++=== Testing creation=off discard=unmap zeroes=unmap === ++ ++{"execute": "blockdev-create", "arguments": ++ {"options": {"driver":"file", "filename":"TEST_DIR/t.IMGFMT", ++ "size":20971520, "preallocation":"off"}, ++ "job-id":"job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job1"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job1"}} ++{"execute": "job-dismiss", "arguments": ++ {"id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job1"}} ++{"return": {}} ++{"execute": "blockdev-add", "arguments": ++ {"node-name": "dst", "driver":"file", ++ "filename":"TEST_DIR/t.IMGFMT", "aio":"threads", ++ "auto-read-only":true, "discard":"unmap", ++ "detect-zeroes":"unmap"}} ++{"return": {}} ++{"execute":"blockdev-mirror", "arguments": ++ {"sync":"full", "device":"src", "target":"dst", ++ "job-id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job2"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "ready", "id": "job2"}} ++{"execute": "job-complete", "arguments": ++ {"id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_READY", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"return": {}} ++{"execute": "blockdev-del", "arguments": ++ {"node-name": "dst"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_COMPLETED", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job2"}} ++{"return": {}} ++Images are identical. ++Destination is sparse; expected sparse ++ ++=== Testing creation=full discard=ignore zeroes=off === ++ ++{"execute": "blockdev-create", "arguments": ++ {"options": {"driver":"file", "filename":"TEST_DIR/t.IMGFMT", ++ "size":20971520, "preallocation":"full"}, ++ "job-id":"job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job1"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job1"}} ++{"execute": "job-dismiss", "arguments": ++ {"id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job1"}} ++{"return": {}} ++{"execute": "blockdev-add", "arguments": ++ {"node-name": "dst", "driver":"file", ++ "filename":"TEST_DIR/t.IMGFMT", "aio":"threads", ++ "auto-read-only":true, "discard":"ignore", ++ "detect-zeroes":"off"}} ++{"return": {}} ++{"execute":"blockdev-mirror", "arguments": ++ {"sync":"full", "device":"src", "target":"dst", ++ "job-id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job2"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "ready", "id": "job2"}} ++{"execute": "job-complete", "arguments": ++ {"id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_READY", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"return": {}} ++{"execute": "blockdev-del", "arguments": ++ {"node-name": "dst"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_COMPLETED", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job2"}} ++{"return": {}} ++Images are identical. ++Destination is full; expected full ++ ++=== Testing creation=full discard=unmap zeroes=off === ++ ++{"execute": "blockdev-create", "arguments": ++ {"options": {"driver":"file", "filename":"TEST_DIR/t.IMGFMT", ++ "size":20971520, "preallocation":"full"}, ++ "job-id":"job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job1"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job1"}} ++{"execute": "job-dismiss", "arguments": ++ {"id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job1"}} ++{"return": {}} ++{"execute": "blockdev-add", "arguments": ++ {"node-name": "dst", "driver":"file", ++ "filename":"TEST_DIR/t.IMGFMT", "aio":"threads", ++ "auto-read-only":true, "discard":"unmap", ++ "detect-zeroes":"off"}} ++{"return": {}} ++{"execute":"blockdev-mirror", "arguments": ++ {"sync":"full", "device":"src", "target":"dst", ++ "job-id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job2"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "ready", "id": "job2"}} ++{"execute": "job-complete", "arguments": ++ {"id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_READY", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"return": {}} ++{"execute": "blockdev-del", "arguments": ++ {"node-name": "dst"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_COMPLETED", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job2"}} ++{"return": {}} ++Images are identical. ++Destination is sparse; expected sparse ++ ++=== Testing creation=full discard=unmap zeroes=unmap === ++ ++{"execute": "blockdev-create", "arguments": ++ {"options": {"driver":"file", "filename":"TEST_DIR/t.IMGFMT", ++ "size":20971520, "preallocation":"full"}, ++ "job-id":"job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job1"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job1"}} ++{"execute": "job-dismiss", "arguments": ++ {"id": "job1"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job1"}} ++{"return": {}} ++{"execute": "blockdev-add", "arguments": ++ {"node-name": "dst", "driver":"file", ++ "filename":"TEST_DIR/t.IMGFMT", "aio":"threads", ++ "auto-read-only":true, "discard":"unmap", ++ "detect-zeroes":"unmap"}} ++{"return": {}} ++{"execute":"blockdev-mirror", "arguments": ++ {"sync":"full", "device":"src", "target":"dst", ++ "job-id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "created", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "running", "id": "job2"}} ++{"return": {}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "ready", "id": "job2"}} ++{"execute": "job-complete", "arguments": ++ {"id":"job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_READY", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"return": {}} ++{"execute": "blockdev-del", "arguments": ++ {"node-name": "dst"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "waiting", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "pending", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "BLOCK_JOB_COMPLETED", "data": {"device": "job2", "len": LEN, "offset": OFFSET, "speed": 0, "type": "mirror"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "concluded", "id": "job2"}} ++{"timestamp": {"seconds": TIMESTAMP, "microseconds": TIMESTAMP}, "event": "JOB_STATUS_CHANGE", "data": {"status": "null", "id": "job2"}} ++{"return": {}} ++Images are identical. ++Destination is sparse; expected sparse ++{"execute":"quit"} ++*** done +-- +2.48.1 + diff --git a/SOURCES/kvm-tests-unit-test-util-sockets-fix-mem-leak-on-error-o.patch b/SOURCES/kvm-tests-unit-test-util-sockets-fix-mem-leak-on-error-o.patch new file mode 100644 index 0000000..d0787f4 --- /dev/null +++ b/SOURCES/kvm-tests-unit-test-util-sockets-fix-mem-leak-on-error-o.patch @@ -0,0 +1,53 @@ +From 83f09a8c65e1fef416e39d9b0a4ead14ed00601e Mon Sep 17 00:00:00 2001 +From: Matheus Tavares Bernardino +Date: Mon, 26 May 2025 10:20:55 -0700 +Subject: [PATCH 14/57] tests/unit/test-util-sockets: fix mem-leak on error + object + +RH-Author: Juraj Marcin +RH-MergeRequest: 369: util/qemu-sockets: Introduce inet socket options controlling TCP keep-alive +RH-Jira: RHEL-67104 +RH-Acked-by: Peter Xu +RH-Acked-by: Miroslav Rezanina +RH-Commit: [7/7] 31c74f784a26812e5c5898efacda9b4069874ed7 (JurajMarcin/centos-src-qemu-kvm) + +The test fails with --enable-asan as the error struct is never freed. +In the case where the test expects a success but it fails, let's also +report the error for debugging (it will be freed internally). + +Fixes 316e8ee8d6 ("util/qemu-sockets: Refactor inet_parse() to use QemuOpts") + +Signed-off-by: Matheus Tavares Bernardino +Reviewed-by: Juraj Marcin +Message-ID: <518d94c7db20060b2a086cf55ee9bffab992a907.1748280011.git.matheus.bernardino@oss.qualcomm.com> +Signed-off-by: Thomas Huth + +(cherry picked from commit 5c54a367265ec19ed94a535cd15d178c16b8cae0) + +JIRA: https://issues.redhat.com/browse/RHEL-67104 + +Signed-off-by: Juraj Marcin +--- + tests/unit/test-util-sockets.c | 4 ++++ + 1 file changed, 4 insertions(+) + +diff --git a/tests/unit/test-util-sockets.c b/tests/unit/test-util-sockets.c +index 8492f4d68f..ee66d727c3 100644 +--- a/tests/unit/test-util-sockets.c ++++ b/tests/unit/test-util-sockets.c +@@ -341,8 +341,12 @@ static void inet_parse_test_helper(const char *str, + int rc = inet_parse(&addr, str, &error); + + if (success) { ++ if (error) { ++ error_report_err(error); ++ } + g_assert_cmpint(rc, ==, 0); + } else { ++ error_free(error); + g_assert_cmpint(rc, <, 0); + } + if (exp_addr != NULL) { +-- +2.39.3 + diff --git a/SOURCES/kvm-ui-vnc-Update-display-update-interval-when-VM-state-.patch b/SOURCES/kvm-ui-vnc-Update-display-update-interval-when-VM-state-.patch index bec1913..9d30908 100644 --- a/SOURCES/kvm-ui-vnc-Update-display-update-interval-when-VM-state-.patch +++ b/SOURCES/kvm-ui-vnc-Update-display-update-interval-when-VM-state-.patch @@ -1,18 +1,18 @@ -From e2931430d3f10dd521e6b4cc7505842fbc8296ec Mon Sep 17 00:00:00 2001 +From a69b8d66fb515cd55cef2fcaa626c350d761f1a9 Mon Sep 17 00:00:00 2001 From: Juraj Marcin Date: Wed, 21 May 2025 17:16:13 +0200 -Subject: [PATCH 01/31] ui/vnc: Update display update interval when VM state +Subject: [PATCH 57/57] ui/vnc: Update display update interval when VM state changes to RUNNING MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RH-Author: Juraj Marcin -RH-MergeRequest: 463: ui/vnc: Update display update interval when VM state changes to RUNNING -RH-Jira: RHEL-100767 +RH-MergeRequest: 385: ui/vnc: Update display update interval when VM state changes to RUNNING +RH-Jira: RHEL-100741 RH-Acked-by: Peter Xu RH-Acked-by: Marc-André Lureau -RH-Commit: [1/1] 60b1a7921296e82b616d055691fe8ac0f2e283b1 +RH-Commit: [1/1] 30dd4790a607d646465c18d621073df997e8850b (JurajMarcin/centos-src-qemu-kvm) If a virtual machine is paused for an extended period time, for example, due to an incoming migration, there are also no changes on the screen. @@ -41,7 +41,7 @@ Signed-off-by: Peter Xu (cherry picked from commit 0310d594d98b39f9dde79b87fd8b0ad16e7c5459) -JIRA: https://issues.redhat.com/browse/RHEL-100767 +JIRA: https://issues.redhat.com/browse/RHEL-100741 Signed-off-by: Juraj Marcin --- diff --git a/SOURCES/kvm-update-Linux-headers-to-KVM-tree-master.patch b/SOURCES/kvm-update-Linux-headers-to-KVM-tree-master.patch new file mode 100644 index 0000000..eb12dfc --- /dev/null +++ b/SOURCES/kvm-update-Linux-headers-to-KVM-tree-master.patch @@ -0,0 +1,62 @@ +From ef8fda39f273b864ab558c48c6f3ff1f28b38e44 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:45 +0200 +Subject: [PATCH 030/115] update Linux headers to KVM tree master + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [30/115] 8ca98dabb9f8b02543e363840ae9a23d0f736f5e (bonzini/rhel-qemu-kvm) + +To fetch the update of TDX + +Signed-off-by: Xiaoyao Li +Link: https://lore.kernel.org/r/20250703024021.3559286-3-xiaoyao.li@intel.com +Signed-off-by: Paolo Bonzini +(cherry picked from commit 25c98a135001559be905a0399669e5cdb3b0a613) +Signed-off-by: Paolo Bonzini +--- + linux-headers/asm-x86/kvm.h | 8 +++++++- + linux-headers/linux/kvm.h | 4 ++++ + 2 files changed, 11 insertions(+), 1 deletion(-) + +diff --git a/linux-headers/asm-x86/kvm.h b/linux-headers/asm-x86/kvm.h +index cd275ae76d..f0c1a730d9 100644 +--- a/linux-headers/asm-x86/kvm.h ++++ b/linux-headers/asm-x86/kvm.h +@@ -963,7 +963,13 @@ struct kvm_tdx_cmd { + struct kvm_tdx_capabilities { + __u64 supported_attrs; + __u64 supported_xfam; +- __u64 reserved[254]; ++ ++ __u64 kernel_tdvmcallinfo_1_r11; ++ __u64 user_tdvmcallinfo_1_r11; ++ __u64 kernel_tdvmcallinfo_1_r12; ++ __u64 user_tdvmcallinfo_1_r12; ++ ++ __u64 reserved[250]; + + /* Configurable CPUID bits for userspace */ + struct kvm_cpuid2 cpuid; +diff --git a/linux-headers/linux/kvm.h b/linux-headers/linux/kvm.h +index 0690743944..32c5885a3c 100644 +--- a/linux-headers/linux/kvm.h ++++ b/linux-headers/linux/kvm.h +@@ -459,6 +459,10 @@ struct kvm_run { + __u64 leaf; + __u64 r11, r12, r13, r14; + } get_tdvmcall_info; ++ struct { ++ __u64 ret; ++ __u64 vector; ++ } setup_event_notify; + }; + } tdx; + /* Fix the size of the union. */ +-- +2.50.1 + diff --git a/SOURCES/kvm-update-Linux-headers-to-v6.16-rc3.patch b/SOURCES/kvm-update-Linux-headers-to-v6.16-rc3.patch new file mode 100644 index 0000000..47a9f4a --- /dev/null +++ b/SOURCES/kvm-update-Linux-headers-to-v6.16-rc3.patch @@ -0,0 +1,497 @@ +From 2fbd65006bac2b17a2fe6b709536bea965517b93 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 18 Jul 2025 18:03:45 +0200 +Subject: [PATCH 029/115] update Linux headers to v6.16-rc3 + +RH-Author: Paolo Bonzini +RH-MergeRequest: 391: TDX support, including attestation and device assignment +RH-Jira: RHEL-15710 RHEL-20798 RHEL-49728 +RH-Acked-by: Yash Mankad +RH-Acked-by: Peter Xu +RH-Acked-by: David Hildenbrand +RH-Commit: [29/115] d92884a0574b2dd9037ee6911312b66b6c14299d (bonzini/rhel-qemu-kvm) + +Signed-off-by: Paolo Bonzini +(cherry picked from commit 688b0756ad2fdbe8effdb66f724a1129f62be7a2) +Signed-off-by: Paolo Bonzini +--- + include/standard-headers/asm-x86/setup_data.h | 13 +- + include/standard-headers/drm/drm_fourcc.h | 45 ++++++ + include/standard-headers/linux/ethtool.h | 134 +++++++++--------- + include/standard-headers/linux/fuse.h | 6 +- + .../linux/input-event-codes.h | 3 +- + include/standard-headers/linux/pci_regs.h | 12 +- + include/standard-headers/linux/virtio_gpu.h | 3 +- + include/standard-headers/linux/virtio_pci.h | 1 + + linux-headers/asm-arm64/kvm.h | 9 +- + linux-headers/asm-x86/kvm.h | 1 + + linux-headers/linux/bits.h | 4 +- + linux-headers/linux/kvm.h | 25 ++++ + linux-headers/linux/vhost.h | 4 +- + 13 files changed, 182 insertions(+), 78 deletions(-) + +diff --git a/include/standard-headers/asm-x86/setup_data.h b/include/standard-headers/asm-x86/setup_data.h +index a483d72f42..2e446c1d85 100644 +--- a/include/standard-headers/asm-x86/setup_data.h ++++ b/include/standard-headers/asm-x86/setup_data.h +@@ -13,7 +13,8 @@ + #define SETUP_CC_BLOB 7 + #define SETUP_IMA 8 + #define SETUP_RNG_SEED 9 +-#define SETUP_ENUM_MAX SETUP_RNG_SEED ++#define SETUP_KEXEC_KHO 10 ++#define SETUP_ENUM_MAX SETUP_KEXEC_KHO + + #define SETUP_INDIRECT (1<<31) + #define SETUP_TYPE_MAX (SETUP_ENUM_MAX | SETUP_INDIRECT) +@@ -78,6 +79,16 @@ struct ima_setup_data { + uint64_t size; + } QEMU_PACKED; + ++/* ++ * Locations of kexec handover metadata ++ */ ++struct kho_data { ++ uint64_t fdt_addr; ++ uint64_t fdt_size; ++ uint64_t scratch_addr; ++ uint64_t scratch_size; ++} QEMU_PACKED; ++ + #endif /* __ASSEMBLER__ */ + + #endif /* _ASM_X86_SETUP_DATA_H */ +diff --git a/include/standard-headers/drm/drm_fourcc.h b/include/standard-headers/drm/drm_fourcc.h +index a8b759dcbc..c8309d378b 100644 +--- a/include/standard-headers/drm/drm_fourcc.h ++++ b/include/standard-headers/drm/drm_fourcc.h +@@ -421,6 +421,7 @@ extern "C" { + #define DRM_FORMAT_MOD_VENDOR_ALLWINNER 0x09 + #define DRM_FORMAT_MOD_VENDOR_AMLOGIC 0x0a + #define DRM_FORMAT_MOD_VENDOR_MTK 0x0b ++#define DRM_FORMAT_MOD_VENDOR_APPLE 0x0c + + /* add more to the end as needed */ + +@@ -1493,6 +1494,50 @@ drm_fourcc_canonicalize_nvidia_format_mod(uint64_t modifier) + /* alias for the most common tiling format */ + #define DRM_FORMAT_MOD_MTK_16L_32S_TILE DRM_FORMAT_MOD_MTK(MTK_FMT_MOD_TILE_16L32S) + ++/* ++ * Apple GPU-tiled layouts. ++ * ++ * Apple GPUs support nonlinear tilings with optional lossless compression. ++ * ++ * GPU-tiled images are divided into 16KiB tiles: ++ * ++ * Bytes per pixel Tile size ++ * --------------- --------- ++ * 1 128x128 ++ * 2 128x64 ++ * 4 64x64 ++ * 8 64x32 ++ * 16 32x32 ++ * ++ * Tiles are raster-order. Pixels within a tile are interleaved (Morton order). ++ * ++ * Compressed images pad the body to 128-bytes and are immediately followed by a ++ * metadata section. The metadata section rounds the image dimensions to ++ * powers-of-two and contains 8 bytes for each 16x16 compression subtile. ++ * Subtiles are interleaved (Morton order). ++ * ++ * All images are 128-byte aligned. ++ * ++ * These layouts fundamentally do not have meaningful strides. No matter how we ++ * specify strides for these layouts, userspace unaware of Apple image layouts ++ * will be unable to use correctly the specified stride for any purpose. ++ * Userspace aware of the image layouts do not use strides. The most "correct" ++ * convention would be setting the image stride to 0. Unfortunately, some ++ * software assumes the stride is at least (width * bytes per pixel). We ++ * therefore require that stride equals (width * bytes per pixel). Since the ++ * stride is arbitrary here, we pick the simplest convention. ++ * ++ * Although containing two sections, compressed image layouts are treated in ++ * software as a single plane. This is modelled after AFBC, a similar ++ * scheme. Attempting to separate the sections to be "explicit" in DRM would ++ * only generate more confusion, as software does not treat the image this way. ++ * ++ * For detailed information on the hardware image layouts, see ++ * https://docs.mesa3d.org/drivers/asahi.html#image-layouts ++ */ ++#define DRM_FORMAT_MOD_APPLE_GPU_TILED fourcc_mod_code(APPLE, 1) ++#define DRM_FORMAT_MOD_APPLE_GPU_TILED_COMPRESSED fourcc_mod_code(APPLE, 2) ++ + /* + * AMD modifiers + * +diff --git a/include/standard-headers/linux/ethtool.h b/include/standard-headers/linux/ethtool.h +index 5d1ad5fdea..cef0d207a6 100644 +--- a/include/standard-headers/linux/ethtool.h ++++ b/include/standard-headers/linux/ethtool.h +@@ -2295,71 +2295,75 @@ static inline int ethtool_validate_duplex(uint8_t duplex) + #define RXH_XFRM_SYM_OR_XOR (1 << 1) + #define RXH_XFRM_NO_CHANGE 0xff + +-/* L2-L4 network traffic flow types */ +-#define TCP_V4_FLOW 0x01 /* hash or spec (tcp_ip4_spec) */ +-#define UDP_V4_FLOW 0x02 /* hash or spec (udp_ip4_spec) */ +-#define SCTP_V4_FLOW 0x03 /* hash or spec (sctp_ip4_spec) */ +-#define AH_ESP_V4_FLOW 0x04 /* hash only */ +-#define TCP_V6_FLOW 0x05 /* hash or spec (tcp_ip6_spec; nfc only) */ +-#define UDP_V6_FLOW 0x06 /* hash or spec (udp_ip6_spec; nfc only) */ +-#define SCTP_V6_FLOW 0x07 /* hash or spec (sctp_ip6_spec; nfc only) */ +-#define AH_ESP_V6_FLOW 0x08 /* hash only */ +-#define AH_V4_FLOW 0x09 /* hash or spec (ah_ip4_spec) */ +-#define ESP_V4_FLOW 0x0a /* hash or spec (esp_ip4_spec) */ +-#define AH_V6_FLOW 0x0b /* hash or spec (ah_ip6_spec; nfc only) */ +-#define ESP_V6_FLOW 0x0c /* hash or spec (esp_ip6_spec; nfc only) */ +-#define IPV4_USER_FLOW 0x0d /* spec only (usr_ip4_spec) */ +-#define IP_USER_FLOW IPV4_USER_FLOW +-#define IPV6_USER_FLOW 0x0e /* spec only (usr_ip6_spec; nfc only) */ +-#define IPV4_FLOW 0x10 /* hash only */ +-#define IPV6_FLOW 0x11 /* hash only */ +-#define ETHER_FLOW 0x12 /* spec only (ether_spec) */ +- +-/* Used for GTP-U IPv4 and IPv6. +- * The format of GTP packets only includes +- * elements such as TEID and GTP version. +- * It is primarily intended for data communication of the UE. +- */ +-#define GTPU_V4_FLOW 0x13 /* hash only */ +-#define GTPU_V6_FLOW 0x14 /* hash only */ +- +-/* Use for GTP-C IPv4 and v6. +- * The format of these GTP packets does not include TEID. +- * Primarily expected to be used for communication +- * to create sessions for UE data communication, +- * commonly referred to as CSR (Create Session Request). +- */ +-#define GTPC_V4_FLOW 0x15 /* hash only */ +-#define GTPC_V6_FLOW 0x16 /* hash only */ +- +-/* Use for GTP-C IPv4 and v6. +- * Unlike GTPC_V4_FLOW, the format of these GTP packets includes TEID. +- * After session creation, it becomes this packet. +- * This is mainly used for requests to realize UE handover. +- */ +-#define GTPC_TEID_V4_FLOW 0x17 /* hash only */ +-#define GTPC_TEID_V6_FLOW 0x18 /* hash only */ +- +-/* Use for GTP-U and extended headers for the PSC (PDU Session Container). +- * The format of these GTP packets includes TEID and QFI. +- * In 5G communication using UPF (User Plane Function), +- * data communication with this extended header is performed. +- */ +-#define GTPU_EH_V4_FLOW 0x19 /* hash only */ +-#define GTPU_EH_V6_FLOW 0x1a /* hash only */ +- +-/* Use for GTP-U IPv4 and v6 PSC (PDU Session Container) extended headers. +- * This differs from GTPU_EH_V(4|6)_FLOW in that it is distinguished by +- * UL/DL included in the PSC. +- * There are differences in the data included based on Downlink/Uplink, +- * and can be used to distinguish packets. +- * The functions described so far are useful when you want to +- * handle communication from the mobile network in UPF, PGW, etc. +- */ +-#define GTPU_UL_V4_FLOW 0x1b /* hash only */ +-#define GTPU_UL_V6_FLOW 0x1c /* hash only */ +-#define GTPU_DL_V4_FLOW 0x1d /* hash only */ +-#define GTPU_DL_V6_FLOW 0x1e /* hash only */ ++enum { ++ /* L2-L4 network traffic flow types */ ++ TCP_V4_FLOW = 0x01, /* hash or spec (tcp_ip4_spec) */ ++ UDP_V4_FLOW = 0x02, /* hash or spec (udp_ip4_spec) */ ++ SCTP_V4_FLOW = 0x03, /* hash or spec (sctp_ip4_spec) */ ++ AH_ESP_V4_FLOW = 0x04, /* hash only */ ++ TCP_V6_FLOW = 0x05, /* hash or spec (tcp_ip6_spec; nfc only) */ ++ UDP_V6_FLOW = 0x06, /* hash or spec (udp_ip6_spec; nfc only) */ ++ SCTP_V6_FLOW = 0x07, /* hash or spec (sctp_ip6_spec; nfc only) */ ++ AH_ESP_V6_FLOW = 0x08, /* hash only */ ++ AH_V4_FLOW = 0x09, /* hash or spec (ah_ip4_spec) */ ++ ESP_V4_FLOW = 0x0a, /* hash or spec (esp_ip4_spec) */ ++ AH_V6_FLOW = 0x0b, /* hash or spec (ah_ip6_spec; nfc only) */ ++ ESP_V6_FLOW = 0x0c, /* hash or spec (esp_ip6_spec; nfc only) */ ++ IPV4_USER_FLOW = 0x0d, /* spec only (usr_ip4_spec) */ ++ IP_USER_FLOW = IPV4_USER_FLOW, ++ IPV6_USER_FLOW = 0x0e, /* spec only (usr_ip6_spec; nfc only) */ ++ IPV4_FLOW = 0x10, /* hash only */ ++ IPV6_FLOW = 0x11, /* hash only */ ++ ETHER_FLOW = 0x12, /* spec only (ether_spec) */ ++ ++ /* Used for GTP-U IPv4 and IPv6. ++ * The format of GTP packets only includes ++ * elements such as TEID and GTP version. ++ * It is primarily intended for data communication of the UE. ++ */ ++ GTPU_V4_FLOW = 0x13, /* hash only */ ++ GTPU_V6_FLOW = 0x14, /* hash only */ ++ ++ /* Use for GTP-C IPv4 and v6. ++ * The format of these GTP packets does not include TEID. ++ * Primarily expected to be used for communication ++ * to create sessions for UE data communication, ++ * commonly referred to as CSR (Create Session Request). ++ */ ++ GTPC_V4_FLOW = 0x15, /* hash only */ ++ GTPC_V6_FLOW = 0x16, /* hash only */ ++ ++ /* Use for GTP-C IPv4 and v6. ++ * Unlike GTPC_V4_FLOW, the format of these GTP packets includes TEID. ++ * After session creation, it becomes this packet. ++ * This is mainly used for requests to realize UE handover. ++ */ ++ GTPC_TEID_V4_FLOW = 0x17, /* hash only */ ++ GTPC_TEID_V6_FLOW = 0x18, /* hash only */ ++ ++ /* Use for GTP-U and extended headers for the PSC (PDU Session Container). ++ * The format of these GTP packets includes TEID and QFI. ++ * In 5G communication using UPF (User Plane Function), ++ * data communication with this extended header is performed. ++ */ ++ GTPU_EH_V4_FLOW = 0x19, /* hash only */ ++ GTPU_EH_V6_FLOW = 0x1a, /* hash only */ ++ ++ /* Use for GTP-U IPv4 and v6 PSC (PDU Session Container) extended headers. ++ * This differs from GTPU_EH_V(4|6)_FLOW in that it is distinguished by ++ * UL/DL included in the PSC. ++ * There are differences in the data included based on Downlink/Uplink, ++ * and can be used to distinguish packets. ++ * The functions described so far are useful when you want to ++ * handle communication from the mobile network in UPF, PGW, etc. ++ */ ++ GTPU_UL_V4_FLOW = 0x1b, /* hash only */ ++ GTPU_UL_V6_FLOW = 0x1c, /* hash only */ ++ GTPU_DL_V4_FLOW = 0x1d, /* hash only */ ++ GTPU_DL_V6_FLOW = 0x1e, /* hash only */ ++ ++ __FLOW_TYPE_COUNT, ++}; + + /* Flag to enable additional fields in struct ethtool_rx_flow_spec */ + #define FLOW_EXT 0x80000000 +diff --git a/include/standard-headers/linux/fuse.h b/include/standard-headers/linux/fuse.h +index a2b5815d89..d8b2fd67e1 100644 +--- a/include/standard-headers/linux/fuse.h ++++ b/include/standard-headers/linux/fuse.h +@@ -232,6 +232,9 @@ + * + * 7.43 + * - add FUSE_REQUEST_TIMEOUT ++ * ++ * 7.44 ++ * - add FUSE_NOTIFY_INC_EPOCH + */ + + #ifndef _LINUX_FUSE_H +@@ -263,7 +266,7 @@ + #define FUSE_KERNEL_VERSION 7 + + /** Minor version number of this interface */ +-#define FUSE_KERNEL_MINOR_VERSION 43 ++#define FUSE_KERNEL_MINOR_VERSION 44 + + /** The node ID of the root inode */ + #define FUSE_ROOT_ID 1 +@@ -667,6 +670,7 @@ enum fuse_notify_code { + FUSE_NOTIFY_RETRIEVE = 5, + FUSE_NOTIFY_DELETE = 6, + FUSE_NOTIFY_RESEND = 7, ++ FUSE_NOTIFY_INC_EPOCH = 8, + FUSE_NOTIFY_CODE_MAX, + }; + +diff --git a/include/standard-headers/linux/input-event-codes.h b/include/standard-headers/linux/input-event-codes.h +index 09ba0ad878..a82ff795e0 100644 +--- a/include/standard-headers/linux/input-event-codes.h ++++ b/include/standard-headers/linux/input-event-codes.h +@@ -925,7 +925,8 @@ + #define SW_MUTE_DEVICE 0x0e /* set = device disabled */ + #define SW_PEN_INSERTED 0x0f /* set = pen inserted */ + #define SW_MACHINE_COVER 0x10 /* set = cover closed */ +-#define SW_MAX_ 0x10 ++#define SW_USB_INSERT 0x11 /* set = USB audio device connected */ ++#define SW_MAX_ 0x11 + #define SW_CNT (SW_MAX_+1) + + /* +diff --git a/include/standard-headers/linux/pci_regs.h b/include/standard-headers/linux/pci_regs.h +index ba326710f9..a3a3e942de 100644 +--- a/include/standard-headers/linux/pci_regs.h ++++ b/include/standard-headers/linux/pci_regs.h +@@ -750,7 +750,8 @@ + #define PCI_EXT_CAP_ID_NPEM 0x29 /* Native PCIe Enclosure Management */ + #define PCI_EXT_CAP_ID_PL_32GT 0x2A /* Physical Layer 32.0 GT/s */ + #define PCI_EXT_CAP_ID_DOE 0x2E /* Data Object Exchange */ +-#define PCI_EXT_CAP_ID_MAX PCI_EXT_CAP_ID_DOE ++#define PCI_EXT_CAP_ID_PL_64GT 0x31 /* Physical Layer 64.0 GT/s */ ++#define PCI_EXT_CAP_ID_MAX PCI_EXT_CAP_ID_PL_64GT + + #define PCI_EXT_CAP_DSN_SIZEOF 12 + #define PCI_EXT_CAP_MCAST_ENDPOINT_SIZEOF 40 +@@ -1144,12 +1145,21 @@ + #define PCI_DLF_CAP 0x04 /* Capabilities Register */ + #define PCI_DLF_EXCHANGE_ENABLE 0x80000000 /* Data Link Feature Exchange Enable */ + ++/* Secondary PCIe Capability 8.0 GT/s */ ++#define PCI_SECPCI_LE_CTRL 0x0c /* Lane Equalization Control Register */ ++ + /* Physical Layer 16.0 GT/s */ + #define PCI_PL_16GT_LE_CTRL 0x20 /* Lane Equalization Control Register */ + #define PCI_PL_16GT_LE_CTRL_DSP_TX_PRESET_MASK 0x0000000F + #define PCI_PL_16GT_LE_CTRL_USP_TX_PRESET_MASK 0x000000F0 + #define PCI_PL_16GT_LE_CTRL_USP_TX_PRESET_SHIFT 4 + ++/* Physical Layer 32.0 GT/s */ ++#define PCI_PL_32GT_LE_CTRL 0x20 /* Lane Equalization Control Register */ ++ ++/* Physical Layer 64.0 GT/s */ ++#define PCI_PL_64GT_LE_CTRL 0x20 /* Lane Equalization Control Register */ ++ + /* Native PCIe Enclosure Management */ + #define PCI_NPEM_CAP 0x04 /* NPEM capability register */ + #define PCI_NPEM_CAP_CAPABLE 0x00000001 /* NPEM Capable */ +diff --git a/include/standard-headers/linux/virtio_gpu.h b/include/standard-headers/linux/virtio_gpu.h +index 6459fdb9fb..00cd3f04af 100644 +--- a/include/standard-headers/linux/virtio_gpu.h ++++ b/include/standard-headers/linux/virtio_gpu.h +@@ -309,8 +309,9 @@ struct virtio_gpu_cmd_submit { + + #define VIRTIO_GPU_CAPSET_VIRGL 1 + #define VIRTIO_GPU_CAPSET_VIRGL2 2 +-/* 3 is reserved for gfxstream */ ++#define VIRTIO_GPU_CAPSET_GFXSTREAM_VULKAN 3 + #define VIRTIO_GPU_CAPSET_VENUS 4 ++#define VIRTIO_GPU_CAPSET_CROSS_DOMAIN 5 + #define VIRTIO_GPU_CAPSET_DRM 6 + + /* VIRTIO_GPU_CMD_GET_CAPSET_INFO */ +diff --git a/include/standard-headers/linux/virtio_pci.h b/include/standard-headers/linux/virtio_pci.h +index 91fec6f502..09e964e6ee 100644 +--- a/include/standard-headers/linux/virtio_pci.h ++++ b/include/standard-headers/linux/virtio_pci.h +@@ -246,6 +246,7 @@ struct virtio_pci_cfg_cap { + #define VIRTIO_ADMIN_CMD_LIST_USE 0x1 + + /* Admin command group type. */ ++#define VIRTIO_ADMIN_GROUP_TYPE_SELF 0x0 + #define VIRTIO_ADMIN_GROUP_TYPE_SRIOV 0x1 + + /* Transitional device admin command. */ +diff --git a/linux-headers/asm-arm64/kvm.h b/linux-headers/asm-arm64/kvm.h +index 4e6aff08df..f4d9baafa1 100644 +--- a/linux-headers/asm-arm64/kvm.h ++++ b/linux-headers/asm-arm64/kvm.h +@@ -419,10 +419,11 @@ enum { + + /* Device Control API on vcpu fd */ + #define KVM_ARM_VCPU_PMU_V3_CTRL 0 +-#define KVM_ARM_VCPU_PMU_V3_IRQ 0 +-#define KVM_ARM_VCPU_PMU_V3_INIT 1 +-#define KVM_ARM_VCPU_PMU_V3_FILTER 2 +-#define KVM_ARM_VCPU_PMU_V3_SET_PMU 3 ++#define KVM_ARM_VCPU_PMU_V3_IRQ 0 ++#define KVM_ARM_VCPU_PMU_V3_INIT 1 ++#define KVM_ARM_VCPU_PMU_V3_FILTER 2 ++#define KVM_ARM_VCPU_PMU_V3_SET_PMU 3 ++#define KVM_ARM_VCPU_PMU_V3_SET_NR_COUNTERS 4 + #define KVM_ARM_VCPU_TIMER_CTRL 1 + #define KVM_ARM_VCPU_TIMER_IRQ_VTIMER 0 + #define KVM_ARM_VCPU_TIMER_IRQ_PTIMER 1 +diff --git a/linux-headers/asm-x86/kvm.h b/linux-headers/asm-x86/kvm.h +index 7fb57ccb2a..cd275ae76d 100644 +--- a/linux-headers/asm-x86/kvm.h ++++ b/linux-headers/asm-x86/kvm.h +@@ -843,6 +843,7 @@ struct kvm_sev_snp_launch_start { + }; + + /* Kept in sync with firmware values for simplicity. */ ++#define KVM_SEV_PAGE_TYPE_INVALID 0x0 + #define KVM_SEV_SNP_PAGE_TYPE_NORMAL 0x1 + #define KVM_SEV_SNP_PAGE_TYPE_ZERO 0x3 + #define KVM_SEV_SNP_PAGE_TYPE_UNMEASURED 0x4 +diff --git a/linux-headers/linux/bits.h b/linux-headers/linux/bits.h +index 58596d18f4..9243f38975 100644 +--- a/linux-headers/linux/bits.h ++++ b/linux-headers/linux/bits.h +@@ -4,9 +4,9 @@ + #ifndef _LINUX_BITS_H + #define _LINUX_BITS_H + +-#define __GENMASK(h, l) (((~_UL(0)) << (l)) & (~_UL(0) >> (BITS_PER_LONG - 1 - (h)))) ++#define __GENMASK(h, l) (((~_UL(0)) << (l)) & (~_UL(0) >> (__BITS_PER_LONG - 1 - (h)))) + +-#define __GENMASK_ULL(h, l) (((~_ULL(0)) << (l)) & (~_ULL(0) >> (BITS_PER_LONG_LONG - 1 - (h)))) ++#define __GENMASK_ULL(h, l) (((~_ULL(0)) << (l)) & (~_ULL(0) >> (__BITS_PER_LONG_LONG - 1 - (h)))) + + #define __GENMASK_U128(h, l) \ + ((_BIT128((h)) << 1) - (_BIT128(l))) +diff --git a/linux-headers/linux/kvm.h b/linux-headers/linux/kvm.h +index 99cc82a275..0690743944 100644 +--- a/linux-headers/linux/kvm.h ++++ b/linux-headers/linux/kvm.h +@@ -178,6 +178,7 @@ struct kvm_xen_exit { + #define KVM_EXIT_NOTIFY 37 + #define KVM_EXIT_LOONGARCH_IOCSR 38 + #define KVM_EXIT_MEMORY_FAULT 39 ++#define KVM_EXIT_TDX 40 + + /* For KVM_EXIT_INTERNAL_ERROR */ + /* Emulate instruction failed. */ +@@ -439,6 +440,27 @@ struct kvm_run { + __u64 gpa; + __u64 size; + } memory_fault; ++ /* KVM_EXIT_TDX */ ++ struct { ++ __u64 flags; ++ __u64 nr; ++ union { ++ struct { ++ __u64 ret; ++ __u64 data[5]; ++ } unknown; ++ struct { ++ __u64 ret; ++ __u64 gpa; ++ __u64 size; ++ } get_quote; ++ struct { ++ __u64 ret; ++ __u64 leaf; ++ __u64 r11, r12, r13, r14; ++ } get_tdvmcall_info; ++ }; ++ } tdx; + /* Fix the size of the union. */ + char padding[256]; + }; +@@ -923,6 +945,9 @@ struct kvm_enable_cap { + #define KVM_CAP_X86_APIC_BUS_CYCLES_NS 237 + #define KVM_CAP_X86_GUEST_MODE 238 + #define KVM_CAP_ARM_WRITABLE_IMP_ID_REGS 239 ++#define KVM_CAP_ARM_EL2 240 ++#define KVM_CAP_ARM_EL2_E2H0 241 ++#define KVM_CAP_RISCV_MP_STATE_RESET 242 + + struct kvm_irq_routing_irqchip { + __u32 irqchip; +diff --git a/linux-headers/linux/vhost.h b/linux-headers/linux/vhost.h +index b95dd84eef..d4b3e2ae13 100644 +--- a/linux-headers/linux/vhost.h ++++ b/linux-headers/linux/vhost.h +@@ -28,10 +28,10 @@ + + /* Set current process as the (exclusive) owner of this file descriptor. This + * must be called before any other vhost command. Further calls to +- * VHOST_OWNER_SET fail until VHOST_OWNER_RESET is called. */ ++ * VHOST_SET_OWNER fail until VHOST_RESET_OWNER is called. */ + #define VHOST_SET_OWNER _IO(VHOST_VIRTIO, 0x01) + /* Give up ownership, and reset the device to default values. +- * Allows subsequent call to VHOST_OWNER_SET to succeed. */ ++ * Allows subsequent call to VHOST_SET_OWNER to succeed. */ + #define VHOST_RESET_OWNER _IO(VHOST_VIRTIO, 0x02) + + /* Set up/modify memory layout */ +-- +2.50.1 + diff --git a/SOURCES/kvm-util-qemu-sockets-Add-support-for-keep-alive-flag-to.patch b/SOURCES/kvm-util-qemu-sockets-Add-support-for-keep-alive-flag-to.patch new file mode 100644 index 0000000..9c53a87 --- /dev/null +++ b/SOURCES/kvm-util-qemu-sockets-Add-support-for-keep-alive-flag-to.patch @@ -0,0 +1,86 @@ +From 644b9e34d8c764598e663eb983e2d6eca4ed2510 Mon Sep 17 00:00:00 2001 +From: Juraj Marcin +Date: Wed, 21 May 2025 15:52:33 +0200 +Subject: [PATCH 11/57] util/qemu-sockets: Add support for keep-alive flag to + passive sockets +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Juraj Marcin +RH-MergeRequest: 369: util/qemu-sockets: Introduce inet socket options controlling TCP keep-alive +RH-Jira: RHEL-67104 +RH-Acked-by: Peter Xu +RH-Acked-by: Miroslav Rezanina +RH-Commit: [4/7] a8ec1996262b2bb657b8fe2e72c9045faee0a64c (JurajMarcin/centos-src-qemu-kvm) + +Commit aec21d3175 (qapi: Add InetSocketAddress member keep-alive) +introduces the keep-alive flag, which enables the SO_KEEPALIVE socket +option, but only on client-side sockets. However, this option is also +useful for server-side sockets, so they can check if a client is still +reachable or drop the connection otherwise. + +This patch enables the SO_KEEPALIVE socket option on passive server-side +sockets if the keep-alive flag is enabled. This socket option is then +inherited by active server-side sockets communicating with connected +clients. + +Signed-off-by: Juraj Marcin +Reviewed-by: Daniel P. Berrangé +Signed-off-by: Daniel P. Berrangé + +(cherry picked from commit 00064705ed1f3943d3634be25da434466c87e7d5) + +JIRA: https://issues.redhat.com/browse/RHEL-67104 + +Signed-off-by: Juraj Marcin +--- + qapi/sockets.json | 4 ++-- + util/qemu-sockets.c | 9 +++------ + 2 files changed, 5 insertions(+), 8 deletions(-) + +diff --git a/qapi/sockets.json b/qapi/sockets.json +index 6a95023315..62797cd027 100644 +--- a/qapi/sockets.json ++++ b/qapi/sockets.json +@@ -56,8 +56,8 @@ + # @ipv6: whether to accept IPv6 addresses, default try both IPv4 and + # IPv6 + # +-# @keep-alive: enable keep-alive when connecting to this socket. Not +-# supported for passive sockets. (Since 4.2) ++# @keep-alive: enable keep-alive when connecting to/listening on this socket. ++# (Since 4.2, not supported for listening sockets until 10.1) + # + # @mptcp: enable multi-path TCP. (Since 6.1) + # +diff --git a/util/qemu-sockets.c b/util/qemu-sockets.c +index 631d0c4023..8fc1f86145 100644 +--- a/util/qemu-sockets.c ++++ b/util/qemu-sockets.c +@@ -236,12 +236,6 @@ static int inet_listen_saddr(InetSocketAddress *saddr, + int saved_errno = 0; + bool socket_created = false; + +- if (saddr->keep_alive) { +- error_setg(errp, "keep-alive option is not supported for passive " +- "sockets"); +- return -1; +- } +- + memset(&ai,0, sizeof(ai)); + ai.ai_flags = AI_PASSIVE; + if (saddr->has_numeric && saddr->numeric) { +@@ -349,6 +343,9 @@ static int inet_listen_saddr(InetSocketAddress *saddr, + goto fail; + } + /* We have a listening socket */ ++ if (inet_set_sockopts(slisten, saddr, errp) < 0) { ++ goto fail; ++ } + freeaddrinfo(res); + return slisten; + } +-- +2.39.3 + diff --git a/SOURCES/kvm-util-qemu-sockets-Introduce-inet-socket-options-cont.patch b/SOURCES/kvm-util-qemu-sockets-Introduce-inet-socket-options-cont.patch new file mode 100644 index 0000000..8839085 --- /dev/null +++ b/SOURCES/kvm-util-qemu-sockets-Introduce-inet-socket-options-cont.patch @@ -0,0 +1,314 @@ +From 3e9458cd71f909474c1dd051f43fd3fbef8d53fd Mon Sep 17 00:00:00 2001 +From: Juraj Marcin +Date: Wed, 21 May 2025 15:52:35 +0200 +Subject: [PATCH 13/57] util/qemu-sockets: Introduce inet socket options + controlling TCP keep-alive +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Juraj Marcin +RH-MergeRequest: 369: util/qemu-sockets: Introduce inet socket options controlling TCP keep-alive +RH-Jira: RHEL-67104 +RH-Acked-by: Peter Xu +RH-Acked-by: Miroslav Rezanina +RH-Commit: [6/7] 3861e7874d5952c53a5020c123b3b2e632149008 (JurajMarcin/centos-src-qemu-kvm) + +With the default TCP stack configuration, it could be even 2 hours +before the connection times out due to the other side not being +reachable. However, in some cases, the application needs to be aware of +a connection issue much sooner. + +This is the case, for example, for postcopy live migration. If there is +no traffic from the migration destination guest (server-side) to the +migration source guest (client-side), the destination keeps waiting for +pages indefinitely and does not switch to the postcopy-paused state. +This can happen, for example, if the destination QEMU instance is +started with the '-S' command line option and the machine is not started +yet, or if the machine is idle and produces no new page faults for +not-yet-migrated pages. + +This patch introduces new inet socket parameters that control count, +idle period, and interval of TCP keep-alive packets before the +connection is considered broken. These parameters are available on +systems where the respective TCP socket options are defined, that +includes Linux, Windows, macOS, but not OpenBSD. Additionally, macOS +defines TCP_KEEPIDLE as TCP_KEEPALIVE instead, so the patch supplies its +own definition. + +The default value for all is 0, which means the system configuration is +used. + +Signed-off-by: Juraj Marcin +Reviewed-by: Daniel P. Berrangé +Signed-off-by: Daniel P. Berrangé + +(cherry picked from commit 1bd4237cb1095d71c16afad3ce93b4a1e453173e) + +JIRA: https://issues.redhat.com/browse/RHEL-67104 + +Signed-off-by: Juraj Marcin +--- + meson.build | 30 +++++++++++++ + qapi/sockets.json | 19 ++++++++ + tests/unit/test-util-sockets.c | 39 +++++++++++++++++ + util/qemu-sockets.c | 80 ++++++++++++++++++++++++++++++++++ + 4 files changed, 168 insertions(+) + +diff --git a/meson.build b/meson.build +index 5bb2b757c3..c4539b66c5 100644 +--- a/meson.build ++++ b/meson.build +@@ -2581,6 +2581,36 @@ config_host_data.set('HAVE_OPTRESET', + cc.has_header_symbol('getopt.h', 'optreset')) + config_host_data.set('HAVE_IPPROTO_MPTCP', + cc.has_header_symbol('netinet/in.h', 'IPPROTO_MPTCP')) ++config_host_data.set('HAVE_TCP_KEEPCNT', ++ cc.has_header_symbol('netinet/tcp.h', 'TCP_KEEPCNT') or ++ cc.compiles(''' ++ #include ++ #ifndef TCP_KEEPCNT ++ #error ++ #endif ++ int main(void) { return 0; }''', ++ name: 'Win32 TCP_KEEPCNT')) ++# On Darwin TCP_KEEPIDLE is available under different name, TCP_KEEPALIVE. ++# https://github.com/apple/darwin-xnu/blob/xnu-4570.1.46/bsd/man/man4/tcp.4#L172 ++config_host_data.set('HAVE_TCP_KEEPIDLE', ++ cc.has_header_symbol('netinet/tcp.h', 'TCP_KEEPIDLE') or ++ cc.has_header_symbol('netinet/tcp.h', 'TCP_KEEPALIVE') or ++ cc.compiles(''' ++ #include ++ #ifndef TCP_KEEPIDLE ++ #error ++ #endif ++ int main(void) { return 0; }''', ++ name: 'Win32 TCP_KEEPIDLE')) ++config_host_data.set('HAVE_TCP_KEEPINTVL', ++ cc.has_header_symbol('netinet/tcp.h', 'TCP_KEEPINTVL') or ++ cc.compiles(''' ++ #include ++ #ifndef TCP_KEEPINTVL ++ #error ++ #endif ++ int main(void) { return 0; }''', ++ name: 'Win32 TCP_KEEPINTVL')) + + # has_member + config_host_data.set('HAVE_SIGEV_NOTIFY_THREAD_ID', +diff --git a/qapi/sockets.json b/qapi/sockets.json +index 62797cd027..f9f559daba 100644 +--- a/qapi/sockets.json ++++ b/qapi/sockets.json +@@ -59,6 +59,22 @@ + # @keep-alive: enable keep-alive when connecting to/listening on this socket. + # (Since 4.2, not supported for listening sockets until 10.1) + # ++# @keep-alive-count: number of keep-alive packets sent before the connection is ++# closed. Only supported for TCP sockets on systems where TCP_KEEPCNT ++# socket option is defined (this includes Linux, Windows, macOS, FreeBSD, ++# but not OpenBSD). When set to 0, system setting is used. (Since 10.1) ++# ++# @keep-alive-idle: time in seconds the connection needs to be idle before ++# sending a keepalive packet. Only supported for TCP sockets on systems ++# where TCP_KEEPIDLE socket option is defined (this includes Linux, ++# Windows, macOS, FreeBSD, but not OpenBSD). When set to 0, system setting ++# is used. (Since 10.1) ++# ++# @keep-alive-interval: time in seconds between keep-alive packets. Only ++# supported for TCP sockets on systems where TCP_KEEPINTVL is defined (this ++# includes Linux, Windows, macOS, FreeBSD, but not OpenBSD). When set to ++# 0, system setting is used. (Since 10.1) ++# + # @mptcp: enable multi-path TCP. (Since 6.1) + # + # Since: 1.3 +@@ -71,6 +87,9 @@ + '*ipv4': 'bool', + '*ipv6': 'bool', + '*keep-alive': 'bool', ++ '*keep-alive-count': { 'type': 'uint32', 'if': 'HAVE_TCP_KEEPCNT' }, ++ '*keep-alive-idle': { 'type': 'uint32', 'if': 'HAVE_TCP_KEEPIDLE' }, ++ '*keep-alive-interval': { 'type': 'uint32', 'if': 'HAVE_TCP_KEEPINTVL' }, + '*mptcp': { 'type': 'bool', 'if': 'HAVE_IPPROTO_MPTCP' } } } + + ## +diff --git a/tests/unit/test-util-sockets.c b/tests/unit/test-util-sockets.c +index 9e39b92e7c..8492f4d68f 100644 +--- a/tests/unit/test-util-sockets.c ++++ b/tests/unit/test-util-sockets.c +@@ -359,6 +359,24 @@ static void inet_parse_test_helper(const char *str, + g_assert_cmpint(addr.ipv6, ==, exp_addr->ipv6); + g_assert_cmpint(addr.has_keep_alive, ==, exp_addr->has_keep_alive); + g_assert_cmpint(addr.keep_alive, ==, exp_addr->keep_alive); ++#ifdef HAVE_TCP_KEEPCNT ++ g_assert_cmpint(addr.has_keep_alive_count, ==, ++ exp_addr->has_keep_alive_count); ++ g_assert_cmpint(addr.keep_alive_count, ==, ++ exp_addr->keep_alive_count); ++#endif ++#ifdef HAVE_TCP_KEEPIDLE ++ g_assert_cmpint(addr.has_keep_alive_idle, ==, ++ exp_addr->has_keep_alive_idle); ++ g_assert_cmpint(addr.keep_alive_idle, ==, ++ exp_addr->keep_alive_idle); ++#endif ++#ifdef HAVE_TCP_KEEPINTVL ++ g_assert_cmpint(addr.has_keep_alive_interval, ==, ++ exp_addr->has_keep_alive_interval); ++ g_assert_cmpint(addr.keep_alive_interval, ==, ++ exp_addr->keep_alive_interval); ++#endif + #ifdef HAVE_IPPROTO_MPTCP + g_assert_cmpint(addr.has_mptcp, ==, exp_addr->has_mptcp); + g_assert_cmpint(addr.mptcp, ==, exp_addr->mptcp); +@@ -460,6 +478,18 @@ static void test_inet_parse_all_options_good(void) + .ipv6 = true, + .has_keep_alive = true, + .keep_alive = true, ++#ifdef HAVE_TCP_KEEPCNT ++ .has_keep_alive_count = true, ++ .keep_alive_count = 10, ++#endif ++#ifdef HAVE_TCP_KEEPIDLE ++ .has_keep_alive_idle = true, ++ .keep_alive_idle = 60, ++#endif ++#ifdef HAVE_TCP_KEEPINTVL ++ .has_keep_alive_interval = true, ++ .keep_alive_interval = 30, ++#endif + #ifdef HAVE_IPPROTO_MPTCP + .has_mptcp = true, + .mptcp = false, +@@ -467,6 +497,15 @@ static void test_inet_parse_all_options_good(void) + }; + inet_parse_test_helper( + "[::1]:5000,numeric=on,to=5006,ipv4=off,ipv6=on,keep-alive=on" ++#ifdef HAVE_TCP_KEEPCNT ++ ",keep-alive-count=10" ++#endif ++#ifdef HAVE_TCP_KEEPIDLE ++ ",keep-alive-idle=60" ++#endif ++#ifdef HAVE_TCP_KEEPINTVL ++ ",keep-alive-interval=30" ++#endif + #ifdef HAVE_IPPROTO_MPTCP + ",mptcp=off" + #endif +diff --git a/util/qemu-sockets.c b/util/qemu-sockets.c +index 8017124c74..ef0a137bd2 100644 +--- a/util/qemu-sockets.c ++++ b/util/qemu-sockets.c +@@ -45,6 +45,14 @@ + # define AI_NUMERICSERV 0 + #endif + ++/* ++ * On macOS TCP_KEEPIDLE is available under a different name, TCP_KEEPALIVE. ++ * https://github.com/apple/darwin-xnu/blob/xnu-4570.1.46/bsd/man/man4/tcp.4#L172 ++ */ ++#if defined(TCP_KEEPALIVE) && !defined(TCP_KEEPIDLE) ++# define TCP_KEEPIDLE TCP_KEEPALIVE ++#endif ++ + + static int inet_getport(struct addrinfo *e) + { +@@ -218,6 +226,42 @@ static int inet_set_sockopts(int sock, InetSocketAddress *saddr, Error **errp) + "Unable to set keep-alive option on socket"); + return -1; + } ++#ifdef HAVE_TCP_KEEPCNT ++ if (saddr->has_keep_alive_count && saddr->keep_alive_count) { ++ int keep_count = saddr->keep_alive_count; ++ ret = setsockopt(sock, IPPROTO_TCP, TCP_KEEPCNT, &keep_count, ++ sizeof(keep_count)); ++ if (ret < 0) { ++ error_setg_errno(errp, errno, ++ "Unable to set TCP keep-alive count option on socket"); ++ return -1; ++ } ++ } ++#endif ++#ifdef HAVE_TCP_KEEPIDLE ++ if (saddr->has_keep_alive_idle && saddr->keep_alive_idle) { ++ int keep_idle = saddr->keep_alive_idle; ++ ret = setsockopt(sock, IPPROTO_TCP, TCP_KEEPIDLE, &keep_idle, ++ sizeof(keep_idle)); ++ if (ret < 0) { ++ error_setg_errno(errp, errno, ++ "Unable to set TCP keep-alive idle option on socket"); ++ return -1; ++ } ++ } ++#endif ++#ifdef HAVE_TCP_KEEPINTVL ++ if (saddr->has_keep_alive_interval && saddr->keep_alive_interval) { ++ int keep_interval = saddr->keep_alive_interval; ++ ret = setsockopt(sock, IPPROTO_TCP, TCP_KEEPINTVL, &keep_interval, ++ sizeof(keep_interval)); ++ if (ret < 0) { ++ error_setg_errno(errp, errno, ++ "Unable to set TCP keep-alive interval option on socket"); ++ return -1; ++ } ++ } ++#endif + } + return 0; + } +@@ -631,6 +675,24 @@ static QemuOptsList inet_opts = { + .name = "keep-alive", + .type = QEMU_OPT_BOOL, + }, ++#ifdef HAVE_TCP_KEEPCNT ++ { ++ .name = "keep-alive-count", ++ .type = QEMU_OPT_NUMBER, ++ }, ++#endif ++#ifdef HAVE_TCP_KEEPIDLE ++ { ++ .name = "keep-alive-idle", ++ .type = QEMU_OPT_NUMBER, ++ }, ++#endif ++#ifdef HAVE_TCP_KEEPINTVL ++ { ++ .name = "keep-alive-interval", ++ .type = QEMU_OPT_NUMBER, ++ }, ++#endif + #ifdef HAVE_IPPROTO_MPTCP + { + .name = "mptcp", +@@ -696,6 +758,24 @@ int inet_parse(InetSocketAddress *addr, const char *str, Error **errp) + addr->has_keep_alive = true; + addr->keep_alive = qemu_opt_get_bool(opts, "keep-alive", false); + } ++#ifdef HAVE_TCP_KEEPCNT ++ if (qemu_opt_find(opts, "keep-alive-count")) { ++ addr->has_keep_alive_count = true; ++ addr->keep_alive_count = qemu_opt_get_number(opts, "keep-alive-count", 0); ++ } ++#endif ++#ifdef HAVE_TCP_KEEPIDLE ++ if (qemu_opt_find(opts, "keep-alive-idle")) { ++ addr->has_keep_alive_idle = true; ++ addr->keep_alive_idle = qemu_opt_get_number(opts, "keep-alive-idle", 0); ++ } ++#endif ++#ifdef HAVE_TCP_KEEPINTVL ++ if (qemu_opt_find(opts, "keep-alive-interval")) { ++ addr->has_keep_alive_interval = true; ++ addr->keep_alive_interval = qemu_opt_get_number(opts, "keep-alive-interval", 0); ++ } ++#endif + #ifdef HAVE_IPPROTO_MPTCP + if (qemu_opt_find(opts, "mptcp")) { + addr->has_mptcp = true; +-- +2.39.3 + diff --git a/SOURCES/kvm-util-qemu-sockets-Refactor-inet_parse-to-use-QemuOpt.patch b/SOURCES/kvm-util-qemu-sockets-Refactor-inet_parse-to-use-QemuOpt.patch new file mode 100644 index 0000000..e533fa2 --- /dev/null +++ b/SOURCES/kvm-util-qemu-sockets-Refactor-inet_parse-to-use-QemuOpt.patch @@ -0,0 +1,461 @@ +From d2155b6fe1f200a588dd5e95d7587e96109da989 Mon Sep 17 00:00:00 2001 +From: Juraj Marcin +Date: Wed, 21 May 2025 15:52:34 +0200 +Subject: [PATCH 12/57] util/qemu-sockets: Refactor inet_parse() to use + QemuOpts +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Juraj Marcin +RH-MergeRequest: 369: util/qemu-sockets: Introduce inet socket options controlling TCP keep-alive +RH-Jira: RHEL-67104 +RH-Acked-by: Peter Xu +RH-Acked-by: Miroslav Rezanina +RH-Commit: [5/7] 7fd97ed112c6259928d94595c2ae23fe1208621e (JurajMarcin/centos-src-qemu-kvm) + +Currently, the inet address parser cannot handle multiple options where +one is prefixed with the name of the other. For example, with the +'keep-alive-idle' option added, the current parser cannot parse +'127.0.0.1:5000,keep-alive-idle=60,keep-alive' correctly. Instead, it +fails with "error parsing 'keep-alive' flag '-idle=60,keep-alive'". + +To resolve these issues, this patch rewrites the inet address parsing +using the QemuOpts parser, which the inet_parse_flag() function tries to +mimic. This new parser supports all previously supported options and on +top of that the 'numeric' flag is now also supported. The only +difference is, the new parser produces an error if an unknown option is +passed, instead of silently ignoring it. + +Signed-off-by: Juraj Marcin +Reviewed-by: Daniel P. Berrangé +Signed-off-by: Daniel P. Berrangé + +(cherry picked from commit 316e8ee8d614f049bfae697570a5e62af450491c) + +JIRA: https://issues.redhat.com/browse/RHEL-67104 + +Signed-off-by: Juraj Marcin +--- + tests/unit/test-util-sockets.c | 196 +++++++++++++++++++++++++++++++++ + util/qemu-sockets.c | 158 +++++++++++++------------- + 2 files changed, 270 insertions(+), 84 deletions(-) + +diff --git a/tests/unit/test-util-sockets.c b/tests/unit/test-util-sockets.c +index 4c9dd0b271..9e39b92e7c 100644 +--- a/tests/unit/test-util-sockets.c ++++ b/tests/unit/test-util-sockets.c +@@ -332,6 +332,177 @@ static void test_socket_unix_abstract(void) + + #endif /* CONFIG_LINUX */ + ++static void inet_parse_test_helper(const char *str, ++ InetSocketAddress *exp_addr, bool success) ++{ ++ InetSocketAddress addr; ++ Error *error = NULL; ++ ++ int rc = inet_parse(&addr, str, &error); ++ ++ if (success) { ++ g_assert_cmpint(rc, ==, 0); ++ } else { ++ g_assert_cmpint(rc, <, 0); ++ } ++ if (exp_addr != NULL) { ++ g_assert_cmpstr(addr.host, ==, exp_addr->host); ++ g_assert_cmpstr(addr.port, ==, exp_addr->port); ++ /* Own members: */ ++ g_assert_cmpint(addr.has_numeric, ==, exp_addr->has_numeric); ++ g_assert_cmpint(addr.numeric, ==, exp_addr->numeric); ++ g_assert_cmpint(addr.has_to, ==, exp_addr->has_to); ++ g_assert_cmpint(addr.to, ==, exp_addr->to); ++ g_assert_cmpint(addr.has_ipv4, ==, exp_addr->has_ipv4); ++ g_assert_cmpint(addr.ipv4, ==, exp_addr->ipv4); ++ g_assert_cmpint(addr.has_ipv6, ==, exp_addr->has_ipv6); ++ g_assert_cmpint(addr.ipv6, ==, exp_addr->ipv6); ++ g_assert_cmpint(addr.has_keep_alive, ==, exp_addr->has_keep_alive); ++ g_assert_cmpint(addr.keep_alive, ==, exp_addr->keep_alive); ++#ifdef HAVE_IPPROTO_MPTCP ++ g_assert_cmpint(addr.has_mptcp, ==, exp_addr->has_mptcp); ++ g_assert_cmpint(addr.mptcp, ==, exp_addr->mptcp); ++#endif ++ } ++ ++ g_free(addr.host); ++ g_free(addr.port); ++} ++ ++static void test_inet_parse_nohost_good(void) ++{ ++ char host[] = ""; ++ char port[] = "5000"; ++ InetSocketAddress exp_addr = { ++ .host = host, ++ .port = port, ++ }; ++ inet_parse_test_helper(":5000", &exp_addr, true); ++} ++ ++static void test_inet_parse_empty_bad(void) ++{ ++ inet_parse_test_helper("", NULL, false); ++} ++ ++static void test_inet_parse_only_colon_bad(void) ++{ ++ inet_parse_test_helper(":", NULL, false); ++} ++ ++static void test_inet_parse_ipv4_good(void) ++{ ++ char host[] = "127.0.0.1"; ++ char port[] = "5000"; ++ InetSocketAddress exp_addr = { ++ .host = host, ++ .port = port, ++ }; ++ inet_parse_test_helper("127.0.0.1:5000", &exp_addr, true); ++} ++ ++static void test_inet_parse_ipv4_noport_bad(void) ++{ ++ inet_parse_test_helper("127.0.0.1", NULL, false); ++} ++ ++static void test_inet_parse_ipv6_good(void) ++{ ++ char host[] = "::1"; ++ char port[] = "5000"; ++ InetSocketAddress exp_addr = { ++ .host = host, ++ .port = port, ++ }; ++ inet_parse_test_helper("[::1]:5000", &exp_addr, true); ++} ++ ++static void test_inet_parse_ipv6_noend_bad(void) ++{ ++ inet_parse_test_helper("[::1", NULL, false); ++} ++ ++static void test_inet_parse_ipv6_noport_bad(void) ++{ ++ inet_parse_test_helper("[::1]:", NULL, false); ++} ++ ++static void test_inet_parse_ipv6_empty_bad(void) ++{ ++ inet_parse_test_helper("[]:5000", NULL, false); ++} ++ ++static void test_inet_parse_hostname_good(void) ++{ ++ char host[] = "localhost"; ++ char port[] = "5000"; ++ InetSocketAddress exp_addr = { ++ .host = host, ++ .port = port, ++ }; ++ inet_parse_test_helper("localhost:5000", &exp_addr, true); ++} ++ ++static void test_inet_parse_all_options_good(void) ++{ ++ char host[] = "::1"; ++ char port[] = "5000"; ++ InetSocketAddress exp_addr = { ++ .host = host, ++ .port = port, ++ .has_numeric = true, ++ .numeric = true, ++ .has_to = true, ++ .to = 5006, ++ .has_ipv4 = true, ++ .ipv4 = false, ++ .has_ipv6 = true, ++ .ipv6 = true, ++ .has_keep_alive = true, ++ .keep_alive = true, ++#ifdef HAVE_IPPROTO_MPTCP ++ .has_mptcp = true, ++ .mptcp = false, ++#endif ++ }; ++ inet_parse_test_helper( ++ "[::1]:5000,numeric=on,to=5006,ipv4=off,ipv6=on,keep-alive=on" ++#ifdef HAVE_IPPROTO_MPTCP ++ ",mptcp=off" ++#endif ++ , &exp_addr, true); ++} ++ ++static void test_inet_parse_all_implicit_bool_good(void) ++{ ++ char host[] = "::1"; ++ char port[] = "5000"; ++ InetSocketAddress exp_addr = { ++ .host = host, ++ .port = port, ++ .has_numeric = true, ++ .numeric = true, ++ .has_to = true, ++ .to = 5006, ++ .has_ipv4 = true, ++ .ipv4 = true, ++ .has_ipv6 = true, ++ .ipv6 = true, ++ .has_keep_alive = true, ++ .keep_alive = true, ++#ifdef HAVE_IPPROTO_MPTCP ++ .has_mptcp = true, ++ .mptcp = true, ++#endif ++ }; ++ inet_parse_test_helper( ++ "[::1]:5000,numeric,to=5006,ipv4,ipv6,keep-alive" ++#ifdef HAVE_IPPROTO_MPTCP ++ ",mptcp" ++#endif ++ , &exp_addr, true); ++} ++ + int main(int argc, char **argv) + { + bool has_ipv4, has_ipv6; +@@ -377,6 +548,31 @@ int main(int argc, char **argv) + test_socket_unix_abstract); + #endif + ++ g_test_add_func("/util/socket/inet-parse/nohost-good", ++ test_inet_parse_nohost_good); ++ g_test_add_func("/util/socket/inet-parse/empty-bad", ++ test_inet_parse_empty_bad); ++ g_test_add_func("/util/socket/inet-parse/only-colon-bad", ++ test_inet_parse_only_colon_bad); ++ g_test_add_func("/util/socket/inet-parse/ipv4-good", ++ test_inet_parse_ipv4_good); ++ g_test_add_func("/util/socket/inet-parse/ipv4-noport-bad", ++ test_inet_parse_ipv4_noport_bad); ++ g_test_add_func("/util/socket/inet-parse/ipv6-good", ++ test_inet_parse_ipv6_good); ++ g_test_add_func("/util/socket/inet-parse/ipv6-noend-bad", ++ test_inet_parse_ipv6_noend_bad); ++ g_test_add_func("/util/socket/inet-parse/ipv6-noport-bad", ++ test_inet_parse_ipv6_noport_bad); ++ g_test_add_func("/util/socket/inet-parse/ipv6-empty-bad", ++ test_inet_parse_ipv6_empty_bad); ++ g_test_add_func("/util/socket/inet-parse/hostname-good", ++ test_inet_parse_hostname_good); ++ g_test_add_func("/util/socket/inet-parse/all-options-good", ++ test_inet_parse_all_options_good); ++ g_test_add_func("/util/socket/inet-parse/all-bare-bool-good", ++ test_inet_parse_all_implicit_bool_good); ++ + end: + return g_test_run(); + } +diff --git a/util/qemu-sockets.c b/util/qemu-sockets.c +index 8fc1f86145..8017124c74 100644 +--- a/util/qemu-sockets.c ++++ b/util/qemu-sockets.c +@@ -30,6 +30,7 @@ + #include "qapi/qobject-input-visitor.h" + #include "qapi/qobject-output-visitor.h" + #include "qemu/cutils.h" ++#include "qemu/option.h" + #include "trace.h" + + #ifndef AI_ADDRCONFIG +@@ -601,115 +602,104 @@ err: + return -1; + } + +-/* compatibility wrapper */ +-static int inet_parse_flag(const char *flagname, const char *optstr, bool *val, +- Error **errp) +-{ +- char *end; +- size_t len; +- +- end = strstr(optstr, ","); +- if (end) { +- if (end[1] == ',') { /* Reject 'ipv6=on,,foo' */ +- error_setg(errp, "error parsing '%s' flag '%s'", flagname, optstr); +- return -1; +- } +- len = end - optstr; +- } else { +- len = strlen(optstr); +- } +- if (len == 0 || (len == 3 && strncmp(optstr, "=on", len) == 0)) { +- *val = true; +- } else if (len == 4 && strncmp(optstr, "=off", len) == 0) { +- *val = false; +- } else { +- error_setg(errp, "error parsing '%s' flag '%s'", flagname, optstr); +- return -1; +- } +- return 0; +-} ++static QemuOptsList inet_opts = { ++ .name = "InetSocketAddress", ++ .head = QTAILQ_HEAD_INITIALIZER(inet_opts.head), ++ .implied_opt_name = "addr", ++ .desc = { ++ { ++ .name = "addr", ++ .type = QEMU_OPT_STRING, ++ }, ++ { ++ .name = "numeric", ++ .type = QEMU_OPT_BOOL, ++ }, ++ { ++ .name = "to", ++ .type = QEMU_OPT_NUMBER, ++ }, ++ { ++ .name = "ipv4", ++ .type = QEMU_OPT_BOOL, ++ }, ++ { ++ .name = "ipv6", ++ .type = QEMU_OPT_BOOL, ++ }, ++ { ++ .name = "keep-alive", ++ .type = QEMU_OPT_BOOL, ++ }, ++#ifdef HAVE_IPPROTO_MPTCP ++ { ++ .name = "mptcp", ++ .type = QEMU_OPT_BOOL, ++ }, ++#endif ++ { /* end of list */ } ++ }, ++}; + + int inet_parse(InetSocketAddress *addr, const char *str, Error **errp) + { +- const char *optstr, *h; +- char host[65]; +- char port[33]; +- int to; +- int pos; +- char *begin; +- ++ QemuOpts *opts = qemu_opts_parse(&inet_opts, str, true, errp); ++ if (!opts) { ++ return -1; ++ } + memset(addr, 0, sizeof(*addr)); + + /* parse address */ +- if (str[0] == ':') { +- /* no host given */ +- host[0] = '\0'; +- if (sscanf(str, ":%32[^,]%n", port, &pos) != 1) { +- error_setg(errp, "error parsing port in address '%s'", str); +- return -1; +- } +- } else if (str[0] == '[') { ++ const char *addr_str = qemu_opt_get(opts, "addr"); ++ if (!addr_str) { ++ error_setg(errp, "error parsing address ''"); ++ return -1; ++ } ++ if (str[0] == '[') { + /* IPv6 addr */ +- if (sscanf(str, "[%64[^]]]:%32[^,]%n", host, port, &pos) != 2) { +- error_setg(errp, "error parsing IPv6 address '%s'", str); ++ const char *ip_end = strstr(addr_str, "]:"); ++ if (!ip_end || ip_end - addr_str < 2 || strlen(ip_end) < 3) { ++ error_setg(errp, "error parsing IPv6 address '%s'", addr_str); + return -1; + } ++ addr->host = g_strndup(addr_str + 1, ip_end - addr_str - 1); ++ addr->port = g_strdup(ip_end + 2); + } else { +- /* hostname or IPv4 addr */ +- if (sscanf(str, "%64[^:]:%32[^,]%n", host, port, &pos) != 2) { +- error_setg(errp, "error parsing address '%s'", str); ++ /* no host, hostname or IPv4 addr */ ++ const char *port = strchr(addr_str, ':'); ++ if (!port || strlen(port) < 2) { ++ error_setg(errp, "error parsing address '%s'", addr_str); + return -1; + } ++ addr->host = g_strndup(addr_str, port - addr_str); ++ addr->port = g_strdup(port + 1); + } + +- addr->host = g_strdup(host); +- addr->port = g_strdup(port); +- + /* parse options */ +- optstr = str + pos; +- h = strstr(optstr, ",to="); +- if (h) { +- h += 4; +- if (sscanf(h, "%d%n", &to, &pos) != 1 || +- (h[pos] != '\0' && h[pos] != ',')) { +- error_setg(errp, "error parsing to= argument"); +- return -1; +- } ++ if (qemu_opt_find(opts, "numeric")) { ++ addr->has_numeric = true, ++ addr->numeric = qemu_opt_get_bool(opts, "numeric", false); ++ } ++ if (qemu_opt_find(opts, "to")) { + addr->has_to = true; +- addr->to = to; ++ addr->to = qemu_opt_get_number(opts, "to", 0); + } +- begin = strstr(optstr, ",ipv4"); +- if (begin) { +- if (inet_parse_flag("ipv4", begin + 5, &addr->ipv4, errp) < 0) { +- return -1; +- } ++ if (qemu_opt_find(opts, "ipv4")) { + addr->has_ipv4 = true; ++ addr->ipv4 = qemu_opt_get_bool(opts, "ipv4", false); + } +- begin = strstr(optstr, ",ipv6"); +- if (begin) { +- if (inet_parse_flag("ipv6", begin + 5, &addr->ipv6, errp) < 0) { +- return -1; +- } ++ if (qemu_opt_find(opts, "ipv6")) { + addr->has_ipv6 = true; ++ addr->ipv6 = qemu_opt_get_bool(opts, "ipv6", false); + } +- begin = strstr(optstr, ",keep-alive"); +- if (begin) { +- if (inet_parse_flag("keep-alive", begin + strlen(",keep-alive"), +- &addr->keep_alive, errp) < 0) +- { +- return -1; +- } ++ if (qemu_opt_find(opts, "keep-alive")) { + addr->has_keep_alive = true; ++ addr->keep_alive = qemu_opt_get_bool(opts, "keep-alive", false); + } + #ifdef HAVE_IPPROTO_MPTCP +- begin = strstr(optstr, ",mptcp"); +- if (begin) { +- if (inet_parse_flag("mptcp", begin + strlen(",mptcp"), +- &addr->mptcp, errp) < 0) +- { +- return -1; +- } ++ if (qemu_opt_find(opts, "mptcp")) { + addr->has_mptcp = true; ++ addr->mptcp = qemu_opt_get_bool(opts, "mptcp", 0); + } + #endif + return 0; +-- +2.39.3 + diff --git a/SOURCES/kvm-util-qemu-sockets-Refactor-setting-client-sockopts-i.patch b/SOURCES/kvm-util-qemu-sockets-Refactor-setting-client-sockopts-i.patch new file mode 100644 index 0000000..6b95ffb --- /dev/null +++ b/SOURCES/kvm-util-qemu-sockets-Refactor-setting-client-sockopts-i.patch @@ -0,0 +1,83 @@ +From cc7fbd3aabe3be1e2966472a151dc618be02ac4c Mon Sep 17 00:00:00 2001 +From: Juraj Marcin +Date: Wed, 21 May 2025 15:52:31 +0200 +Subject: [PATCH 09/57] util/qemu-sockets: Refactor setting client sockopts + into a separate function +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Juraj Marcin +RH-MergeRequest: 369: util/qemu-sockets: Introduce inet socket options controlling TCP keep-alive +RH-Jira: RHEL-67104 +RH-Acked-by: Peter Xu +RH-Acked-by: Miroslav Rezanina +RH-Commit: [2/7] b9b258d9a21a31ead1dd4488f1f55f1728fa8b41 (JurajMarcin/centos-src-qemu-kvm) + +This is done in preparation for enabling the SO_KEEPALIVE support for +server sockets and adding settings for more TCP keep-alive socket +options. + +Signed-off-by: Juraj Marcin +Reviewed-by: Daniel P. Berrangé +Signed-off-by: Daniel P. Berrangé + +(cherry picked from commit b8b5278aca78be4a1c2e7cbb11c6be176f63706d) + +JIRA: https://issues.redhat.com/browse/RHEL-67104 + +Signed-off-by: Juraj Marcin +--- + util/qemu-sockets.c | 29 +++++++++++++++++++---------- + 1 file changed, 19 insertions(+), 10 deletions(-) + +diff --git a/util/qemu-sockets.c b/util/qemu-sockets.c +index 60c44b2b56..2c0e4883ce 100644 +--- a/util/qemu-sockets.c ++++ b/util/qemu-sockets.c +@@ -205,6 +205,22 @@ static int try_bind(int socket, InetSocketAddress *saddr, struct addrinfo *e) + #endif + } + ++static int inet_set_sockopts(int sock, InetSocketAddress *saddr, Error **errp) ++{ ++ if (saddr->keep_alive) { ++ int keep_alive = 1; ++ int ret = setsockopt(sock, SOL_SOCKET, SO_KEEPALIVE, ++ &keep_alive, sizeof(keep_alive)); ++ ++ if (ret < 0) { ++ error_setg_errno(errp, errno, ++ "Unable to set keep-alive option on socket"); ++ return -1; ++ } ++ } ++ return 0; ++} ++ + static int inet_listen_saddr(InetSocketAddress *saddr, + int port_offset, + int num, +@@ -476,16 +492,9 @@ int inet_connect_saddr(InetSocketAddress *saddr, Error **errp) + return sock; + } + +- if (saddr->keep_alive) { +- int val = 1; +- int ret = setsockopt(sock, SOL_SOCKET, SO_KEEPALIVE, +- &val, sizeof(val)); +- +- if (ret < 0) { +- error_setg_errno(errp, errno, "Unable to set KEEPALIVE"); +- close(sock); +- return -1; +- } ++ if (inet_set_sockopts(sock, saddr, errp) < 0) { ++ close(sock); ++ return -1; + } + + return sock; +-- +2.39.3 + diff --git a/SOURCES/kvm-util-qemu-sockets-Refactor-success-and-failure-paths.patch b/SOURCES/kvm-util-qemu-sockets-Refactor-success-and-failure-paths.patch new file mode 100644 index 0000000..de5c984 --- /dev/null +++ b/SOURCES/kvm-util-qemu-sockets-Refactor-success-and-failure-paths.patch @@ -0,0 +1,141 @@ +From 17af9176a1fff5af0d4c1bf1beb52228f7fc324d Mon Sep 17 00:00:00 2001 +From: Juraj Marcin +Date: Wed, 21 May 2025 15:52:32 +0200 +Subject: [PATCH 10/57] util/qemu-sockets: Refactor success and failure paths + in inet_listen_saddr() +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Juraj Marcin +RH-MergeRequest: 369: util/qemu-sockets: Introduce inet socket options controlling TCP keep-alive +RH-Jira: RHEL-67104 +RH-Acked-by: Peter Xu +RH-Acked-by: Miroslav Rezanina +RH-Commit: [3/7] 5d5f0e704cdf3454a3c402ea7b130aa42705eb9f (JurajMarcin/centos-src-qemu-kvm) + +To get a listening socket, we need to first create a socket, try binding +it to a certain port, and lastly starting listening to it. Each of these +operations can fail due to various reasons, one of them being that the +requested address/port is already in use. In such case, the function +tries the same process with a new port number. + +This patch refactors the port number loop, so the success path is no +longer buried inside the 'if' statements in the middle of the loop. Now, +the success path is not nested and ends at the end of the iteration +after successful socket creation, binding, and listening. In case any of +the operations fails, it either continues to the next iteration (and the +next port) or jumps out of the loop to handle the error and exits the +function. + +Signed-off-by: Juraj Marcin +Reviewed-by: Daniel P. Berrangé +Signed-off-by: Daniel P. Berrangé + +(cherry picked from commit 911e0f2c6e2d00c985affa75ec188c8edcf480f2) + +JIRA: https://issues.redhat.com/browse/RHEL-67104 + +Signed-off-by: Juraj Marcin +--- + util/qemu-sockets.c | 51 ++++++++++++++++++++++++--------------------- + 1 file changed, 27 insertions(+), 24 deletions(-) + +diff --git a/util/qemu-sockets.c b/util/qemu-sockets.c +index 2c0e4883ce..631d0c4023 100644 +--- a/util/qemu-sockets.c ++++ b/util/qemu-sockets.c +@@ -303,11 +303,20 @@ static int inet_listen_saddr(InetSocketAddress *saddr, + port_min = inet_getport(e); + port_max = saddr->has_to ? saddr->to + port_offset : port_min; + for (p = port_min; p <= port_max; p++) { ++ if (slisten >= 0) { ++ /* ++ * We have a socket we tried with the previous port. It cannot ++ * be rebound, we need to close it and create a new one. ++ */ ++ close(slisten); ++ slisten = -1; ++ } + inet_setport(e, p); + + slisten = create_fast_reuse_socket(e); + if (slisten < 0) { +- /* First time we expect we might fail to create the socket ++ /* ++ * First time we expect we might fail to create the socket + * eg if 'e' has AF_INET6 but ipv6 kmod is not loaded. + * Later iterations should always succeed if first iteration + * worked though, so treat that as fatal. +@@ -317,40 +326,38 @@ static int inet_listen_saddr(InetSocketAddress *saddr, + } else { + error_setg_errno(errp, errno, + "Failed to recreate failed listening socket"); +- goto listen_failed; ++ goto fail; + } + } + socket_created = true; + + rc = try_bind(slisten, saddr, e); + if (rc < 0) { +- if (errno != EADDRINUSE) { +- error_setg_errno(errp, errno, "Failed to bind socket"); +- goto listen_failed; +- } +- } else { +- if (!listen(slisten, num)) { +- goto listen_ok; ++ if (errno == EADDRINUSE) { ++ /* This port is already used, try the next one */ ++ continue; + } +- if (errno != EADDRINUSE) { +- error_setg_errno(errp, errno, "Failed to listen on socket"); +- goto listen_failed; ++ error_setg_errno(errp, errno, "Failed to bind socket"); ++ goto fail; ++ } ++ if (listen(slisten, num)) { ++ if (errno == EADDRINUSE) { ++ /* This port is already used, try the next one */ ++ continue; + } ++ error_setg_errno(errp, errno, "Failed to listen on socket"); ++ goto fail; + } +- /* Someone else managed to bind to the same port and beat us +- * to listen on it! Socket semantics does not allow us to +- * recover from this situation, so we need to recreate the +- * socket to allow bind attempts for subsequent ports: +- */ +- close(slisten); +- slisten = -1; ++ /* We have a listening socket */ ++ freeaddrinfo(res); ++ return slisten; + } + } + error_setg_errno(errp, errno, + socket_created ? + "Failed to find an available port" : + "Failed to create a socket"); +-listen_failed: ++fail: + saved_errno = errno; + if (slisten >= 0) { + close(slisten); +@@ -358,10 +365,6 @@ listen_failed: + freeaddrinfo(res); + errno = saved_errno; + return -1; +- +-listen_ok: +- freeaddrinfo(res); +- return slisten; + } + + #ifdef _WIN32 +-- +2.39.3 + diff --git a/SOURCES/kvm-vfio-helpers-Align-mmaps.patch b/SOURCES/kvm-vfio-helpers-Align-mmaps.patch index 419dc88..784e024 100644 --- a/SOURCES/kvm-vfio-helpers-Align-mmaps.patch +++ b/SOURCES/kvm-vfio-helpers-Align-mmaps.patch @@ -1,17 +1,17 @@ -From da69744f6441f4a63215841ae9add2c1ef631047 Mon Sep 17 00:00:00 2001 +From 0e733c43122688a40b0bad9cf9af43ac3655fa30 Mon Sep 17 00:00:00 2001 From: Alex Williamson Date: Tue, 22 Oct 2024 14:08:29 -0600 -Subject: [PATCH 2/2] vfio/helpers: Align mmaps +Subject: [PATCH 5/5] vfio/helpers: Align mmaps MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit -RH-Author: Alex Williamson -RH-MergeRequest: 468: vfio/helpers: Align mmaps [9.6.z] -RH-Jira: RHEL-107314 +RH-Author: Donald Dutile +RH-MergeRequest: 366: Improve VFIO mmapping performance with huge pfnmaps +RH-Jira: RHEL-88533 RH-Acked-by: Cédric Le Goater -RH-Acked-by: Peter Xu -RH-Commit: [2/2] 4941ab64a5c0508eefa6bc5f923d6fcc38fdda4b +RH-Acked-by: Alex Williamson +RH-Commit: [2/2] f0e99cf993f82796352376bc7280342729ea5624 (ddutile/qemu-kvm) Thanks to work by Peter Xu, support is introduced in Linux v6.12 to allow pfnmap insertions at PMD and PUD levels of the page table. This @@ -37,7 +37,10 @@ Signed-off-by: Alex Williamson Reviewed-by: Peter Xu Reviewed-by: Cédric Le Goater (cherry picked from commit 00b519c0bca0e933ed22e2e6f8bca6b23f41f950) -Signed-off-by: Alex Williamson + +Jira: https://issues.redhat.com/browse/RHEL-88533 + +Signed-off-by: Donald Dutile --- hw/vfio/helpers.c | 32 ++++++++++++++++++++++++++++++-- 1 file changed, 30 insertions(+), 2 deletions(-) diff --git a/SOURCES/kvm-vfio-helpers-Refactor-vfio_region_mmap-error-handlin.patch b/SOURCES/kvm-vfio-helpers-Refactor-vfio_region_mmap-error-handlin.patch index 5503890..6b06e31 100644 --- a/SOURCES/kvm-vfio-helpers-Refactor-vfio_region_mmap-error-handlin.patch +++ b/SOURCES/kvm-vfio-helpers-Refactor-vfio_region_mmap-error-handlin.patch @@ -1,17 +1,17 @@ -From 41ea67ec82122d12ed26a98fe32d29e90f8fd282 Mon Sep 17 00:00:00 2001 +From f3af9e4476546c0bc814f78d5dd1047ec60768e8 Mon Sep 17 00:00:00 2001 From: Alex Williamson Date: Tue, 22 Oct 2024 14:08:28 -0600 -Subject: [PATCH 1/2] vfio/helpers: Refactor vfio_region_mmap() error handling +Subject: [PATCH 4/5] vfio/helpers: Refactor vfio_region_mmap() error handling MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit -RH-Author: Alex Williamson -RH-MergeRequest: 468: vfio/helpers: Align mmaps [9.6.z] -RH-Jira: RHEL-107314 +RH-Author: Donald Dutile +RH-MergeRequest: 366: Improve VFIO mmapping performance with huge pfnmaps +RH-Jira: RHEL-88533 RH-Acked-by: Cédric Le Goater -RH-Acked-by: Peter Xu -RH-Commit: [1/2] b91e8d009b8a6ff91bf273211272b101bc1c1146 +RH-Acked-by: Alex Williamson +RH-Commit: [1/2] b83c7dbc6a6037b465a141961ae810e5551fad30 (ddutile/qemu-kvm) Move error handling code to the end of the function so that it can more easily be shared by new mmap failure conditions. No functional change @@ -21,7 +21,10 @@ Signed-off-by: Alex Williamson Reviewed-by: Peter Xu Reviewed-by: Cédric Le Goater (cherry picked from commit 49915c0d2c9868e6f25e52e4d839943611b69e98) -Signed-off-by: Alex Williamson + +Jira: https://issues.redhat.com/browse/RHEL-88533 + +Signed-off-by: Donald Dutile --- hw/vfio/helpers.c | 34 +++++++++++++++++----------------- 1 file changed, 17 insertions(+), 17 deletions(-) diff --git a/SOURCES/kvm-vfio-pci-Delete-local-pm_cap.patch b/SOURCES/kvm-vfio-pci-Delete-local-pm_cap.patch new file mode 100644 index 0000000..0d18886 --- /dev/null +++ b/SOURCES/kvm-vfio-pci-Delete-local-pm_cap.patch @@ -0,0 +1,81 @@ +From 80be4b7d44d4721bacaa6205a47f2d898a090c6b Mon Sep 17 00:00:00 2001 +From: Alex Williamson +Date: Tue, 25 Feb 2025 14:52:27 -0700 +Subject: [PATCH 4/7] vfio/pci: Delete local pm_cap +MIME-Version: 1.0 +Content-Type: text/plain; charset=UTF-8 +Content-Transfer-Encoding: 8bit + +RH-Author: Eric Auger +RH-MergeRequest: 348: PCI: Implement basic PCI PM capability backing +RH-Jira: RHEL-7301 +RH-Acked-by: Cédric Le Goater +RH-Acked-by: Alex Williamson +RH-Acked-by: Jon Maloy +RH-Commit: [4/6] 85bd6b15af7c483e36e265c12b7b1689a4872f4c (eauger1/centos-qemu-kvm) + +This is now redundant to PCIDevice.pm_cap. + +Cc: Cédric Le Goater +Reviewed-by: Zhenzhong Duan +Reviewed-by: Eric Auger +Signed-off-by: Alex Williamson +Reviewed-by: Michael S. Tsirkin +Link: https://lore.kernel.org/qemu-devel/20250225215237.3314011-4-alex.williamson@redhat.com +Signed-off-by: Cédric Le Goater +(cherry picked from commit 05c6a8eff6298675080aa2692ee05a310b3483b4) +Signed-off-by: Eric Auger +--- + hw/vfio/pci.c | 9 ++++----- + hw/vfio/pci.h | 1 - + 2 files changed, 4 insertions(+), 6 deletions(-) + +diff --git a/hw/vfio/pci.c b/hw/vfio/pci.c +index e18b57d864..595b5c9b25 100644 +--- a/hw/vfio/pci.c ++++ b/hw/vfio/pci.c +@@ -2219,7 +2219,6 @@ static bool vfio_add_std_cap(VFIOPCIDevice *vdev, uint8_t pos, Error **errp) + break; + case PCI_CAP_ID_PM: + vfio_check_pm_reset(vdev, pos); +- vdev->pm_cap = pos; + ret = pci_pm_init(pdev, pos, errp) >= 0; + /* + * PCI-core config space emulation needs write access to the power +@@ -2416,17 +2415,17 @@ void vfio_pci_pre_reset(VFIOPCIDevice *vdev) + vfio_disable_interrupts(vdev); + + /* Make sure the device is in D0 */ +- if (vdev->pm_cap) { ++ if (pdev->pm_cap) { + uint16_t pmcsr; + uint8_t state; + +- pmcsr = vfio_pci_read_config(pdev, vdev->pm_cap + PCI_PM_CTRL, 2); ++ pmcsr = vfio_pci_read_config(pdev, pdev->pm_cap + PCI_PM_CTRL, 2); + state = pmcsr & PCI_PM_CTRL_STATE_MASK; + if (state) { + pmcsr &= ~PCI_PM_CTRL_STATE_MASK; +- vfio_pci_write_config(pdev, vdev->pm_cap + PCI_PM_CTRL, pmcsr, 2); ++ vfio_pci_write_config(pdev, pdev->pm_cap + PCI_PM_CTRL, pmcsr, 2); + /* vfio handles the necessary delay here */ +- pmcsr = vfio_pci_read_config(pdev, vdev->pm_cap + PCI_PM_CTRL, 2); ++ pmcsr = vfio_pci_read_config(pdev, pdev->pm_cap + PCI_PM_CTRL, 2); + state = pmcsr & PCI_PM_CTRL_STATE_MASK; + if (state) { + error_report("vfio: Unable to power on device, stuck in D%d", +diff --git a/hw/vfio/pci.h b/hw/vfio/pci.h +index 0d3c93fb2e..ca8d55f8b2 100644 +--- a/hw/vfio/pci.h ++++ b/hw/vfio/pci.h +@@ -161,7 +161,6 @@ struct VFIOPCIDevice { + int32_t bootindex; + uint32_t igd_gms; + OffAutoPCIBAR msix_relo; +- uint8_t pm_cap; + uint8_t nv_gpudirect_clique; + bool pci_aer; + bool req_enabled; +-- +2.48.1 + diff --git a/SOURCES/kvm-virtio-kconfig-memory-devices-are-PCI-only.patch b/SOURCES/kvm-virtio-kconfig-memory-devices-are-PCI-only.patch new file mode 100644 index 0000000..0ddc3e9 --- /dev/null +++ b/SOURCES/kvm-virtio-kconfig-memory-devices-are-PCI-only.patch @@ -0,0 +1,87 @@ +From a582cf6f68febba05e20548f643c8be637eab7b8 Mon Sep 17 00:00:00 2001 +From: Paolo Bonzini +Date: Fri, 6 Sep 2024 12:16:58 +0200 +Subject: [PATCH 01/26] virtio: kconfig: memory devices are PCI only + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [1/26] 0f0eab06b6f79f84c2e8d4fee28309b3c7c57414 (thuth/qemu-kvm-cs) + +Virtio memory devices rely on PCI BARs to expose the contents of memory. +Because of this they cannot be used (yet) with virtio-mmio or virtio-ccw. +In fact the code that is common to virtio-mem and virtio-pmem, which +is in hw/virtio/virtio-md-pci.c, is only included if CONFIG_VIRTIO_PCI +is set. Reproduce the same condition in the Kconfig file, only allowing +VIRTIO_MEM and VIRTIO_PMEM to be defined if the transport supports it. + +Without this patch it is possible to create a configuration with +CONFIG_VIRTIO_PCI=n and CONFIG_VIRTIO_MEM=y, but that causes a +linking failure. + +Message-ID: <20240906101658.514470-1-pbonzini@redhat.com> +Reported-by: Michael Tokarev +Reviewed-by: David Hildenbrand +Signed-off-by: Paolo Bonzini +Signed-off-by: David Hildenbrand +(cherry picked from commit 8d018fe59a0beff580ac6b3399d642c4277d9dd0) +Signed-off-by: Thomas Huth +--- + hw/virtio/Kconfig | 11 +++++++++++ + 1 file changed, 11 insertions(+) + +diff --git a/hw/virtio/Kconfig b/hw/virtio/Kconfig +index aa63ff7fd4..0afec2ae92 100644 +--- a/hw/virtio/Kconfig ++++ b/hw/virtio/Kconfig +@@ -16,6 +16,7 @@ config VIRTIO_PCI + default y if PCI_DEVICES + depends on PCI + select VIRTIO ++ select VIRTIO_MD_SUPPORTED + + config VIRTIO_MMIO + bool +@@ -35,10 +36,17 @@ config VIRTIO_CRYPTO + default y + depends on VIRTIO + ++# not all virtio transports support memory devices; if none does, ++# no need to include the code ++config VIRTIO_MD_SUPPORTED ++ bool ++ + config VIRTIO_MD + bool ++ depends on VIRTIO_MD_SUPPORTED + select MEM_DEVICE + ++# selected by the board if it has the required support code + config VIRTIO_PMEM_SUPPORTED + bool + +@@ -46,9 +54,11 @@ config VIRTIO_PMEM + bool + default y + depends on VIRTIO ++ depends on VIRTIO_MD_SUPPORTED + depends on VIRTIO_PMEM_SUPPORTED + select VIRTIO_MD + ++# selected by the board if it has the required support code + config VIRTIO_MEM_SUPPORTED + bool + +@@ -57,6 +67,7 @@ config VIRTIO_MEM + default y + depends on VIRTIO + depends on LINUX ++ depends on VIRTIO_MD_SUPPORTED + depends on VIRTIO_MEM_SUPPORTED + select VIRTIO_MD + +-- +2.48.1 + diff --git a/SOURCES/kvm-virtio-mem-Add-support-for-suspend-wake-up-with-plug.patch b/SOURCES/kvm-virtio-mem-Add-support-for-suspend-wake-up-with-plug.patch new file mode 100644 index 0000000..ba58998 --- /dev/null +++ b/SOURCES/kvm-virtio-mem-Add-support-for-suspend-wake-up-with-plug.patch @@ -0,0 +1,79 @@ +From 001200670ce9076a34419828e7e7ba92f19a80b7 Mon Sep 17 00:00:00 2001 +From: Juraj Marcin +Date: Wed, 4 Sep 2024 12:37:15 +0200 +Subject: [PATCH 08/26] virtio-mem: Add support for suspend+wake-up with + plugged memory + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [8/26] 08e25d41e32b3ac2bf5e0266f9c7e91739eda4d4 (thuth/qemu-kvm-cs) + +Before, the virtio-mem device would unplug all the memory with any reset +of the device, including during the wake-up of the guest from a +suspended state. Due to this, the virtio-mem driver in the Linux kernel +disallowed suspend-to-ram requests in the guest when the +VIRTIO_MEM_F_PERSISTENT_SUSPEND feature is not exposed by QEMU. + +This patch adds the code to skip the reset on wake-up and exposes +theVIRTIO_MEM_F_PERSISTENT_SUSPEND feature to the guest kernel driver +when suspending is possible in QEMU (currently only x86). + +Message-ID: <20240904103722.946194-5-jmarcin@redhat.com> +Reviewed-by: David Hildenbrand +Signed-off-by: Juraj Marcin +Signed-off-by: David Hildenbrand +(cherry picked from commit 1f5f49056d0f140568805d66f33396ed5cd90369) +Signed-off-by: Thomas Huth +--- + hw/virtio/virtio-mem.c | 10 ++++++++++ + hw/virtio/virtio-qmp.c | 3 +++ + 2 files changed, 13 insertions(+) + +diff --git a/hw/virtio/virtio-mem.c b/hw/virtio/virtio-mem.c +index 025ae4abac..51642a15ef 100644 +--- a/hw/virtio/virtio-mem.c ++++ b/hw/virtio/virtio-mem.c +@@ -883,6 +883,9 @@ static uint64_t virtio_mem_get_features(VirtIODevice *vdev, uint64_t features, + if (vmem->unplugged_inaccessible == ON_OFF_AUTO_ON) { + virtio_add_feature(&features, VIRTIO_MEM_F_UNPLUGGED_INACCESSIBLE); + } ++ if (qemu_wakeup_suspend_enabled()) { ++ virtio_add_feature(&features, VIRTIO_MEM_F_PERSISTENT_SUSPEND); ++ } + return features; + } + +@@ -1842,6 +1845,13 @@ static void virtio_mem_system_reset_hold(Object *obj, ResetType type) + { + VirtIOMEM *vmem = VIRTIO_MEM(obj); + ++ /* ++ * When waking up from standby/suspend-to-ram, do not unplug any memory. ++ */ ++ if (type == RESET_TYPE_WAKEUP) { ++ return; ++ } ++ + /* + * During usual resets, we will unplug all memory and shrink the usable + * region size. This is, however, not possible in all scenarios. Then, +diff --git a/hw/virtio/virtio-qmp.c b/hw/virtio/virtio-qmp.c +index 1dd96ed20f..cccc6fe761 100644 +--- a/hw/virtio/virtio-qmp.c ++++ b/hw/virtio/virtio-qmp.c +@@ -450,6 +450,9 @@ static const qmp_virtio_feature_map_t virtio_mem_feature_map[] = { + FEATURE_ENTRY(VIRTIO_MEM_F_UNPLUGGED_INACCESSIBLE, \ + "VIRTIO_MEM_F_UNPLUGGED_INACCESSIBLE: Unplugged memory cannot be " + "accessed"), ++ FEATURE_ENTRY(VIRTIO_MEM_F_PERSISTENT_SUSPEND, \ ++ "VIRTIO_MEM_F_PERSISTENT_SUSPND: Plugged memory will remain " ++ "plugged when suspending+resuming"), + { -1, "" } + }; + #endif +-- +2.48.1 + diff --git a/SOURCES/kvm-virtio-mem-Use-new-Resettable-framework-instead-of-L.patch b/SOURCES/kvm-virtio-mem-Use-new-Resettable-framework-instead-of-L.patch new file mode 100644 index 0000000..5bb3c9e --- /dev/null +++ b/SOURCES/kvm-virtio-mem-Use-new-Resettable-framework-instead-of-L.patch @@ -0,0 +1,141 @@ +From 6bc0cdecdc736d642bb6c040e07d79a0a3e591ea Mon Sep 17 00:00:00 2001 +From: Juraj Marcin +Date: Wed, 4 Sep 2024 12:37:14 +0200 +Subject: [PATCH 07/26] virtio-mem: Use new Resettable framework instead of + LegacyReset + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [7/26] 31fddbeb4aaf6794b83399a7e2996f01d918d748 (thuth/qemu-kvm-cs) + +LegacyReset does not pass ResetType to the reset callback method, which +the new Resettable framework uses. Due to this, virtio-mem cannot use +the new RESET_TYPE_WAKEUP to skip the reset during wake-up from a +suspended state. + +This patch adds overrides Resettable interface methods in VirtIOMEMClass +to use the new Resettable framework and replaces +qemu_[un]register_reset() calls with qemu_[un]register_resettable(). + +Message-ID: <20240904103722.946194-4-jmarcin@redhat.com> +Reviewed-by: David Hildenbrand +Signed-off-by: Juraj Marcin +Signed-off-by: David Hildenbrand +(cherry picked from commit c009a311e93963860cfba917605a4bf903a06bce) +Signed-off-by: Thomas Huth +--- + hw/virtio/virtio-mem.c | 38 +++++++++++++++++++++------------- + include/hw/virtio/virtio-mem.h | 4 ++++ + 2 files changed, 28 insertions(+), 14 deletions(-) + +diff --git a/hw/virtio/virtio-mem.c b/hw/virtio/virtio-mem.c +index ba11aa4646..025ae4abac 100644 +--- a/hw/virtio/virtio-mem.c ++++ b/hw/virtio/virtio-mem.c +@@ -895,18 +895,6 @@ static int virtio_mem_validate_features(VirtIODevice *vdev) + return 0; + } + +-static void virtio_mem_system_reset(void *opaque) +-{ +- VirtIOMEM *vmem = VIRTIO_MEM(opaque); +- +- /* +- * During usual resets, we will unplug all memory and shrink the usable +- * region size. This is, however, not possible in all scenarios. Then, +- * the guest has to deal with this manually (VIRTIO_MEM_REQ_UNPLUG_ALL). +- */ +- virtio_mem_unplug_all(vmem); +-} +- + static void virtio_mem_prepare_mr(VirtIOMEM *vmem) + { + const uint64_t region_size = memory_region_size(&vmem->memdev->mr); +@@ -1123,7 +1111,7 @@ static void virtio_mem_device_realize(DeviceState *dev, Error **errp) + vmstate_register_any(VMSTATE_IF(vmem), + &vmstate_virtio_mem_device_early, vmem); + } +- qemu_register_reset(virtio_mem_system_reset, vmem); ++ qemu_register_resettable(OBJECT(vmem)); + + /* + * Set ourselves as RamDiscardManager before the plug handler maps the +@@ -1143,7 +1131,7 @@ static void virtio_mem_device_unrealize(DeviceState *dev) + * found via an address space anymore. Unset ourselves. + */ + memory_region_set_ram_discard_manager(&vmem->memdev->mr, NULL); +- qemu_unregister_reset(virtio_mem_system_reset, vmem); ++ qemu_unregister_resettable(OBJECT(vmem)); + if (vmem->early_migration) { + vmstate_unregister(VMSTATE_IF(vmem), &vmstate_virtio_mem_device_early, + vmem); +@@ -1844,12 +1832,31 @@ static void virtio_mem_unplug_request_check(VirtIOMEM *vmem, Error **errp) + } + } + ++static ResettableState *virtio_mem_get_reset_state(Object *obj) ++{ ++ VirtIOMEM *vmem = VIRTIO_MEM(obj); ++ return &vmem->reset_state; ++} ++ ++static void virtio_mem_system_reset_hold(Object *obj, ResetType type) ++{ ++ VirtIOMEM *vmem = VIRTIO_MEM(obj); ++ ++ /* ++ * During usual resets, we will unplug all memory and shrink the usable ++ * region size. This is, however, not possible in all scenarios. Then, ++ * the guest has to deal with this manually (VIRTIO_MEM_REQ_UNPLUG_ALL). ++ */ ++ virtio_mem_unplug_all(vmem); ++} ++ + static void virtio_mem_class_init(ObjectClass *klass, void *data) + { + DeviceClass *dc = DEVICE_CLASS(klass); + VirtioDeviceClass *vdc = VIRTIO_DEVICE_CLASS(klass); + VirtIOMEMClass *vmc = VIRTIO_MEM_CLASS(klass); + RamDiscardManagerClass *rdmc = RAM_DISCARD_MANAGER_CLASS(klass); ++ ResettableClass *rc = RESETTABLE_CLASS(klass); + + device_class_set_props(dc, virtio_mem_properties); + dc->vmsd = &vmstate_virtio_mem; +@@ -1876,6 +1883,9 @@ static void virtio_mem_class_init(ObjectClass *klass, void *data) + rdmc->replay_discarded = virtio_mem_rdm_replay_discarded; + rdmc->register_listener = virtio_mem_rdm_register_listener; + rdmc->unregister_listener = virtio_mem_rdm_unregister_listener; ++ ++ rc->get_state = virtio_mem_get_reset_state; ++ rc->phases.hold = virtio_mem_system_reset_hold; + } + + static const TypeInfo virtio_mem_info = { +diff --git a/include/hw/virtio/virtio-mem.h b/include/hw/virtio/virtio-mem.h +index 5f5b02b8f9..a1af144c28 100644 +--- a/include/hw/virtio/virtio-mem.h ++++ b/include/hw/virtio/virtio-mem.h +@@ -14,6 +14,7 @@ + #define HW_VIRTIO_MEM_H + + #include "standard-headers/linux/virtio_mem.h" ++#include "hw/resettable.h" + #include "hw/virtio/virtio.h" + #include "qapi/qapi-types-misc.h" + #include "sysemu/hostmem.h" +@@ -115,6 +116,9 @@ struct VirtIOMEM { + + /* listeners to notify on plug/unplug activity. */ + QLIST_HEAD(, RamDiscardListener) rdl_list; ++ ++ /* State of the resettable container */ ++ ResettableState reset_state; + }; + + struct VirtIOMEMClass { +-- +2.48.1 + diff --git a/SOURCES/kvm-virtio-mem-don-t-warn-about-THP-sizes-on-a-kernel-wi.patch b/SOURCES/kvm-virtio-mem-don-t-warn-about-THP-sizes-on-a-kernel-wi.patch new file mode 100644 index 0000000..f527a70 --- /dev/null +++ b/SOURCES/kvm-virtio-mem-don-t-warn-about-THP-sizes-on-a-kernel-wi.patch @@ -0,0 +1,59 @@ +From f4052d25199bfce8ce29a173934a805fe1cf7e3e Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Tue, 10 Sep 2024 18:34:33 +0200 +Subject: [PATCH 25/26] virtio-mem: don't warn about THP sizes on a kernel + without THP support + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [25/26] 5dff17ef818722db8f1fa87cff5b7777afc3c814 (thuth/qemu-kvm-cs) + +If the config directory in sysfs does not exist at all, we are dealing +with a system that does not support THPs. Simply use 1 MiB block size +then, instead of warning "Could not detect THP size, falling back to +..." and falling back to the default THP size. + +Cc: "Michael S. Tsirkin" +Cc: Gavin Shan +Cc: Juraj Marcin +Signed-off-by: David Hildenbrand +Message-Id: <20240910163433.2100295-1-david@redhat.com> +Reviewed-by: Michael S. Tsirkin +Signed-off-by: Michael S. Tsirkin +(cherry picked from commit 95b717a8154b955de2782305f305b63f357b0576) +Signed-off-by: Thomas Huth +--- + hw/virtio/virtio-mem.c | 7 +++++++ + 1 file changed, 7 insertions(+) + +diff --git a/hw/virtio/virtio-mem.c b/hw/virtio/virtio-mem.c +index c9f8a23bbc..4977658312 100644 +--- a/hw/virtio/virtio-mem.c ++++ b/hw/virtio/virtio-mem.c +@@ -90,6 +90,7 @@ static uint32_t virtio_mem_default_thp_size(void) + static uint32_t thp_size; + + #define HPAGE_PMD_SIZE_PATH "/sys/kernel/mm/transparent_hugepage/hpage_pmd_size" ++#define HPAGE_PATH "/sys/kernel/mm/transparent_hugepage/" + static uint32_t virtio_mem_thp_size(void) + { + gchar *content = NULL; +@@ -100,6 +101,12 @@ static uint32_t virtio_mem_thp_size(void) + return thp_size; + } + ++ /* No THP -> no restrictions. */ ++ if (!g_file_test(HPAGE_PATH, G_FILE_TEST_EXISTS)) { ++ thp_size = VIRTIO_MEM_MIN_BLOCK_SIZE; ++ return thp_size; ++ } ++ + /* + * Try to probe the actual THP size, fallback to (sane but eventually + * incorrect) default sizes. +-- +2.48.1 + diff --git a/SOURCES/kvm-virtio-mem-unplug-memory-only-during-system-resets-n.patch b/SOURCES/kvm-virtio-mem-unplug-memory-only-during-system-resets-n.patch new file mode 100644 index 0000000..cfb49f4 --- /dev/null +++ b/SOURCES/kvm-virtio-mem-unplug-memory-only-during-system-resets-n.patch @@ -0,0 +1,258 @@ +From e5f2bb584154eef665211228f1ac3113e2acc269 Mon Sep 17 00:00:00 2001 +From: David Hildenbrand +Date: Fri, 25 Oct 2024 12:41:03 +0200 +Subject: [PATCH 09/26] virtio-mem: unplug memory only during system resets, + not device resets + +RH-Author: Thomas Huth +RH-MergeRequest: 351: Enable virtio-mem support on s390x +RH-Jira: RHEL-72977 +RH-Acked-by: David Hildenbrand +RH-Acked-by: Juraj Marcin +RH-Commit: [9/26] 7c5ddd4d3fd0d19caa946bcbe98cb5732404978b (thuth/qemu-kvm-cs) + +We recently converted from the LegacyReset to the new reset framework +in commit c009a311e939 ("virtio-mem: Use new Resettable framework instead +of LegacyReset") to be able to use the ResetType to filter out wakeup +resets. + +However, this change had an undesired implications: as we override the +Resettable interface methods in VirtIOMEMClass, the reset handler will +not only get called during system resets (i.e., qemu_devices_reset()) +but also during any direct or indirect device rests (e.g., +device_cold_reset()). + +Further, we might now receive two reset callbacks during +qemu_devices_reset(), first when reset by a parent and later when reset +directly. + +The memory state of virtio-mem devices is rather special: it's supposed to +be persistent/unchanged during most resets (similar to resetting a hard +disk will not destroy the data), unless actually cold-resetting the whole +system (different to a hard disk where a reboot will not destroy the data): +ripping out system RAM is something guest OSes don't particularly enjoy, +but we want to detect when rebooting to an OS that does not support +virtio-mem and wouldn't be able to detect+use the memory -- and we want +to force-defragment hotplugged memory to also shrink the usable device +memory region. So we rally want to catch system resets to do that. + +On supported targets (e.g., x86), getting a cold reset on the +device/parent triggers is not that easy (but looks like PCI code +might trigger it), so this implication went unnoticed. + +However, with upcoming s390x support it is problematic: during +kdump, s390x triggers a subsystem reset, ending up in +s390_machine_reset() and calling only subsystem_reset() instead of +qemu_devices_reset() -- because it's not a full system reset. + +In subsystem_reset(), s390x performs a device_cold_reset() of any +TYPE_VIRTUAL_CSS_BRIDGE device, which ends up resetting all children, +including the virtio-mem device. Consequently, we wrongly detect a system +reset and unplug all device memory, resulting in hotplugged memory not +getting included in the crash dump -- undesired. + +We really must not mess with hotplugged memory state during simple +device resets. To fix, create+register a new reset object that will only +get triggered during qemu_devices_reset() calls, but not during any other +resets as it is logically not the child of any other object. + +Message-ID: <20241025104103.342188-1-david@redhat.com> +Acked-by: Michael S. Tsirkin +Cc: "Michael S. Tsirkin" +Cc: Juraj Marcin +Cc: Peter Maydell +Signed-off-by: David Hildenbrand +(cherry picked from commit 713484d0389c9d1cbb87eca060361281248b69f5) +Signed-off-by: Thomas Huth +--- + hw/virtio/virtio-mem.c | 103 +++++++++++++++++++++++---------- + include/hw/virtio/virtio-mem.h | 13 ++++- + 2 files changed, 84 insertions(+), 32 deletions(-) + +diff --git a/hw/virtio/virtio-mem.c b/hw/virtio/virtio-mem.c +index 51642a15ef..00da98b6e1 100644 +--- a/hw/virtio/virtio-mem.c ++++ b/hw/virtio/virtio-mem.c +@@ -949,6 +949,7 @@ static void virtio_mem_device_realize(DeviceState *dev, Error **errp) + VirtIOMEM *vmem = VIRTIO_MEM(dev); + uint64_t page_size; + RAMBlock *rb; ++ Object *obj; + int ret; + + if (!vmem->memdev) { +@@ -1114,7 +1115,28 @@ static void virtio_mem_device_realize(DeviceState *dev, Error **errp) + vmstate_register_any(VMSTATE_IF(vmem), + &vmstate_virtio_mem_device_early, vmem); + } +- qemu_register_resettable(OBJECT(vmem)); ++ ++ /* ++ * We only want to unplug all memory to start with a clean slate when ++ * it is safe for the guest -- during system resets that call ++ * qemu_devices_reset(). ++ * ++ * We'll filter out selected qemu_devices_reset() calls used for other ++ * purposes, like resetting all devices during wakeup from suspend on ++ * x86 based on the reset type passed to qemu_devices_reset(). ++ * ++ * Unplugging all memory during simple device resets can result in the VM ++ * unexpectedly losing RAM, corrupting VM state. ++ * ++ * Simple device resets (or resets triggered by getting a parent device ++ * reset) must not change the state of plugged memory blocks. Therefore, ++ * we need a dedicated reset object that only gets called during ++ * qemu_devices_reset(). ++ */ ++ obj = object_new(TYPE_VIRTIO_MEM_SYSTEM_RESET); ++ vmem->system_reset = VIRTIO_MEM_SYSTEM_RESET(obj); ++ vmem->system_reset->vmem = vmem; ++ qemu_register_resettable(obj); + + /* + * Set ourselves as RamDiscardManager before the plug handler maps the +@@ -1134,7 +1156,10 @@ static void virtio_mem_device_unrealize(DeviceState *dev) + * found via an address space anymore. Unset ourselves. + */ + memory_region_set_ram_discard_manager(&vmem->memdev->mr, NULL); +- qemu_unregister_resettable(OBJECT(vmem)); ++ ++ qemu_unregister_resettable(OBJECT(vmem->system_reset)); ++ object_unref(OBJECT(vmem->system_reset)); ++ + if (vmem->early_migration) { + vmstate_unregister(VMSTATE_IF(vmem), &vmstate_virtio_mem_device_early, + vmem); +@@ -1835,38 +1860,12 @@ static void virtio_mem_unplug_request_check(VirtIOMEM *vmem, Error **errp) + } + } + +-static ResettableState *virtio_mem_get_reset_state(Object *obj) +-{ +- VirtIOMEM *vmem = VIRTIO_MEM(obj); +- return &vmem->reset_state; +-} +- +-static void virtio_mem_system_reset_hold(Object *obj, ResetType type) +-{ +- VirtIOMEM *vmem = VIRTIO_MEM(obj); +- +- /* +- * When waking up from standby/suspend-to-ram, do not unplug any memory. +- */ +- if (type == RESET_TYPE_WAKEUP) { +- return; +- } +- +- /* +- * During usual resets, we will unplug all memory and shrink the usable +- * region size. This is, however, not possible in all scenarios. Then, +- * the guest has to deal with this manually (VIRTIO_MEM_REQ_UNPLUG_ALL). +- */ +- virtio_mem_unplug_all(vmem); +-} +- + static void virtio_mem_class_init(ObjectClass *klass, void *data) + { + DeviceClass *dc = DEVICE_CLASS(klass); + VirtioDeviceClass *vdc = VIRTIO_DEVICE_CLASS(klass); + VirtIOMEMClass *vmc = VIRTIO_MEM_CLASS(klass); + RamDiscardManagerClass *rdmc = RAM_DISCARD_MANAGER_CLASS(klass); +- ResettableClass *rc = RESETTABLE_CLASS(klass); + + device_class_set_props(dc, virtio_mem_properties); + dc->vmsd = &vmstate_virtio_mem; +@@ -1893,9 +1892,6 @@ static void virtio_mem_class_init(ObjectClass *klass, void *data) + rdmc->replay_discarded = virtio_mem_rdm_replay_discarded; + rdmc->register_listener = virtio_mem_rdm_register_listener; + rdmc->unregister_listener = virtio_mem_rdm_unregister_listener; +- +- rc->get_state = virtio_mem_get_reset_state; +- rc->phases.hold = virtio_mem_system_reset_hold; + } + + static const TypeInfo virtio_mem_info = { +@@ -1918,3 +1914,48 @@ static void virtio_register_types(void) + } + + type_init(virtio_register_types) ++ ++OBJECT_DEFINE_SIMPLE_TYPE_WITH_INTERFACES(VirtioMemSystemReset, virtio_mem_system_reset, VIRTIO_MEM_SYSTEM_RESET, OBJECT, { TYPE_RESETTABLE_INTERFACE }, { }) ++ ++static void virtio_mem_system_reset_init(Object *obj) ++{ ++} ++ ++static void virtio_mem_system_reset_finalize(Object *obj) ++{ ++} ++ ++static ResettableState *virtio_mem_system_reset_get_state(Object *obj) ++{ ++ VirtioMemSystemReset *vmem_reset = VIRTIO_MEM_SYSTEM_RESET(obj); ++ ++ return &vmem_reset->reset_state; ++} ++ ++static void virtio_mem_system_reset_hold(Object *obj, ResetType type) ++{ ++ VirtioMemSystemReset *vmem_reset = VIRTIO_MEM_SYSTEM_RESET(obj); ++ VirtIOMEM *vmem = vmem_reset->vmem; ++ ++ /* ++ * When waking up from standby/suspend-to-ram, do not unplug any memory. ++ */ ++ if (type == RESET_TYPE_WAKEUP) { ++ return; ++ } ++ ++ /* ++ * During usual resets, we will unplug all memory and shrink the usable ++ * region size. This is, however, not possible in all scenarios. Then, ++ * the guest has to deal with this manually (VIRTIO_MEM_REQ_UNPLUG_ALL). ++ */ ++ virtio_mem_unplug_all(vmem); ++} ++ ++static void virtio_mem_system_reset_class_init(ObjectClass *klass, void *data) ++{ ++ ResettableClass *rc = RESETTABLE_CLASS(klass); ++ ++ rc->get_state = virtio_mem_system_reset_get_state; ++ rc->phases.hold = virtio_mem_system_reset_hold; ++} +diff --git a/include/hw/virtio/virtio-mem.h b/include/hw/virtio/virtio-mem.h +index a1af144c28..550ce585b2 100644 +--- a/include/hw/virtio/virtio-mem.h ++++ b/include/hw/virtio/virtio-mem.h +@@ -25,6 +25,10 @@ + OBJECT_DECLARE_TYPE(VirtIOMEM, VirtIOMEMClass, + VIRTIO_MEM) + ++#define TYPE_VIRTIO_MEM_SYSTEM_RESET "virtio-mem-system-reset" ++ ++OBJECT_DECLARE_SIMPLE_TYPE(VirtioMemSystemReset, VIRTIO_MEM_SYSTEM_RESET) ++ + #define VIRTIO_MEM_MEMDEV_PROP "memdev" + #define VIRTIO_MEM_NODE_PROP "node" + #define VIRTIO_MEM_SIZE_PROP "size" +@@ -117,8 +121,15 @@ struct VirtIOMEM { + /* listeners to notify on plug/unplug activity. */ + QLIST_HEAD(, RamDiscardListener) rdl_list; + +- /* State of the resettable container */ ++ /* Catch system resets -> qemu_devices_reset() only. */ ++ VirtioMemSystemReset *system_reset; ++}; ++ ++struct VirtioMemSystemReset { ++ Object parent; ++ + ResettableState reset_state; ++ VirtIOMEM *vmem; + }; + + struct VirtIOMEMClass { +-- +2.48.1 + diff --git a/SOURCES/kvm-virtio-net-disable-USO-for-virt-rhel9.6.patch b/SOURCES/kvm-virtio-net-disable-USO-for-virt-rhel9.6.patch new file mode 100644 index 0000000..632c5aa --- /dev/null +++ b/SOURCES/kvm-virtio-net-disable-USO-for-virt-rhel9.6.patch @@ -0,0 +1,135 @@ +From a7cd7f5b3bd6df30e75532fb19b645c5349f6183 Mon Sep 17 00:00:00 2001 +From: Shaoqin Huang +Date: Thu, 24 Apr 2025 04:48:29 -0400 +Subject: [PATCH 1/5] virtio-net: disable USO for virt-rhel9.6 + +RH-Author: Shaoqin Huang +RH-MergeRequest: 353: virtio-net: disable USO for virt-rhel9.6 +RH-Jira: RHEL-80313 +RH-Acked-by: Thomas Huth +RH-Acked-by: Eric Auger +RH-Commit: [1/2] c7099480e656106219040d45ce7b76b19376227a (shahuang/qemu-kvm) + +JIRA: https://issues.redhat.com/browse/RHEL-80313 +Upstream Status: RHEL only + +RHEL9 kernels have USO* disabled while RHEL10 has it enabled, this can +cause the migration to fail when running a RHEL9 qemu on a RHEL10 kernel +and then migrate to a RHEL9 kernel. + +Make sure the virt-rhel9.6 machine type in RHEL9 stay the same +independent of the kernel. + +Signed-off-by: Shaoqin Huang +--- + hw/arm/virt.c | 3 +++ + hw/core/machine.c | 15 +++++++++------ + hw/i386/pc_piix.c | 1 + + hw/i386/pc_q35.c | 3 +++ + hw/s390x/s390-virtio-ccw.c | 2 ++ + include/hw/boards.h | 3 +++ + 6 files changed, 21 insertions(+), 6 deletions(-) + +diff --git a/hw/arm/virt.c b/hw/arm/virt.c +index c5270a5abc..896deaa025 100644 +--- a/hw/arm/virt.c ++++ b/hw/arm/virt.c +@@ -3600,6 +3600,9 @@ DEFINE_VIRT_MACHINE(2, 6) + static void virt_rhel_machine_9_6_0_options(MachineClass *mc) + { + compat_props_add(mc->compat_props, arm_rhel9_compat, arm_rhel9_compat_len); ++ ++ /* NB: remember to move this line to the *latest* RHEL 9 machine */ ++ compat_props_add(mc->compat_props, hw_compat_rhel_9, hw_compat_rhel_9_len); + } + DEFINE_VIRT_MACHINE_AS_LATEST(9, 6, 0) + +diff --git a/hw/core/machine.c b/hw/core/machine.c +index add42660f8..37751f6b9b 100644 +--- a/hw/core/machine.c ++++ b/hw/core/machine.c +@@ -305,6 +305,15 @@ GlobalProperty hw_compat_2_1[] = { + }; + const size_t hw_compat_2_1_len = G_N_ELEMENTS(hw_compat_2_1); + ++/* Apply this to all RHEL9 boards going backward and forward */ ++GlobalProperty hw_compat_rhel_9[] = { ++ /* supported by userspace, but RHEL 9 *kernels* do not support USO. */ ++ { TYPE_VIRTIO_NET, "host_uso", "off"}, ++ { TYPE_VIRTIO_NET, "guest_uso4", "off"}, ++ { TYPE_VIRTIO_NET, "guest_uso6", "off"}, ++}; ++const size_t hw_compat_rhel_9_len = G_N_ELEMENTS(hw_compat_rhel_9); ++ + /* + * RHEL only: machine types for previous major releases are deprecated + */ +@@ -341,12 +350,6 @@ GlobalProperty hw_compat_rhel_9_5[] = { + const size_t hw_compat_rhel_9_5_len = G_N_ELEMENTS(hw_compat_rhel_9_5); + + GlobalProperty hw_compat_rhel_9_4[] = { +- /* hw_compat_rhel_9_4 from hw_compat_8_0 */ +- { TYPE_VIRTIO_NET, "host_uso", "off"}, +- /* hw_compat_rhel_9_4 from hw_compat_8_0 */ +- { TYPE_VIRTIO_NET, "guest_uso4", "off"}, +- /* hw_compat_rhel_9_4 from hw_compat_8_0 */ +- { TYPE_VIRTIO_NET, "guest_uso6", "off"}, + /* hw_compat_rhel_9_4 from hw_compat_8_1 */ + { TYPE_PCI_BRIDGE, "x-pci-express-writeable-slt-bug", "true" }, + /* hw_compat_rhel_9_4 from hw_compat_8_1 */ +diff --git a/hw/i386/pc_piix.c b/hw/i386/pc_piix.c +index 656abb5d39..10764bf596 100644 +--- a/hw/i386/pc_piix.c ++++ b/hw/i386/pc_piix.c +@@ -929,6 +929,7 @@ static void pc_i440fx_rhel_machine_7_6_0_options(MachineClass *m) + compat_props_add(m->compat_props, pc_rhel_8_0_compat, pc_rhel_8_0_compat_len); + compat_props_add(m->compat_props, hw_compat_rhel_7_6, hw_compat_rhel_7_6_len); + compat_props_add(m->compat_props, pc_rhel_7_6_compat, pc_rhel_7_6_compat_len); ++ compat_props_add(m->compat_props, hw_compat_rhel_9, hw_compat_rhel_9_len); + } + + DEFINE_I440FX_MACHINE(7, 6, 0); +diff --git a/hw/i386/pc_q35.c b/hw/i386/pc_q35.c +index 578f63524f..5bf08be0fb 100644 +--- a/hw/i386/pc_q35.c ++++ b/hw/i386/pc_q35.c +@@ -679,6 +679,9 @@ static void pc_q35_rhel_machine_9_6_0_options(MachineClass *m) + m->desc = "RHEL-9.6.0 PC (Q35 + ICH9, 2009)"; + pcmc->smbios_stream_product = "RHEL"; + pcmc->smbios_stream_version = "9.6.0"; ++ ++ /* NB: remember to move this line to the *latest* RHEL 9 machine */ ++ compat_props_add(m->compat_props, hw_compat_rhel_9, hw_compat_rhel_9_len); + } + + DEFINE_Q35_MACHINE_BUGFIX(9, 6, 0); +diff --git a/hw/s390x/s390-virtio-ccw.c b/hw/s390x/s390-virtio-ccw.c +index 9f4ad01789..312e8f18aa 100644 +--- a/hw/s390x/s390-virtio-ccw.c ++++ b/hw/s390x/s390-virtio-ccw.c +@@ -1348,6 +1348,8 @@ static void ccw_rhel_machine_9_6_0_instance_options(MachineState *machine) + + static void ccw_rhel_machine_9_6_0_class_options(MachineClass *mc) + { ++ /* NB: remember to move this line to the *latest* RHEL 9 machine */ ++ compat_props_add(mc->compat_props, hw_compat_rhel_9, hw_compat_rhel_9_len); + } + DEFINE_CCW_MACHINE_AS_LATEST(9, 6, 0); + +diff --git a/include/hw/boards.h b/include/hw/boards.h +index fe011b1e86..8f3fa40cf9 100644 +--- a/include/hw/boards.h ++++ b/include/hw/boards.h +@@ -803,6 +803,9 @@ extern const size_t hw_compat_2_2_len; + extern GlobalProperty hw_compat_2_1[]; + extern const size_t hw_compat_2_1_len; + ++extern GlobalProperty hw_compat_rhel_9[]; ++extern const size_t hw_compat_rhel_9_len; ++ + extern GlobalProperty hw_compat_rhel_9_6[]; + extern const size_t hw_compat_rhel_9_6_len; + +-- +2.48.1 + diff --git a/SOURCES/qemu-ga.sysconfig b/SOURCES/qemu-ga.sysconfig index 736b471..b574514 100644 --- a/SOURCES/qemu-ga.sysconfig +++ b/SOURCES/qemu-ga.sysconfig @@ -13,7 +13,7 @@ # # You can get the list of RPC commands using "qemu-ga --allow-rpcs='?'". # There should be no spaces between commas and commands in the allow list. -FILTER_RPC_ARGS="--allow-rpcs=guest-sync-delimited,guest-sync,guest-ping,guest-get-time,guest-set-time,guest-info,guest-shutdown,guest-fsfreeze-status,guest-fsfreeze-freeze,guest-fsfreeze-freeze-list,guest-fsfreeze-thaw,guest-fstrim,guest-suspend-disk,guest-suspend-ram,guest-suspend-hybrid,guest-network-get-interfaces,guest-get-vcpus,guest-set-vcpus,guest-get-disks,guest-get-fsinfo,guest-set-user-password,guest-get-memory-blocks,guest-set-memory-blocks,guest-get-memory-block-info,guest-get-host-name,guest-get-users,guest-get-timezone,guest-get-osinfo,guest-get-devices,guest-ssh-get-authorized-keys,guest-ssh-add-authorized-keys,guest-ssh-remove-authorized-keys,guest-get-diskstats,guest-get-cpustats" +FILTER_RPC_ARGS="--allow-rpcs=guest-sync-delimited,guest-sync,guest-ping,guest-get-time,guest-set-time,guest-info,guest-shutdown,guest-fsfreeze-status,guest-fsfreeze-freeze,guest-fsfreeze-freeze-list,guest-fsfreeze-thaw,guest-fstrim,guest-suspend-disk,guest-suspend-ram,guest-suspend-hybrid,guest-network-get-interfaces,guest-get-vcpus,guest-set-vcpus,guest-get-disks,guest-get-fsinfo,guest-set-user-password,guest-get-memory-blocks,guest-set-memory-blocks,guest-get-memory-block-info,guest-get-host-name,guest-get-users,guest-get-timezone,guest-get-osinfo,guest-get-devices,guest-ssh-get-authorized-keys,guest-ssh-add-authorized-keys,guest-ssh-remove-authorized-keys,guest-get-diskstats,guest-get-cpustats,guest-network-get-route,guest-get-load" # Fsfreeze hook script specification. # diff --git a/SPECS/qemu-kvm.spec b/SPECS/qemu-kvm.spec index 5390a45..07e383c 100644 --- a/SPECS/qemu-kvm.spec +++ b/SPECS/qemu-kvm.spec @@ -149,7 +149,7 @@ Obsoletes: %{name}-block-ssh <= %{epoch}:%{version} \ Summary: QEMU is a machine emulator and virtualizer Name: qemu-kvm Version: 9.1.0 -Release: 15%{?rcrel}%{?dist}%{?cc_suffix}.9 +Release: 29%{?rcrel}%{?dist}%{?cc_suffix}.3 # Epoch because we pushed a qemu-1.0 package. AIUI this can't ever be dropped # Epoch 15 used for RHEL 8 # Epoch 17 used for RHEL 9 (due to release versioning offset in RHEL 8.5) @@ -465,90 +465,746 @@ Patch147: kvm-iotests-Add-qsd-migrate-case.patch # For RHEL-54296 - Provide QMP command for block device reactivation after migration [rhel-9.5] # For RHEL-78397 - backport fix for double migration of a paused VM (disk activation rewrite) Patch148: kvm-iotests-Add-NBD-based-tests-for-inactive-nodes.patch -# For RHEL-80622 - Allow libvirt to restart passt/vhost-user when the process is killed [rhel-9] -Patch149: kvm-net-vhost-user-add-QAPI-events-to-report-connection-.patch -# For RHEL-83000 - [qemu-guest-agent][RFE] Report CPU load average [rhel-9.6.z] -Patch150: kvm-qga-implement-a-guest-get-load-command.patch -# For RHEL-87734 - QEMU sends unaligned discards on 4K devices [rhel-9.6.z] -Patch151: kvm-file-posix-probe-discard-alignment-on-Linux-block-de.patch -# For RHEL-87734 - QEMU sends unaligned discards on 4K devices [rhel-9.6.z] -Patch152: kvm-block-io-skip-head-tail-requests-on-EINVAL.patch -# For RHEL-87734 - QEMU sends unaligned discards on 4K devices [rhel-9.6.z] -Patch153: kvm-file-posix-Fix-crash-on-discard_granularity-0.patch -# For RHEL-92077 - Fix x86 M-type compats [rhel-9.6.z] -Patch154: kvm-hw-i386-Fix-machine-type-compatibility.patch -# For RHEL-95407 - Support multipath failover with scsi-block [rhel-9.6.z] -Patch155: kvm-file-posix-Define-DM_MPATH_PROBE_PATHS.patch -# For RHEL-95407 - Support multipath failover with scsi-block [rhel-9.6.z] -Patch156: kvm-file-posix-Probe-paths-and-retry-SG_IO-on-potential-.patch -# For RHEL-100767 - Video stuck after switchover phase when play one video during migration [rhel-9.6.z] -Patch157: kvm-ui-vnc-Update-display-update-interval-when-VM-state-.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch158: kvm-include-qemu-compiler-add-QEMU_UNINITIALIZED-attribu.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch159: kvm-hw-virtio-virtio-avoid-cost-of-ftrivial-auto-var-ini.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch160: kvm-block-skip-automatic-zero-init-of-large-array-in-ioq.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch161: kvm-chardev-char-fd-skip-automatic-zero-init-of-large-ar.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch162: kvm-chardev-char-pty-skip-automatic-zero-init-of-large-a.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch163: kvm-chardev-char-socket-skip-automatic-zero-init-of-larg.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch164: kvm-hw-audio-ac97-skip-automatic-zero-init-of-large-arra.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch165: kvm-hw-audio-cs4231a-skip-automatic-zero-init-of-large-a.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch166: kvm-hw-audio-es1370-skip-automatic-zero-init-of-large-ar.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch167: kvm-hw-audio-gus-skip-automatic-zero-init-of-large-array.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch168: kvm-hw-audio-marvell_88w8618-skip-automatic-zero-init-of.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch169: kvm-hw-audio-sb16-skip-automatic-zero-init-of-large-arra.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch170: kvm-hw-audio-via-ac97-skip-automatic-zero-init-of-large-.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch171: kvm-hw-char-sclpconsole-lm-skip-automatic-zero-init-of-l.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch172: kvm-hw-dma-xlnx_csu_dma-skip-automatic-zero-init-of-larg.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch173: kvm-hw-display-vmware_vga-skip-automatic-zero-init-of-la.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch174: kvm-hw-hyperv-syndbg-skip-automatic-zero-init-of-large-a.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch175: kvm-hw-misc-aspeed_hace-skip-automatic-zero-init-of-larg.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch176: kvm-hw-net-rtl8139-skip-automatic-zero-init-of-large-arr.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch177: kvm-hw-net-tulip-skip-automatic-zero-init-of-large-array.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch178: kvm-hw-net-virtio-net-skip-automatic-zero-init-of-large-.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch179: kvm-hw-net-xgamc-skip-automatic-zero-init-of-large-array.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch180: kvm-hw-nvme-ctrl-skip-automatic-zero-init-of-large-array.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch181: kvm-hw-ppc-spapr_tpm_proxy-skip-automatic-zero-init-of-l.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch182: kvm-hw-usb-hcd-ohci-skip-automatic-zero-init-of-large-ar.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch183: kvm-hw-scsi-lsi53c895a-skip-automatic-zero-init-of-large.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch184: kvm-hw-scsi-megasas-skip-automatic-zero-init-of-large-ar.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch185: kvm-hw-ufs-lu-skip-automatic-zero-init-of-large-array.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch186: kvm-net-socket-skip-automatic-zero-init-of-large-array.patch -# For RHEL-99887 - -ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z] -Patch187: kvm-net-stream-skip-automatic-zero-init-of-large-array.patch -# For RHEL-107314 - Improve VFIO mmapping performance with huge pfnmaps [rhel-9.6.z] -Patch188: kvm-vfio-helpers-Refactor-vfio_region_mmap-error-handlin.patch -# For RHEL-107314 - Improve VFIO mmapping performance with huge pfnmaps [rhel-9.6.z] -Patch189: kvm-vfio-helpers-Align-mmaps.patch -# For RHEL-108725 - Openstack guest becomes inaccessible via network when storage network on the hypervisor is disabled/lost [rhel-9.6.z] -Patch190: kvm-rbd-Fix-.bdrv_get_specific_info-implementation.patch +# For RHEL-7188 - [intel iommu][PF] DMAR: DRHD: handling fault status reg +Patch149: kvm-hw-virtio-virtio-iommu-Migrate-to-3-phase-reset.patch +# For RHEL-7188 - [intel iommu][PF] DMAR: DRHD: handling fault status reg +Patch150: kvm-hw-i386-intel-iommu-Migrate-to-3-phase-reset.patch +# For RHEL-7188 - [intel iommu][PF] DMAR: DRHD: handling fault status reg +Patch151: kvm-hw-arm-smmuv3-Move-reset-to-exit-phase.patch +# For RHEL-7188 - [intel iommu][PF] DMAR: DRHD: handling fault status reg +Patch152: kvm-hw-vfio-common-Add-a-trace-point-in-vfio_reset_handl.patch +# For RHEL-7188 - [intel iommu][PF] DMAR: DRHD: handling fault status reg +Patch153: kvm-docs-devel-reset-Document-reset-expectations-for-DMA.patch +# For RHEL-69622 - [qemu-guest-agent][RFE] Report CPU load average +Patch154: kvm-qga-implement-a-guest-get-load-command.patch +# For RHEL-69775 - Guest crashed on the target host when the migration was canceled +Patch155: kvm-migration-Fix-UAF-for-incoming-migration-on-Migratio.patch +# For RHEL-47340 - [Qemu RHEL-9] qemu-trace-stap should handle lack of stap more gracefully +Patch156: kvm-scripts-improve-error-from-qemu-trace-stap-on-missin.patch +# For RHEL-7301 - [intel iommu] VFIO_MAP_DMA failed: Bad address on system_powerdown +Patch157: kvm-hw-pci-Rename-has_power-to-enabled.patch +# For RHEL-7301 - [intel iommu] VFIO_MAP_DMA failed: Bad address on system_powerdown +Patch158: kvm-hw-pci-Basic-support-for-PCI-power-management.patch +# For RHEL-7301 - [intel iommu] VFIO_MAP_DMA failed: Bad address on system_powerdown +Patch159: kvm-pci-Use-PCI-PM-capability-initializer.patch +# For RHEL-7301 - [intel iommu] VFIO_MAP_DMA failed: Bad address on system_powerdown +Patch160: kvm-vfio-pci-Delete-local-pm_cap.patch +# For RHEL-7301 - [intel iommu] VFIO_MAP_DMA failed: Bad address on system_powerdown +Patch161: kvm-pcie-virtio-Remove-redundant-pm_cap.patch +# For RHEL-7301 - [intel iommu] VFIO_MAP_DMA failed: Bad address on system_powerdown +Patch162: kvm-hw-vfio-pci-Re-order-pre-reset.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch163: kvm-virtio-kconfig-memory-devices-are-PCI-only.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch164: kvm-hw-s390-ccw-device-Convert-to-three-phase-reset.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch165: kvm-hw-s390-virtio-ccw-Convert-to-three-phase-reset.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch166: kvm-target-s390-Convert-CPU-to-Resettable-interface.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch167: kvm-reset-Use-ResetType-for-qemu_devices_reset-and-Machi.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch168: kvm-reset-Add-RESET_TYPE_WAKEUP.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch169: kvm-virtio-mem-Use-new-Resettable-framework-instead-of-L.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch170: kvm-virtio-mem-Add-support-for-suspend-wake-up-with-plug.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch171: kvm-virtio-mem-unplug-memory-only-during-system-resets-n.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch172: kvm-s390x-s390-virtio-ccw-don-t-crash-on-weird-RAM-sizes.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch173: kvm-s390x-s390-virtio-hcall-remove-hypercall-registratio.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch174: kvm-s390x-s390-virtio-hcall-prepare-for-more-diag500-hyp.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch175: kvm-s390x-rename-s390-virtio-hcall-to-s390-hypercall.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch176: kvm-s390x-s390-virtio-ccw-move-setting-the-maximum-guest.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch177: kvm-s390x-introduce-s390_get_memory_limit.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch178: kvm-s390x-s390-hypercall-introduce-DIAG500-STORAGE_LIMIT.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch179: kvm-s390x-s390-stattrib-kvm-prepare-for-memory-devices-a.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch180: kvm-s390x-s390-skeys-prepare-for-memory-devices.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch181: kvm-s390x-s390-virtio-ccw-prepare-for-memory-devices.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch182: kvm-s390x-pv-prepare-for-memory-devices.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch183: kvm-s390x-remember-the-maximum-page-size.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch184: kvm-s390x-virtio-ccw-add-support-for-virtio-based-memory.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch185: kvm-s390x-virtio-mem-support.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch186: kvm-hw-virtio-Also-include-md-stubs-in-case-CONFIG_VIRTI.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch187: kvm-virtio-mem-don-t-warn-about-THP-sizes-on-a-kernel-wi.patch +# For RHEL-72977 - [IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part +Patch188: kvm-redhat-Enable-virtio-mem-on-s390x.patch +# For RHEL-7130 - [Hyper-V][RHEL9.2] Nested Hyper-V on KVM: L1 Windows VM with BIOS mode fails to boot up when using '-cpu host,hv_passthrough’ flag +Patch189: kvm-target-i386-Fix-conditional-CONFIG_SYNDBG-enablement.patch +# For RHEL-7130 - [Hyper-V][RHEL9.2] Nested Hyper-V on KVM: L1 Windows VM with BIOS mode fails to boot up when using '-cpu host,hv_passthrough’ flag +Patch190: kvm-target-i386-Exclude-hv-syndbg-from-hv-passthrough.patch +# For RHEL-80313 - Unable to migrate VM from RHEL10.0/qemu-kvm-9.6 to RHEL9.6/qemu-kvm-9.6 +Patch191: kvm-virtio-net-disable-USO-for-virt-rhel9.6.patch +# For RHEL-80313 - Unable to migrate VM from RHEL10.0/qemu-kvm-9.6 to RHEL9.6/qemu-kvm-9.6 +Patch192: kvm-arm-Use-arm_virt_compat_set-to-apply-the-compat.patch +# For RHEL-86032 - QEMU sends unaligned discards on 4K devices [RHEL-9.7] +Patch193: kvm-file-posix-probe-discard-alignment-on-Linux-block-de.patch +# For RHEL-86032 - QEMU sends unaligned discards on 4K devices [RHEL-9.7] +Patch194: kvm-block-io-skip-head-tail-requests-on-EINVAL.patch +# For RHEL-86032 - QEMU sends unaligned discards on 4K devices [RHEL-9.7] +Patch195: kvm-file-posix-Fix-crash-on-discard_granularity-0.patch +# For RHEL-88153 - [s390x] valgrind not working with qemu-kvm for non-x86 builds +Patch196: kvm-meson-configure-add-valgrind-option-en-dis-able-valg.patch +# For RHEL-88153 - [s390x] valgrind not working with qemu-kvm for non-x86 builds +Patch197: kvm-hw-i386-Fix-machine-type-compatibility.patch +# For RHEL-88533 - Improve VFIO mmapping performance with huge pfnmaps +Patch198: kvm-vfio-helpers-Refactor-vfio_region_mmap-error-handlin.patch +# For RHEL-88533 - Improve VFIO mmapping performance with huge pfnmaps +Patch199: kvm-vfio-helpers-Align-mmaps.patch +# For RHEL-85159 - Video stuck about 1 min after switchover phase when play one video during postcopy-preempt migration +Patch200: kvm-migration-postcopy-Spatial-locality-page-hint-for-pr.patch +# For RHEL-95120 - Allow libvirt to restart passt/vhost-user when the process is killed [rhel-9.7] +Patch201: kvm-net-vhost-user-add-QAPI-events-to-report-connection-.patch +# For RHEL-95408 - Support multipath failover with scsi-block [rhel-9] +Patch202: kvm-file-posix-Define-DM_MPATH_PROBE_PATHS.patch +# For RHEL-95408 - Support multipath failover with scsi-block [rhel-9] +Patch203: kvm-file-posix-Probe-paths-and-retry-SG_IO-on-potential-.patch +# For RHEL-11430 - [IBM 9.7 FEAT] KVM: Performance Enhanced Refresh PCI Translation - qemu part +Patch204: kvm-s390x-pci-add-support-for-guests-that-request-direct.patch +# For RHEL-11430 - [IBM 9.7 FEAT] KVM: Performance Enhanced Refresh PCI Translation - qemu part +Patch205: kvm-s390x-pci-indicate-QEMU-supports-relaxed-translation.patch +# For RHEL-82906 - --migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7] +# For RHEL-83015 - Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7] +Patch206: kvm-block-Expand-block-status-mode-from-bool-to-flags.patch +# For RHEL-82906 - --migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7] +# For RHEL-83015 - Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7] +Patch207: kvm-file-posix-gluster-Handle-zero-block-status-hint-bet.patch +# For RHEL-82906 - --migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7] +# For RHEL-83015 - Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7] +Patch208: kvm-block-Let-bdrv_co_is_zero_fast-consolidate-adjacent-.patch +# For RHEL-82906 - --migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7] +# For RHEL-83015 - Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7] +Patch209: kvm-block-Add-new-bdrv_co_is_all_zeroes-function.patch +# For RHEL-82906 - --migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7] +# For RHEL-83015 - Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7] +Patch210: kvm-iotests-Improve-iotest-194-to-mirror-data.patch +# For RHEL-82906 - --migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7] +# For RHEL-83015 - Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7] +Patch211: kvm-mirror-Minor-refactoring.patch +# For RHEL-82906 - --migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7] +# For RHEL-83015 - Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7] +Patch212: kvm-mirror-Pass-full-sync-mode-rather-than-bool-to-inter.patch +# For RHEL-82906 - --migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7] +# For RHEL-83015 - Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7] +Patch213: kvm-mirror-Allow-QMP-override-to-declare-target-already-.patch +# For RHEL-82906 - --migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7] +# For RHEL-83015 - Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7] +Patch214: kvm-mirror-Drop-redundant-zero_target-parameter.patch +# For RHEL-82906 - --migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7] +# For RHEL-83015 - Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7] +Patch215: kvm-mirror-Skip-pre-zeroing-destination-if-it-is-already.patch +# For RHEL-82906 - --migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7] +# For RHEL-83015 - Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7] +Patch216: kvm-mirror-Skip-writing-zeroes-when-target-is-already-ze.patch +# For RHEL-82906 - --migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7] +# For RHEL-83015 - Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7] +Patch217: kvm-iotests-common.rc-add-disk_usage-function.patch +# For RHEL-82906 - --migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7] +# For RHEL-83015 - Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7] +Patch218: kvm-tests-Add-iotest-mirror-sparse-for-recent-patches.patch +# For RHEL-82906 - --migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7] +# For RHEL-83015 - Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7] +Patch219: kvm-mirror-Reduce-I-O-when-destination-is-detect-zeroes-.patch +# For RHEL-98554 - [s390x][RHEL9.7.0][virtio_block] there would be memory leak with virtio_blk disks +Patch220: kvm-s390x-Fix-leak-in-machine_set_loadparm.patch +# For RHEL-98554 - [s390x][RHEL9.7.0][virtio_block] there would be memory leak with virtio_blk disks +Patch221: kvm-hw-s390x-ccw-device-Fix-memory-leak-in-loadparm-sett.patch +# For RHEL-66202 - [AMDSERVER 9.6 Feature] qemu: Interrupt Remap support for emulated amd viommu +Patch222: kvm-amd_iommu-Rename-variable-mmio-to-mr_mmio.patch +# For RHEL-66202 - [AMDSERVER 9.6 Feature] qemu: Interrupt Remap support for emulated amd viommu +Patch223: kvm-amd_iommu-Add-support-for-pass-though-mode.patch +# For RHEL-66202 - [AMDSERVER 9.6 Feature] qemu: Interrupt Remap support for emulated amd viommu +Patch224: kvm-amd_iommu-Use-shared-memory-region-for-Interrupt-Rem.patch +# For RHEL-66202 - [AMDSERVER 9.6 Feature] qemu: Interrupt Remap support for emulated amd viommu +Patch225: kvm-amd_iommu-Send-notification-when-invalidate-interrup.patch +# For RHEL-66202 - [AMDSERVER 9.6 Feature] qemu: Interrupt Remap support for emulated amd viommu +Patch226: kvm-amd_iommu-Check-APIC-ID-255-for-XTSup.patch +# For RHEL-67104 - postcopy on the destination host can't switch into pause status under the network issue if boot VM with '-S' +Patch227: kvm-io-Fix-partial-struct-copy-in-qio_dns_resolver_looku.patch +# For RHEL-67104 - postcopy on the destination host can't switch into pause status under the network issue if boot VM with '-S' +Patch228: kvm-util-qemu-sockets-Refactor-setting-client-sockopts-i.patch +# For RHEL-67104 - postcopy on the destination host can't switch into pause status under the network issue if boot VM with '-S' +Patch229: kvm-util-qemu-sockets-Refactor-success-and-failure-paths.patch +# For RHEL-67104 - postcopy on the destination host can't switch into pause status under the network issue if boot VM with '-S' +Patch230: kvm-util-qemu-sockets-Add-support-for-keep-alive-flag-to.patch +# For RHEL-67104 - postcopy on the destination host can't switch into pause status under the network issue if boot VM with '-S' +Patch231: kvm-util-qemu-sockets-Refactor-inet_parse-to-use-QemuOpt.patch +# For RHEL-67104 - postcopy on the destination host can't switch into pause status under the network issue if boot VM with '-S' +Patch232: kvm-util-qemu-sockets-Introduce-inet-socket-options-cont.patch +# For RHEL-67104 - postcopy on the destination host can't switch into pause status under the network issue if boot VM with '-S' +Patch233: kvm-tests-unit-test-util-sockets-fix-mem-leak-on-error-o.patch +# For RHEL-52649 - [AMDSERVER 9.6 Feature] Turin: Qemu EPYC-Turin Model +Patch234: kvm-target-i386-Expose-bits-related-to-SRSO-vulnerabilit.patch +# For RHEL-52649 - [AMDSERVER 9.6 Feature] Turin: Qemu EPYC-Turin Model +Patch235: kvm-target-i386-Add-PerfMonV2-feature-bit.patch +# For RHEL-52649 - [AMDSERVER 9.6 Feature] Turin: Qemu EPYC-Turin Model +Patch236: kvm-target-i386-Update-EPYC-CPU-model-for-Cache-property.patch +# For RHEL-52649 - [AMDSERVER 9.6 Feature] Turin: Qemu EPYC-Turin Model +Patch237: kvm-target-i386-Update-EPYC-Rome-CPU-model-for-Cache-pro.patch +# For RHEL-52649 - [AMDSERVER 9.6 Feature] Turin: Qemu EPYC-Turin Model +Patch238: kvm-target-i386-Update-EPYC-Milan-CPU-model-for-Cache-pr.patch +# For RHEL-52649 - [AMDSERVER 9.6 Feature] Turin: Qemu EPYC-Turin Model +Patch239: kvm-target-i386-Add-couple-of-feature-bits-in-CPUID_Fn80.patch +# For RHEL-52649 - [AMDSERVER 9.6 Feature] Turin: Qemu EPYC-Turin Model +Patch240: kvm-target-i386-Update-EPYC-Genoa-for-Cache-property-per.patch +# For RHEL-52649 - [AMDSERVER 9.6 Feature] Turin: Qemu EPYC-Turin Model +Patch241: kvm-target-i386-Add-support-for-EPYC-Turin-model.patch +# For RHEL-70926 - Qemu/amd-iommu: Advertise a suitable device id +Patch242: kvm-hw-i386-amd_iommu-Assign-pci-id-0x1419-for-the-AMD-I.patch +# For RHEL-70925 - Qemu/amd-iommu: Add ability to manually specify the AMDVI-PCI device +Patch243: kvm-hw-i386-amd_iommu-Isolate-AMDVI-PCI-from-amd-iommu-d.patch +# For RHEL-70925 - Qemu/amd-iommu: Add ability to manually specify the AMDVI-PCI device +Patch244: kvm-hw-i386-amd_iommu-Allow-migration-when-explicitly-cr.patch +# For RHEL-70925 - Qemu/amd-iommu: Add ability to manually specify the AMDVI-PCI device +Patch245: kvm-Enable-amd-iommu-device.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch246: kvm-include-qemu-compiler-add-QEMU_UNINITIALIZED-attribu.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch247: kvm-hw-virtio-virtio-avoid-cost-of-ftrivial-auto-var-ini.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch248: kvm-block-skip-automatic-zero-init-of-large-array-in-ioq.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch249: kvm-chardev-char-fd-skip-automatic-zero-init-of-large-ar.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch250: kvm-chardev-char-pty-skip-automatic-zero-init-of-large-a.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch251: kvm-chardev-char-socket-skip-automatic-zero-init-of-larg.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch252: kvm-hw-audio-ac97-skip-automatic-zero-init-of-large-arra.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch253: kvm-hw-audio-cs4231a-skip-automatic-zero-init-of-large-a.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch254: kvm-hw-audio-es1370-skip-automatic-zero-init-of-large-ar.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch255: kvm-hw-audio-gus-skip-automatic-zero-init-of-large-array.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch256: kvm-hw-audio-marvell_88w8618-skip-automatic-zero-init-of.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch257: kvm-hw-audio-sb16-skip-automatic-zero-init-of-large-arra.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch258: kvm-hw-audio-via-ac97-skip-automatic-zero-init-of-large-.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch259: kvm-hw-char-sclpconsole-lm-skip-automatic-zero-init-of-l.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch260: kvm-hw-dma-xlnx_csu_dma-skip-automatic-zero-init-of-larg.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch261: kvm-hw-display-vmware_vga-skip-automatic-zero-init-of-la.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch262: kvm-hw-hyperv-syndbg-skip-automatic-zero-init-of-large-a.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch263: kvm-hw-misc-aspeed_hace-skip-automatic-zero-init-of-larg.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch264: kvm-hw-net-rtl8139-skip-automatic-zero-init-of-large-arr.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch265: kvm-hw-net-tulip-skip-automatic-zero-init-of-large-array.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch266: kvm-hw-net-virtio-net-skip-automatic-zero-init-of-large-.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch267: kvm-hw-net-xgamc-skip-automatic-zero-init-of-large-array.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch268: kvm-hw-nvme-ctrl-skip-automatic-zero-init-of-large-array.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch269: kvm-hw-ppc-spapr_tpm_proxy-skip-automatic-zero-init-of-l.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch270: kvm-hw-usb-hcd-ohci-skip-automatic-zero-init-of-large-ar.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch271: kvm-hw-scsi-lsi53c895a-skip-automatic-zero-init-of-large.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch272: kvm-hw-scsi-megasas-skip-automatic-zero-init-of-large-ar.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch273: kvm-hw-ufs-lu-skip-automatic-zero-init-of-large-array.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch274: kvm-net-socket-skip-automatic-zero-init-of-large-array.patch +# For RHEL-99888 - -ftrivial-auto-var-init=zero reduced performance [rhel-9] +Patch275: kvm-net-stream-skip-automatic-zero-init-of-large-array.patch +# For RHEL-100741 - Video stuck after switchover phase when play one video during migration [rhel-9] +Patch276: kvm-ui-vnc-Update-display-update-interval-when-VM-state-.patch +# For RHEL-108726 - Openstack guest becomes inaccessible via network when storage network on the hypervisor is disabled/lost [rhel-9] +Patch277: kvm-rbd-Fix-.bdrv_get_specific_info-implementation.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch278: kvm-target-i386-Make-invtsc-migratable-when-user-sets-ts.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch279: kvm-target-i386-Enable-fdp-excptn-only-and-zero-fcs-fds.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch280: kvm-kvm-i386-make-kvm_filter_msr-and-related-definitions.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch281: kvm-kvm-remove-unnecessary-ifdef.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch282: kvm-crypto-Define-macros-for-hash-algorithm-digest-lengt.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch283: kvm-i386-cpu-Drop-the-check-of-phys_bits-in-host_cpu_rea.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch284: kvm-i386-cpu-Extract-a-common-fucntion-to-setup-value-of.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch285: kvm-i386-cpu-Drop-the-variable-smp_cores-and-smp_threads.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch286: kvm-i386-cpu-Drop-cores_per_pkg-in-cpu_x86_cpuid.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch287: kvm-i386-topology-Update-the-comment-of-x86_apicid_from_.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch288: kvm-i386-topology-Introduce-helpers-for-various-topology.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch289: kvm-i386-cpu-Track-a-X86CPUTopoInfo-directly-in-CPUX86St.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch290: kvm-i386-cpu-Hoist-check-of-CPUID_EXT3_TOPOEXT-against-t.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch291: kvm-cpu-Remove-nr_cores-from-struct-CPUState.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch292: kvm-i386-cpu-Set-up-CPUID_HT-in-x86_cpu_expand_features-.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch293: kvm-i386-cpu-Set-and-track-CPUID_EXT3_CMP_LEG-in-env-fea.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch294: kvm-i386-Remove-unused-parameter-uint32_t-bit-in-feature.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch295: kvm-target-i386-Print-CPUID-subleaf-info-for-unsupported.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch296: kvm-target-i386-sev-Reduce-system-specific-declarations.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch297: kvm-physmem-replace-assertion-with-error.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch298: kvm-redhat-target-i386-add-CPUID-and-MSR-bits-from-Clear.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch299: kvm-qom-reverse-order-of-instance_post_init-calls.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch300: kvm-target-i386-Remove-AccelCPUClass-cpu_class_init-need.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch301: kvm-i386-cpu-Consolidate-the-helper-to-get-Host-s-vendor.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch302: kvm-rocker-do-not-pollute-the-namespace.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch303: kvm-linux-headers-Update-to-Linux-v6.14-rc3.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch304: kvm-linux-headers-Update-to-Linux-v6.15-rc3.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch305: kvm-linux-headers-update-from-6.15-kvm-next.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch306: kvm-update-Linux-headers-to-v6.16-rc3.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch307: kvm-update-Linux-headers-to-KVM-tree-master.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch308: kvm-i386-Introduce-tdx-guest-object.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch309: kvm-i386-tdx-Implement-tdx_kvm_type-for-TDX.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch310: kvm-i386-tdx-Implement-tdx_kvm_init-to-initialize-TDX-VM.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch311: kvm-i386-tdx-Get-tdx_capabilities-via-KVM_TDX_CAPABILITI.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch312: kvm-i386-tdx-Introduce-is_tdx_vm-helper-and-cache-tdx_gu.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch313: kvm-kvm-Introduce-kvm_arch_pre_create_vcpu.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch314: kvm-i386-tdx-Initialize-TDX-before-creating-TD-vcpus.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch315: kvm-i386-tdx-Add-property-sept-ve-disable-for-tdx-guest-.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch316: kvm-i386-tdx-Make-sept_ve_disable-set-by-default.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch317: kvm-i386-tdx-Wire-CPU-features-up-with-attributes-of-TD-.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch318: kvm-i386-tdx-Validate-TD-attributes.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch319: kvm-i386-tdx-Support-user-configurable-mrconfigid-mrowne.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch320: kvm-i386-tdx-Set-APIC-bus-rate-to-match-with-what-TDX-mo.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch321: kvm-i386-tdx-Implement-user-specified-tsc-frequency.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch322: kvm-i386-tdx-load-TDVF-for-TD-guest.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch323: kvm-i386-tdvf-Introduce-function-to-parse-TDVF-metadata.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch324: kvm-i386-tdx-Parse-TDVF-metadata-for-TDX-VM.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch325: kvm-i386-tdx-Don-t-initialize-pc.rom-for-TDX-VMs.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch326: kvm-i386-tdx-Track-mem_ptr-for-each-firmware-entry-of-TD.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch327: kvm-i386-tdx-Track-RAM-entries-for-TDX-VM.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch328: kvm-headers-Add-definitions-from-UEFI-spec-for-volumes-r.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch329: kvm-i386-tdx-Setup-the-TD-HOB-list.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch330: kvm-i386-tdx-Add-TDVF-memory-via-KVM_TDX_INIT_MEM_REGION.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch331: kvm-i386-tdx-Call-KVM_TDX_INIT_VCPU-to-initialize-TDX-vc.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch332: kvm-i386-tdx-Finalize-TDX-VM.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch333: kvm-i386-tdx-Enable-user-exit-on-KVM_HC_MAP_GPA_RANGE.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch334: kvm-i386-tdx-Handle-KVM_SYSTEM_EVENT_TDX_FATAL.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch335: kvm-i386-tdx-Wire-TDX_REPORT_FATAL_ERROR-with-GuestPanic.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch336: kvm-kvm-Check-KVM_CAP_MAX_VCPUS-at-vm-level.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch337: kvm-i386-cpu-introduce-x86_confidential_guest_cpu_instan.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch338: kvm-i386-tdx-implement-tdx_cpu_instance_init.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch339: kvm-i386-cpu-Introduce-enable_cpuid_0x1f-to-force-exposi.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch340: kvm-i386-tdx-Force-exposing-CPUID-0x1f.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch341: kvm-i386-tdx-Set-kvm_readonly_mem_enabled-to-false-for-T.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch342: kvm-i386-tdx-Disable-SMM-for-TDX-VMs.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch343: kvm-i386-tdx-Disable-PIC-for-TDX-VMs.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch344: kvm-i386-tdx-Set-and-check-kernel_irqchip-mode-for-TDX.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch345: kvm-i386-tdx-Don-t-synchronize-guest-tsc-for-TDs.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch346: kvm-i386-tdx-Only-configure-MSR_IA32_UCODE_REV-in-kvm_in.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch347: kvm-i386-apic-Skip-kvm_apic_put-for-TDX.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch348: kvm-cpu-Don-t-set-vcpu_dirty-when-guest_state_protected.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch349: kvm-i386-cgs-Rename-mask_cpuid_features-to-adjust_cpuid_.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch350: kvm-i386-tdx-Implement-adjust_cpuid_features-for-TDX.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch351: kvm-i386-tdx-Add-TDX-fixed1-bits-to-supported-CPUIDs.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch352: kvm-i386-tdx-Add-supported-CPUID-bits-related-to-TD-Attr.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch353: kvm-i386-tdx-Add-supported-CPUID-bits-relates-to-XFAM.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch354: kvm-i386-tdx-Add-XFD-to-supported-bit-of-TDX.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch355: kvm-i386-tdx-Define-supported-KVM-features-for-TDX.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch356: kvm-i386-cgs-Introduce-x86_confidential_guest_check_feat.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch357: kvm-i386-tdx-Fetch-and-validate-CPUID-of-TD-guest.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch358: kvm-i386-tdx-Don-t-treat-SYSCALL-as-unavailable.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch359: kvm-i386-tdx-Make-invtsc-default-on.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch360: kvm-i386-tdx-Validate-phys_bits-against-host-value.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch361: kvm-docs-Add-TDX-documentation.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch362: kvm-i386-tdx-Fix-build-on-32-bit-host.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch363: kvm-i386-tdvf-Fix-build-on-32-bit-host.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch364: kvm-i386-cpu-Move-adjustment-of-CPUID_EXT_PDCM-before-fe.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch365: kvm-i386-tdx-Error-and-exit-when-named-cpu-model-is-requ.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch366: kvm-i386-cpu-Rename-enable_cpuid_0x1f-to-force_cpuid_0x1.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch367: kvm-i386-tdx-Fix-the-typo-of-the-comment-of-struct-TdxGu.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch368: kvm-i386-tdx-Clarify-the-error-message-of-mrconfigid-mro.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch369: kvm-i386-tdx-handle-TDG.VP.VMCALL-GetTdVmCallInfo.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch370: kvm-i386-tdx-handle-TDG.VP.VMCALL-GetQuote.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch371: kvm-target-i386-move-max_features-to-class.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch372: kvm-target-i386-nvmm-whpx-add-accel-CPU-class-that-sets-.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch373: kvm-target-i386-allow-reordering-max_x86_cpu_initfn-vs-a.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch374: kvm-target-i386-move-accel_cpu_instance_init-to-.instanc.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch375: kvm-target-i386-merge-host_cpu_instance_init-and-host_cp.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch376: kvm-i386-tdx-Remove-enumeration-of-GetQuote-in-tdx_handl.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch377: kvm-i386-tdx-Set-value-of-GetTdVmCallInfo-based-on-capab.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch378: kvm-i386-tdx-handle-TDVMCALL_SETUP_EVENT_NOTIFY_INTERRUP.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch379: kvm-i386-tdx-Fix-the-report-of-gpa-in-QAPI.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch380: kvm-i386-tdx-Remove-task-watch-only-when-it-s-valid.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch381: kvm-i386-tdx-Don-t-mask-off-CPUID_EXT_PDCM.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch382: kvm-i386-cpu-Move-x86_ext_save_areas-initialization-to-..patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch383: kvm-target-i386-tdx-fix-locking-for-interrupt-injection.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch384: kvm-i386-cpu-Cleanup-host_cpu_max_instance_init.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch385: kvm-i386-tdx-Remove-the-redundant-qemu_mutex_init-tdx-lo.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch386: kvm-redhat-enable-CONFIG_TDX.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch387: kvm-redhat-allow-5-level-paging-for-TDX-VMs.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch388: kvm-memory-Export-a-helper-to-get-intersection-of-a-Memo.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch389: kvm-memory-Change-memory_region_set_ram_discard_manager-.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch390: kvm-memory-Unify-the-definiton-of-ReplayRamPopulate-and-.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch391: kvm-ram-block-attributes-Introduce-RamBlockAttributes-to.patch +# For RHEL-15710 - [Intel 9.7 FEAT] TDX: QEMU Support +# For RHEL-20798 - [Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD +# For RHEL-49728 - [Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock +Patch392: kvm-physmem-Support-coordinated-discarding-of-RAM-with-g.patch +# For RHEL-17614 - VM reports Vulnerable to spec_rstack_overflow when reading status in '/sys/devices/system/cpu/vulnerabilities/' +Patch393: kvm-target-i386-Expose-IBPB-BRTYPE-and-SBPB-CPUID-bits-t.patch +# For RHEL-120502 - [rhel9] Backport "arm/kvm: report registers we failed to set" [rhel-9.7.z] +Patch394: kvm-arm-kvm-report-registers-we-failed-to-set.patch +# For RHEL-120125 - CVE-2025-11234 qemu-kvm: VNC WebSocket handshake use-after-free [rhel-9.7.z] +Patch395: kvm-io-move-websock-resource-release-to-close-method.patch +# For RHEL-120125 - CVE-2025-11234 qemu-kvm: VNC WebSocket handshake use-after-free [rhel-9.7.z] +Patch396: kvm-io-fix-use-after-free-in-websocket-handshake-code.patch %if %{have_clang} BuildRequires: clang @@ -626,6 +1282,9 @@ BuildRequires: pulseaudio-libs-devel BuildRequires: spice-protocol BuildRequires: capstone-devel BuildRequires: python3-tomli +%ifarch %{valgrind_arches} +BuildRequires: valgrind-devel +%endif # Requires for qemu-kvm package Requires: %{name}-core = %{epoch}:%{version}-%{release} @@ -705,6 +1364,8 @@ This package provides documentation and auxiliary programs used with %{name}. %package tools Summary: %{name} support tools +Recommends: systemtap-client +Recommends: systemtap-devel %description tools %{name}-tools provides various tools related to %{name} usage. @@ -1000,6 +1661,7 @@ ulimit -n 10240 --disable-u2f \\\ --disable-usb-redir \\\ --disable-user \\\ + --disable-valgrind \\\ --disable-vde \\\ --disable-vdi \\\ --disable-vduse-blk-export \\\ @@ -1122,6 +1784,9 @@ run_configure \ --enable-tpm \ %if %{have_usbredir} --enable-usb-redir \ +%endif +%ifarch %{valgrind_arches} + --enable-valgrind \ %endif --enable-vdi \ --enable-vhost-kernel \ @@ -1615,81 +2280,370 @@ useradd -r -u 107 -g qemu -G kvm -d / -s /sbin/nologin \ %endif %changelog -* Mon Aug 18 2025 Jon Maloy - 9.1.0-15.el9_6.9 -- kvm-rbd-Fix-.bdrv_get_specific_info-implementation.patch [RHEL-108725] -- Resolves: RHEL-108725 - (Openstack guest becomes inaccessible via network when storage network on the hypervisor is disabled/lost [rhel-9.6.z]) +* Mon Nov 17 2025 Jon Maloy - 9.1.0-29.el9_7.3 +- kvm-io-move-websock-resource-release-to-close-method.patch [RHEL-120125] +- kvm-io-fix-use-after-free-in-websocket-handshake-code.patch [RHEL-120125] +- Resolves: RHEL-120125 + (CVE-2025-11234 qemu-kvm: VNC WebSocket handshake use-after-free [rhel-9.7.z]) -* Tue Aug 05 2025 Jon Maloy - 9.1.0-15.el9_6.8 -- kvm-vfio-helpers-Refactor-vfio_region_mmap-error-handlin.patch [RHEL-107314] -- kvm-vfio-helpers-Align-mmaps.patch [RHEL-107314] -- Resolves: RHEL-107314 - (Improve VFIO mmapping performance with huge pfnmaps [rhel-9.6.z]) +* Wed Nov 12 2025 Jon Maloy - 9.1.0-29.el9_7.2 +- kvm-io-move-websock-resource-release-to-close-method.patch [RHEL-120125] +- kvm-io-fix-use-after-free-in-websocket-handshake-code.patch [RHEL-120125] +- Resolves: RHEL-120125 + (CVE-2025-11234 qemu-kvm: VNC WebSocket handshake use-after-free [rhel-9.7.z]) -* Fri Jul 04 2025 Miroslav Rezanina - 9.1.0-15.el9_6.7 -- kvm-ui-vnc-Update-display-update-interval-when-VM-state-.patch [RHEL-100767] -- kvm-include-qemu-compiler-add-QEMU_UNINITIALIZED-attribu.patch [RHEL-99887] -- kvm-hw-virtio-virtio-avoid-cost-of-ftrivial-auto-var-ini.patch [RHEL-99887] -- kvm-block-skip-automatic-zero-init-of-large-array-in-ioq.patch [RHEL-99887] -- kvm-chardev-char-fd-skip-automatic-zero-init-of-large-ar.patch [RHEL-99887] -- kvm-chardev-char-pty-skip-automatic-zero-init-of-large-a.patch [RHEL-99887] -- kvm-chardev-char-socket-skip-automatic-zero-init-of-larg.patch [RHEL-99887] -- kvm-hw-audio-ac97-skip-automatic-zero-init-of-large-arra.patch [RHEL-99887] -- kvm-hw-audio-cs4231a-skip-automatic-zero-init-of-large-a.patch [RHEL-99887] -- kvm-hw-audio-es1370-skip-automatic-zero-init-of-large-ar.patch [RHEL-99887] -- kvm-hw-audio-gus-skip-automatic-zero-init-of-large-array.patch [RHEL-99887] -- kvm-hw-audio-marvell_88w8618-skip-automatic-zero-init-of.patch [RHEL-99887] -- kvm-hw-audio-sb16-skip-automatic-zero-init-of-large-arra.patch [RHEL-99887] -- kvm-hw-audio-via-ac97-skip-automatic-zero-init-of-large-.patch [RHEL-99887] -- kvm-hw-char-sclpconsole-lm-skip-automatic-zero-init-of-l.patch [RHEL-99887] -- kvm-hw-dma-xlnx_csu_dma-skip-automatic-zero-init-of-larg.patch [RHEL-99887] -- kvm-hw-display-vmware_vga-skip-automatic-zero-init-of-la.patch [RHEL-99887] -- kvm-hw-hyperv-syndbg-skip-automatic-zero-init-of-large-a.patch [RHEL-99887] -- kvm-hw-misc-aspeed_hace-skip-automatic-zero-init-of-larg.patch [RHEL-99887] -- kvm-hw-net-rtl8139-skip-automatic-zero-init-of-large-arr.patch [RHEL-99887] -- kvm-hw-net-tulip-skip-automatic-zero-init-of-large-array.patch [RHEL-99887] -- kvm-hw-net-virtio-net-skip-automatic-zero-init-of-large-.patch [RHEL-99887] -- kvm-hw-net-xgamc-skip-automatic-zero-init-of-large-array.patch [RHEL-99887] -- kvm-hw-nvme-ctrl-skip-automatic-zero-init-of-large-array.patch [RHEL-99887] -- kvm-hw-ppc-spapr_tpm_proxy-skip-automatic-zero-init-of-l.patch [RHEL-99887] -- kvm-hw-usb-hcd-ohci-skip-automatic-zero-init-of-large-ar.patch [RHEL-99887] -- kvm-hw-scsi-lsi53c895a-skip-automatic-zero-init-of-large.patch [RHEL-99887] -- kvm-hw-scsi-megasas-skip-automatic-zero-init-of-large-ar.patch [RHEL-99887] -- kvm-hw-ufs-lu-skip-automatic-zero-init-of-large-array.patch [RHEL-99887] -- kvm-net-socket-skip-automatic-zero-init-of-large-array.patch [RHEL-99887] -- kvm-net-stream-skip-automatic-zero-init-of-large-array.patch [RHEL-99887] -- Resolves: RHEL-100767 - (Video stuck after switchover phase when play one video during migration [rhel-9.6.z]) -- Resolves: RHEL-99887 - (-ftrivial-auto-var-init=zero reduced performance [rhel-9.6.z]) +* Mon Nov 03 2025 Jon Maloy - 9.1.0-29.el9_7.1 +- kvm-arm-kvm-report-registers-we-failed-to-set.patch [RHEL-120502] +- Resolves: RHEL-120502 + ([rhel9] Backport "arm/kvm: report registers we failed to set" [rhel-9.7.z]) -* Mon Jun 09 2025 Jon Maloy - 9.1.0-15.el9_6.6 -- kvm-file-posix-Define-DM_MPATH_PROBE_PATHS.patch [RHEL-95407] -- kvm-file-posix-Probe-paths-and-retry-SG_IO-on-potential-.patch [RHEL-95407] -- Resolves: RHEL-95407 - (Support multipath failover with scsi-block [rhel-9.6.z]) +* Tue Sep 16 2025 Jon Maloy - 9.1.0-29 +- kvm-target-i386-Expose-IBPB-BRTYPE-and-SBPB-CPUID-bits-t.patch [RHEL-17614] +- Resolves: RHEL-17614 + (VM reports Vulnerable to spec_rstack_overflow when reading status in '/sys/devices/system/cpu/vulnerabilities/') -* Mon May 26 2025 Jon Maloy - 9.1.0-15.el9_6.5 -- kvm-hw-i386-Fix-machine-type-compatibility.patch [RHEL-92077] -- Resolves: RHEL-92077 - (Fix x86 M-type compats [rhel-9.6.z]) +* Mon Sep 15 2025 Jon Maloy - 9.1.0-28 +- kvm-target-i386-Expose-IBPB-BRTYPE-and-SBPB-CPUID-bits-t.patch [RHEL-17614] +- Resolves: RHEL-17614 + (VM reports Vulnerable to spec_rstack_overflow when reading status in '/sys/devices/system/cpu/vulnerabilities/') -* Mon May 05 2025 Jon Maloy - 9.1.0-15.el9_6.4 -- kvm-file-posix-probe-discard-alignment-on-Linux-block-de.patch [RHEL-87734] -- kvm-block-io-skip-head-tail-requests-on-EINVAL.patch [RHEL-87734] -- kvm-file-posix-Fix-crash-on-discard_granularity-0.patch [RHEL-87734] -- Resolves: RHEL-87734 - (QEMU sends unaligned discards on 4K devices [rhel-9.6.z]) +* Tue Sep 09 2025 Jon Maloy - 9.1.0-27 +- kvm-target-i386-Make-invtsc-migratable-when-user-sets-ts.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-target-i386-Enable-fdp-excptn-only-and-zero-fcs-fds.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-kvm-i386-make-kvm_filter_msr-and-related-definitions.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-kvm-remove-unnecessary-ifdef.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-crypto-Define-macros-for-hash-algorithm-digest-lengt.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-Drop-the-check-of-phys_bits-in-host_cpu_rea.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-Extract-a-common-fucntion-to-setup-value-of.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-Drop-the-variable-smp_cores-and-smp_threads.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-Drop-cores_per_pkg-in-cpu_x86_cpuid.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-topology-Update-the-comment-of-x86_apicid_from_.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-topology-Introduce-helpers-for-various-topology.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-Track-a-X86CPUTopoInfo-directly-in-CPUX86St.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-Hoist-check-of-CPUID_EXT3_TOPOEXT-against-t.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-cpu-Remove-nr_cores-from-struct-CPUState.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-Set-up-CPUID_HT-in-x86_cpu_expand_features-.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-Set-and-track-CPUID_EXT3_CMP_LEG-in-env-fea.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-Remove-unused-parameter-uint32_t-bit-in-feature.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-target-i386-Print-CPUID-subleaf-info-for-unsupported.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-target-i386-sev-Reduce-system-specific-declarations.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-physmem-replace-assertion-with-error.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-redhat-target-i386-add-CPUID-and-MSR-bits-from-Clear.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-qom-reverse-order-of-instance_post_init-calls.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-target-i386-Remove-AccelCPUClass-cpu_class_init-need.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-Consolidate-the-helper-to-get-Host-s-vendor.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-rocker-do-not-pollute-the-namespace.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-linux-headers-Update-to-Linux-v6.14-rc3.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-linux-headers-Update-to-Linux-v6.15-rc3.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-linux-headers-update-from-6.15-kvm-next.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-update-Linux-headers-to-v6.16-rc3.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-update-Linux-headers-to-KVM-tree-master.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-Introduce-tdx-guest-object.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Implement-tdx_kvm_type-for-TDX.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Implement-tdx_kvm_init-to-initialize-TDX-VM.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Get-tdx_capabilities-via-KVM_TDX_CAPABILITI.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Introduce-is_tdx_vm-helper-and-cache-tdx_gu.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-kvm-Introduce-kvm_arch_pre_create_vcpu.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Initialize-TDX-before-creating-TD-vcpus.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Add-property-sept-ve-disable-for-tdx-guest-.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Make-sept_ve_disable-set-by-default.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Wire-CPU-features-up-with-attributes-of-TD-.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Validate-TD-attributes.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Support-user-configurable-mrconfigid-mrowne.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Set-APIC-bus-rate-to-match-with-what-TDX-mo.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Implement-user-specified-tsc-frequency.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-load-TDVF-for-TD-guest.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdvf-Introduce-function-to-parse-TDVF-metadata.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Parse-TDVF-metadata-for-TDX-VM.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Don-t-initialize-pc.rom-for-TDX-VMs.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Track-mem_ptr-for-each-firmware-entry-of-TD.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Track-RAM-entries-for-TDX-VM.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-headers-Add-definitions-from-UEFI-spec-for-volumes-r.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Setup-the-TD-HOB-list.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Add-TDVF-memory-via-KVM_TDX_INIT_MEM_REGION.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Call-KVM_TDX_INIT_VCPU-to-initialize-TDX-vc.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Finalize-TDX-VM.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Enable-user-exit-on-KVM_HC_MAP_GPA_RANGE.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Handle-KVM_SYSTEM_EVENT_TDX_FATAL.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Wire-TDX_REPORT_FATAL_ERROR-with-GuestPanic.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-kvm-Check-KVM_CAP_MAX_VCPUS-at-vm-level.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-introduce-x86_confidential_guest_cpu_instan.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-implement-tdx_cpu_instance_init.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-Introduce-enable_cpuid_0x1f-to-force-exposi.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Force-exposing-CPUID-0x1f.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Set-kvm_readonly_mem_enabled-to-false-for-T.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Disable-SMM-for-TDX-VMs.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Disable-PIC-for-TDX-VMs.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Set-and-check-kernel_irqchip-mode-for-TDX.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Don-t-synchronize-guest-tsc-for-TDs.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Only-configure-MSR_IA32_UCODE_REV-in-kvm_in.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-apic-Skip-kvm_apic_put-for-TDX.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-cpu-Don-t-set-vcpu_dirty-when-guest_state_protected.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cgs-Rename-mask_cpuid_features-to-adjust_cpuid_.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Implement-adjust_cpuid_features-for-TDX.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Add-TDX-fixed1-bits-to-supported-CPUIDs.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Add-supported-CPUID-bits-related-to-TD-Attr.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Add-supported-CPUID-bits-relates-to-XFAM.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Add-XFD-to-supported-bit-of-TDX.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Define-supported-KVM-features-for-TDX.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cgs-Introduce-x86_confidential_guest_check_feat.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Fetch-and-validate-CPUID-of-TD-guest.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Don-t-treat-SYSCALL-as-unavailable.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Make-invtsc-default-on.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Validate-phys_bits-against-host-value.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-docs-Add-TDX-documentation.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Fix-build-on-32-bit-host.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdvf-Fix-build-on-32-bit-host.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-Move-adjustment-of-CPUID_EXT_PDCM-before-fe.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Error-and-exit-when-named-cpu-model-is-requ.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-Rename-enable_cpuid_0x1f-to-force_cpuid_0x1.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Fix-the-typo-of-the-comment-of-struct-TdxGu.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Clarify-the-error-message-of-mrconfigid-mro.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-handle-TDG.VP.VMCALL-GetTdVmCallInfo.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-handle-TDG.VP.VMCALL-GetQuote.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-target-i386-move-max_features-to-class.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-target-i386-nvmm-whpx-add-accel-CPU-class-that-sets-.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-target-i386-allow-reordering-max_x86_cpu_initfn-vs-a.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-target-i386-move-accel_cpu_instance_init-to-.instanc.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-target-i386-merge-host_cpu_instance_init-and-host_cp.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Remove-enumeration-of-GetQuote-in-tdx_handl.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Set-value-of-GetTdVmCallInfo-based-on-capab.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-handle-TDVMCALL_SETUP_EVENT_NOTIFY_INTERRUP.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Fix-the-report-of-gpa-in-QAPI.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Remove-task-watch-only-when-it-s-valid.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Don-t-mask-off-CPUID_EXT_PDCM.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-Move-x86_ext_save_areas-initialization-to-..patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-target-i386-tdx-fix-locking-for-interrupt-injection.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-cpu-Cleanup-host_cpu_max_instance_init.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-i386-tdx-Remove-the-redundant-qemu_mutex_init-tdx-lo.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-redhat-enable-CONFIG_TDX.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-redhat-allow-5-level-paging-for-TDX-VMs.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-memory-Export-a-helper-to-get-intersection-of-a-Memo.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-memory-Change-memory_region_set_ram_discard_manager-.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-memory-Unify-the-definiton-of-ReplayRamPopulate-and-.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-ram-block-attributes-Introduce-RamBlockAttributes-to.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- kvm-physmem-Support-coordinated-discarding-of-RAM-with-g.patch [RHEL-15710 RHEL-20798 RHEL-49728] +- Resolves: RHEL-15710 + ([Intel 9.7 FEAT] TDX: QEMU Support) +- Resolves: RHEL-20798 + ([Intel 9.6 FEAT] TDX: host: Virt-QEMU: Add safe device pass-through for TD) +- Resolves: RHEL-49728 + ([Intel 9.7 FEAT] Virt-QEMU: TDX: Allow to configure apic bus clock) -* Thu Apr 03 2025 Jon Maloy - 9.1.0-15.el9_6.3 -- kvm-qga-implement-a-guest-get-load-command.patch [RHEL-83000] -- Resolves: RHEL-83000 - ([qemu-guest-agent][RFE] Report CPU load average [rhel-9.6.z]) +* Wed Aug 20 2025 Jon Maloy - 9.1.0-26 +- kvm-rbd-Fix-.bdrv_get_specific_info-implementation.patch [RHEL-108726] +- Resolves: RHEL-108726 + (Openstack guest becomes inaccessible via network when storage network on the hypervisor is disabled/lost [rhel-9]) -* Thu Apr 03 2025 Jon Maloy - 9.1.0-15.el9_6.2 -- kvm-net-vhost-user-add-QAPI-events-to-report-connection-.patch [RHEL-80622] -- Resolves: RHEL-80622 - (Allow libvirt to restart passt/vhost-user when the process is killed [rhel-9]) +* Tue Jul 08 2025 Miroslav Rezanina - 9.1.0-25 +- kvm-s390x-Fix-leak-in-machine_set_loadparm.patch [RHEL-98554] +- kvm-hw-s390x-ccw-device-Fix-memory-leak-in-loadparm-sett.patch [RHEL-98554] +- kvm-amd_iommu-Rename-variable-mmio-to-mr_mmio.patch [RHEL-66202] +- kvm-amd_iommu-Add-support-for-pass-though-mode.patch [RHEL-66202] +- kvm-amd_iommu-Use-shared-memory-region-for-Interrupt-Rem.patch [RHEL-66202] +- kvm-amd_iommu-Send-notification-when-invalidate-interrup.patch [RHEL-66202] +- kvm-amd_iommu-Check-APIC-ID-255-for-XTSup.patch [RHEL-66202] +- kvm-io-Fix-partial-struct-copy-in-qio_dns_resolver_looku.patch [RHEL-67104] +- kvm-util-qemu-sockets-Refactor-setting-client-sockopts-i.patch [RHEL-67104] +- kvm-util-qemu-sockets-Refactor-success-and-failure-paths.patch [RHEL-67104] +- kvm-util-qemu-sockets-Add-support-for-keep-alive-flag-to.patch [RHEL-67104] +- kvm-util-qemu-sockets-Refactor-inet_parse-to-use-QemuOpt.patch [RHEL-67104] +- kvm-util-qemu-sockets-Introduce-inet-socket-options-cont.patch [RHEL-67104] +- kvm-tests-unit-test-util-sockets-fix-mem-leak-on-error-o.patch [RHEL-67104] +- kvm-target-i386-Expose-bits-related-to-SRSO-vulnerabilit.patch [RHEL-52649] +- kvm-target-i386-Add-PerfMonV2-feature-bit.patch [RHEL-52649] +- kvm-target-i386-Update-EPYC-CPU-model-for-Cache-property.patch [RHEL-52649] +- kvm-target-i386-Update-EPYC-Rome-CPU-model-for-Cache-pro.patch [RHEL-52649] +- kvm-target-i386-Update-EPYC-Milan-CPU-model-for-Cache-pr.patch [RHEL-52649] +- kvm-target-i386-Add-couple-of-feature-bits-in-CPUID_Fn80.patch [RHEL-52649] +- kvm-target-i386-Update-EPYC-Genoa-for-Cache-property-per.patch [RHEL-52649] +- kvm-target-i386-Add-support-for-EPYC-Turin-model.patch [RHEL-52649] +- kvm-hw-i386-amd_iommu-Assign-pci-id-0x1419-for-the-AMD-I.patch [RHEL-70926] +- kvm-hw-i386-amd_iommu-Isolate-AMDVI-PCI-from-amd-iommu-d.patch [RHEL-70925] +- kvm-hw-i386-amd_iommu-Allow-migration-when-explicitly-cr.patch [RHEL-70925] +- kvm-Enable-amd-iommu-device.patch [RHEL-70925] +- kvm-include-qemu-compiler-add-QEMU_UNINITIALIZED-attribu.patch [RHEL-99888] +- kvm-hw-virtio-virtio-avoid-cost-of-ftrivial-auto-var-ini.patch [RHEL-99888] +- kvm-block-skip-automatic-zero-init-of-large-array-in-ioq.patch [RHEL-99888] +- kvm-chardev-char-fd-skip-automatic-zero-init-of-large-ar.patch [RHEL-99888] +- kvm-chardev-char-pty-skip-automatic-zero-init-of-large-a.patch [RHEL-99888] +- kvm-chardev-char-socket-skip-automatic-zero-init-of-larg.patch [RHEL-99888] +- kvm-hw-audio-ac97-skip-automatic-zero-init-of-large-arra.patch [RHEL-99888] +- kvm-hw-audio-cs4231a-skip-automatic-zero-init-of-large-a.patch [RHEL-99888] +- kvm-hw-audio-es1370-skip-automatic-zero-init-of-large-ar.patch [RHEL-99888] +- kvm-hw-audio-gus-skip-automatic-zero-init-of-large-array.patch [RHEL-99888] +- kvm-hw-audio-marvell_88w8618-skip-automatic-zero-init-of.patch [RHEL-99888] +- kvm-hw-audio-sb16-skip-automatic-zero-init-of-large-arra.patch [RHEL-99888] +- kvm-hw-audio-via-ac97-skip-automatic-zero-init-of-large-.patch [RHEL-99888] +- kvm-hw-char-sclpconsole-lm-skip-automatic-zero-init-of-l.patch [RHEL-99888] +- kvm-hw-dma-xlnx_csu_dma-skip-automatic-zero-init-of-larg.patch [RHEL-99888] +- kvm-hw-display-vmware_vga-skip-automatic-zero-init-of-la.patch [RHEL-99888] +- kvm-hw-hyperv-syndbg-skip-automatic-zero-init-of-large-a.patch [RHEL-99888] +- kvm-hw-misc-aspeed_hace-skip-automatic-zero-init-of-larg.patch [RHEL-99888] +- kvm-hw-net-rtl8139-skip-automatic-zero-init-of-large-arr.patch [RHEL-99888] +- kvm-hw-net-tulip-skip-automatic-zero-init-of-large-array.patch [RHEL-99888] +- kvm-hw-net-virtio-net-skip-automatic-zero-init-of-large-.patch [RHEL-99888] +- kvm-hw-net-xgamc-skip-automatic-zero-init-of-large-array.patch [RHEL-99888] +- kvm-hw-nvme-ctrl-skip-automatic-zero-init-of-large-array.patch [RHEL-99888] +- kvm-hw-ppc-spapr_tpm_proxy-skip-automatic-zero-init-of-l.patch [RHEL-99888] +- kvm-hw-usb-hcd-ohci-skip-automatic-zero-init-of-large-ar.patch [RHEL-99888] +- kvm-hw-scsi-lsi53c895a-skip-automatic-zero-init-of-large.patch [RHEL-99888] +- kvm-hw-scsi-megasas-skip-automatic-zero-init-of-large-ar.patch [RHEL-99888] +- kvm-hw-ufs-lu-skip-automatic-zero-init-of-large-array.patch [RHEL-99888] +- kvm-net-socket-skip-automatic-zero-init-of-large-array.patch [RHEL-99888] +- kvm-net-stream-skip-automatic-zero-init-of-large-array.patch [RHEL-99888] +- kvm-ui-vnc-Update-display-update-interval-when-VM-state-.patch [RHEL-100741] +- Resolves: RHEL-98554 + ([s390x][RHEL9.7.0][virtio_block] there would be memory leak with virtio_blk disks) +- Resolves: RHEL-66202 + ([AMDSERVER 9.6 Feature] qemu: Interrupt Remap support for emulated amd viommu) +- Resolves: RHEL-67104 + (postcopy on the destination host can't switch into pause status under the network issue if boot VM with '-S') +- Resolves: RHEL-52649 + ([AMDSERVER 9.6 Feature] Turin: Qemu EPYC-Turin Model) +- Resolves: RHEL-70926 + (Qemu/amd-iommu: Advertise a suitable device id) +- Resolves: RHEL-70925 + (Qemu/amd-iommu: Add ability to manually specify the AMDVI-PCI device) +- Resolves: RHEL-99888 + (-ftrivial-auto-var-init=zero reduced performance [rhel-9]) +- Resolves: RHEL-100741 + (Video stuck after switchover phase when play one video during migration [rhel-9]) + +* Mon Jun 16 2025 Jon Maloy - 9.1.0-24 +- kvm-s390x-pci-add-support-for-guests-that-request-direct.patch [RHEL-11430] +- kvm-s390x-pci-indicate-QEMU-supports-relaxed-translation.patch [RHEL-11430] +- kvm-block-Expand-block-status-mode-from-bool-to-flags.patch [RHEL-82906 RHEL-83015] +- kvm-file-posix-gluster-Handle-zero-block-status-hint-bet.patch [RHEL-82906 RHEL-83015] +- kvm-block-Let-bdrv_co_is_zero_fast-consolidate-adjacent-.patch [RHEL-82906 RHEL-83015] +- kvm-block-Add-new-bdrv_co_is_all_zeroes-function.patch [RHEL-82906 RHEL-83015] +- kvm-iotests-Improve-iotest-194-to-mirror-data.patch [RHEL-82906 RHEL-83015] +- kvm-mirror-Minor-refactoring.patch [RHEL-82906 RHEL-83015] +- kvm-mirror-Pass-full-sync-mode-rather-than-bool-to-inter.patch [RHEL-82906 RHEL-83015] +- kvm-mirror-Allow-QMP-override-to-declare-target-already-.patch [RHEL-82906 RHEL-83015] +- kvm-mirror-Drop-redundant-zero_target-parameter.patch [RHEL-82906 RHEL-83015] +- kvm-mirror-Skip-pre-zeroing-destination-if-it-is-already.patch [RHEL-82906 RHEL-83015] +- kvm-mirror-Skip-writing-zeroes-when-target-is-already-ze.patch [RHEL-82906 RHEL-83015] +- kvm-iotests-common.rc-add-disk_usage-function.patch [RHEL-82906 RHEL-83015] +- kvm-tests-Add-iotest-mirror-sparse-for-recent-patches.patch [RHEL-82906 RHEL-83015] +- kvm-mirror-Reduce-I-O-when-destination-is-detect-zeroes-.patch [RHEL-82906 RHEL-83015] +- Resolves: RHEL-11430 + ([IBM 9.7 FEAT] KVM: Performance Enhanced Refresh PCI Translation - qemu part) +- Resolves: RHEL-82906 + (--migrate-disks-detect-zeroes doesn't take effect for disk migration [rhel-9.7]) +- Resolves: RHEL-83015 + (Disk size of target raw image is full allocated when doing mirror with default discard value [rhel-9.7]) + +* Mon Jun 09 2025 Jon Maloy - 9.1.0-23 +- kvm-net-vhost-user-add-QAPI-events-to-report-connection-.patch [RHEL-95120] +- kvm-file-posix-Define-DM_MPATH_PROBE_PATHS.patch [RHEL-95408] +- kvm-file-posix-Probe-paths-and-retry-SG_IO-on-potential-.patch [RHEL-95408] +- Resolves: RHEL-95120 + (Allow libvirt to restart passt/vhost-user when the process is killed [rhel-9.7]) +- Resolves: RHEL-95408 + (Support multipath failover with scsi-block [rhel-9]) + +* Mon Jun 02 2025 Jon Maloy - 9.1.0-22 +- kvm-migration-postcopy-Spatial-locality-page-hint-for-pr.patch [RHEL-85159] +- kvm-Allow-guest-network-get-route-guest-get-load-QGA-com.patch [RHEL-91605 RHEL-91606] +- Resolves: RHEL-85159 + (Video stuck about 1 min after switchover phase when play one video during postcopy-preempt migration) +- Resolves: RHEL-91605 + ([qemu-guest-agent] Add new api 'guest-network-get-route' to allow-rpc [RHEL-9]) +- Resolves: RHEL-91606 + ([qemu-guest-agent] Enable 'guest-get-load' by default [RHEL-9]) + +* Mon May 26 2025 Jon Maloy - 9.1.0-21 +- kvm-meson-configure-add-valgrind-option-en-dis-able-valg.patch [RHEL-88153] +- kvm-distro-add-an-explicit-valgrind-devel-build-dep.patch [RHEL-88153] +- kvm-hw-i386-Fix-machine-type-compatibility.patch [RHEL-91307] +- kvm-vfio-helpers-Refactor-vfio_region_mmap-error-handlin.patch [RHEL-88533] +- kvm-vfio-helpers-Align-mmaps.patch [RHEL-88533] +- Resolves: RHEL-88153 + ([s390x] valgrind not working with qemu-kvm for non-x86 builds) +- Resolves: RHEL-91307 + (Fix x86 M-type compats) +- Resolves: RHEL-88533 + (Improve VFIO mmapping performance with huge pfnmaps) + +* Tue May 13 2025 Jon Maloy - 9.1.0-20 +- kvm-virtio-net-disable-USO-for-virt-rhel9.6.patch [RHEL-80313] +- kvm-arm-Use-arm_virt_compat_set-to-apply-the-compat.patch [RHEL-80313] +- kvm-file-posix-probe-discard-alignment-on-Linux-block-de.patch [RHEL-86032] +- kvm-block-io-skip-head-tail-requests-on-EINVAL.patch [RHEL-86032] +- kvm-file-posix-Fix-crash-on-discard_granularity-0.patch [RHEL-86032] +- Resolves: RHEL-80313 + (Unable to migrate VM from RHEL10.0/qemu-kvm-9.6 to RHEL9.6/qemu-kvm-9.6) +- Resolves: RHEL-86032 + (QEMU sends unaligned discards on 4K devices [RHEL-9.7]) + +* Mon Apr 28 2025 Jon Maloy - 9.1.0-19 +- kvm-target-i386-Fix-conditional-CONFIG_SYNDBG-enablement.patch [RHEL-7130] +- kvm-target-i386-Exclude-hv-syndbg-from-hv-passthrough.patch [RHEL-7130] +- Resolves: RHEL-7130 + ([Hyper-V][RHEL9.2] Nested Hyper-V on KVM: L1 Windows VM with BIOS mode fails to boot up when using '-cpu host,hv_passthrough’ flag) + +* Mon Apr 14 2025 Jon Maloy - 9.1.0-18 +- kvm-virtio-kconfig-memory-devices-are-PCI-only.patch [RHEL-72977] +- kvm-hw-s390-ccw-device-Convert-to-three-phase-reset.patch [RHEL-72977] +- kvm-hw-s390-virtio-ccw-Convert-to-three-phase-reset.patch [RHEL-72977] +- kvm-target-s390-Convert-CPU-to-Resettable-interface.patch [RHEL-72977] +- kvm-reset-Use-ResetType-for-qemu_devices_reset-and-Machi.patch [RHEL-72977] +- kvm-reset-Add-RESET_TYPE_WAKEUP.patch [RHEL-72977] +- kvm-virtio-mem-Use-new-Resettable-framework-instead-of-L.patch [RHEL-72977] +- kvm-virtio-mem-Add-support-for-suspend-wake-up-with-plug.patch [RHEL-72977] +- kvm-virtio-mem-unplug-memory-only-during-system-resets-n.patch [RHEL-72977] +- kvm-s390x-s390-virtio-ccw-don-t-crash-on-weird-RAM-sizes.patch [RHEL-72977] +- kvm-s390x-s390-virtio-hcall-remove-hypercall-registratio.patch [RHEL-72977] +- kvm-s390x-s390-virtio-hcall-prepare-for-more-diag500-hyp.patch [RHEL-72977] +- kvm-s390x-rename-s390-virtio-hcall-to-s390-hypercall.patch [RHEL-72977] +- kvm-s390x-s390-virtio-ccw-move-setting-the-maximum-guest.patch [RHEL-72977] +- kvm-s390x-introduce-s390_get_memory_limit.patch [RHEL-72977] +- kvm-s390x-s390-hypercall-introduce-DIAG500-STORAGE_LIMIT.patch [RHEL-72977] +- kvm-s390x-s390-stattrib-kvm-prepare-for-memory-devices-a.patch [RHEL-72977] +- kvm-s390x-s390-skeys-prepare-for-memory-devices.patch [RHEL-72977] +- kvm-s390x-s390-virtio-ccw-prepare-for-memory-devices.patch [RHEL-72977] +- kvm-s390x-pv-prepare-for-memory-devices.patch [RHEL-72977] +- kvm-s390x-remember-the-maximum-page-size.patch [RHEL-72977] +- kvm-s390x-virtio-ccw-add-support-for-virtio-based-memory.patch [RHEL-72977] +- kvm-s390x-virtio-mem-support.patch [RHEL-72977] +- kvm-hw-virtio-Also-include-md-stubs-in-case-CONFIG_VIRTI.patch [RHEL-72977] +- kvm-virtio-mem-don-t-warn-about-THP-sizes-on-a-kernel-wi.patch [RHEL-72977] +- kvm-redhat-Enable-virtio-mem-on-s390x.patch [RHEL-72977] +- Resolves: RHEL-72977 + ([IBM 9.7 FEAT] KVM: Enable virtio-mem support - qemu part) + +* Mon Mar 31 2025 Jon Maloy - 9.1.0-17 +- kvm-hw-pci-Rename-has_power-to-enabled.patch [RHEL-7301] +- kvm-hw-pci-Basic-support-for-PCI-power-management.patch [RHEL-7301] +- kvm-pci-Use-PCI-PM-capability-initializer.patch [RHEL-7301] +- kvm-vfio-pci-Delete-local-pm_cap.patch [RHEL-7301] +- kvm-pcie-virtio-Remove-redundant-pm_cap.patch [RHEL-7301] +- kvm-hw-vfio-pci-Re-order-pre-reset.patch [RHEL-7301] +- kvm-Also-recommend-systemtap-devel-from-qemu-tools.patch [RHEL-47340] +- Resolves: RHEL-7301 + ([intel iommu] VFIO_MAP_DMA failed: Bad address on system_powerdown) +- Resolves: RHEL-47340 + ([Qemu RHEL-9] qemu-trace-stap should handle lack of stap more gracefully) + +* Thu Mar 20 2025 Jon Maloy - 9.1.0-16 +- kvm-hw-virtio-virtio-iommu-Migrate-to-3-phase-reset.patch [RHEL-7188] +- kvm-hw-i386-intel-iommu-Migrate-to-3-phase-reset.patch [RHEL-7188] +- kvm-hw-arm-smmuv3-Move-reset-to-exit-phase.patch [RHEL-7188] +- kvm-hw-vfio-common-Add-a-trace-point-in-vfio_reset_handl.patch [RHEL-7188] +- kvm-docs-devel-reset-Document-reset-expectations-for-DMA.patch [RHEL-7188] +- kvm-qga-implement-a-guest-get-load-command.patch [RHEL-69622] +- kvm-migration-Fix-UAF-for-incoming-migration-on-Migratio.patch [RHEL-69775] +- kvm-scripts-improve-error-from-qemu-trace-stap-on-missin.patch [RHEL-47340] +- kvm-Recommend-systemtap-client-from-qemu-tools.patch [RHEL-47340] +- Resolves: RHEL-7188 + ([intel iommu][PF] DMAR: DRHD: handling fault status reg) +- Resolves: RHEL-69622 + ([qemu-guest-agent][RFE] Report CPU load average) +- Resolves: RHEL-69775 + (Guest crashed on the target host when the migration was canceled) +- Resolves: RHEL-47340 + ([Qemu RHEL-9] qemu-trace-stap should handle lack of stap more gracefully) * Mon Feb 17 2025 Jon Maloy - 9.1.0-15 - kvm-net-Fix-announce_self.patch [RHEL-73891]